Vendor dependencies

This commit is contained in:
2026-08-01 16:11:49 +03:00
parent 7f139a0241
commit 6b5e7f0f8b
29706 changed files with 9575646 additions and 0 deletions
File diff suppressed because one or more lines are too long
+6
View File
@@ -0,0 +1,6 @@
{
"git": {
"sha1": "63c6578f22bb8eda35a983e5c937d79c16b2a355"
},
"path_in_vcs": ""
}
+8
View File
@@ -0,0 +1,8 @@
This project is licensed under either of
* Apache License, Version 2.0, ([LICENSE-APACHE](LICENSE-APACHE) or
https://www.apache.org/licenses/LICENSE-2.0)
* MIT license ([LICENSE-MIT](LICENSE-MIT) or
https://opensource.org/licenses/MIT)
at your option.
+159
View File
@@ -0,0 +1,159 @@
# This file is automatically @generated by Cargo.
# It is not intended for manual editing.
version = 3
[[package]]
name = "bstr"
version = "1.13.0"
dependencies = [
"memchr",
"quickcheck",
"regex-automata",
"serde_core",
"ucd-parse",
"unicode-segmentation",
]
[[package]]
name = "cfg-if"
version = "1.0.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9330f8b2ff13f34540b44e946ef35111825727b38d33286ef986142615121801"
[[package]]
name = "getrandom"
version = "0.2.17"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ff2abc00be7fca6ebc474524697ae276ad847ad0a6b3faa4bcb027e9a4614ad0"
dependencies = [
"cfg-if",
"libc",
"wasi",
]
[[package]]
name = "libc"
version = "0.2.181"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "459427e2af2b9c839b132acb702a1c654d95e10f8c326bfc2ad11310e458b1c5"
[[package]]
name = "memchr"
version = "2.8.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f8ca58f447f06ed17d5fc4043ce1b10dd205e060fb3ce5b979b8ed8e59ff3f79"
[[package]]
name = "proc-macro2"
version = "1.0.106"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8fd00f0bb2e90d81d1044c2b32617f68fcb9fa3bb7640c23e9c748e53fb30934"
dependencies = [
"unicode-ident",
]
[[package]]
name = "quickcheck"
version = "1.0.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "588f6378e4dd99458b60ec275b4477add41ce4fa9f64dcba6f15adccb19b50d6"
dependencies = [
"rand",
]
[[package]]
name = "quote"
version = "1.0.44"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "21b2ebcf727b7760c461f091f9f0f539b77b8e87f2fd88131e7f1b433b3cece4"
dependencies = [
"proc-macro2",
]
[[package]]
name = "rand"
version = "0.8.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "34af8d1a0e25924bc5b7c43c079c942339d8f0a8b57c39049bef581b46327404"
dependencies = [
"rand_core",
]
[[package]]
name = "rand_core"
version = "0.6.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ec0be4795e2f6a28069bec0b5ff3e2ac9bafc99e6a9a7dc3547996c5c816922c"
dependencies = [
"getrandom",
]
[[package]]
name = "regex-automata"
version = "0.4.14"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "6e1dd4122fc1595e8162618945476892eefca7b88c52820e74af6262213cae8f"
[[package]]
name = "regex-lite"
version = "0.1.9"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "cab834c73d247e67f4fae452806d17d3c7501756d98c8808d7c9c7aa7d18f973"
[[package]]
name = "serde_core"
version = "1.0.228"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "41d385c7d4ca58e59fc732af25c3983b67ac852c1a25000afe1175de458b67ad"
dependencies = [
"serde_derive",
]
[[package]]
name = "serde_derive"
version = "1.0.228"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d540f220d3187173da220f885ab66608367b6574e925011a9353e4badda91d79"
dependencies = [
"proc-macro2",
"quote",
"syn",
]
[[package]]
name = "syn"
version = "2.0.114"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d4d107df263a3013ef9b1879b0df87d706ff80f65a86ea879bd9c31f9b307c2a"
dependencies = [
"proc-macro2",
"quote",
"unicode-ident",
]
[[package]]
name = "ucd-parse"
version = "0.1.13"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c06ff81122fcbf4df4c1660b15f7e3336058e7aec14437c9f85c6b31a0f279b9"
dependencies = [
"regex-lite",
]
[[package]]
name = "unicode-ident"
version = "1.0.23"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "537dd038a89878be9b64dd4bd1b260315c1bb94f4d784956b81e27a088d9a09e"
[[package]]
name = "unicode-segmentation"
version = "1.12.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f6ccf251212114b54433ec949fd6a7841275f9ada20dddd2f29e9ceea4501493"
[[package]]
name = "wasi"
version = "0.11.1+wasi-snapshot-preview1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ccf3ec651a847eb01de73ccad15eb7d99f80485de043efb2f370cd654f4ea44b"
+148
View File
@@ -0,0 +1,148 @@
# THIS FILE IS AUTOMATICALLY GENERATED BY CARGO
#
# When uploading crates to the registry Cargo will automatically
# "normalize" Cargo.toml files for maximal compatibility
# with all versions of Cargo and also rewrite `path` dependencies
# to registry (e.g., crates.io) dependencies.
#
# If you are reading this file be aware that the original Cargo.toml
# will likely look very different (and much more reasonable).
# See Cargo.toml.orig for the original contents.
[package]
edition = "2021"
rust-version = "1.65"
name = "bstr"
version = "1.13.0"
authors = ["Andrew Gallant <jamslam@gmail.com>"]
build = false
exclude = [
"/.github",
"/scripts",
"/src/unicode/data",
]
autolib = false
autobins = false
autoexamples = false
autotests = false
autobenches = false
description = "A string type that is not required to be valid UTF-8."
homepage = "https://github.com/BurntSushi/bstr"
documentation = "https://docs.rs/bstr"
readme = "README.md"
keywords = [
"string",
"str",
"byte",
"bytes",
"text",
]
categories = [
"text-processing",
"encoding",
]
license = "MIT OR Apache-2.0"
repository = "https://github.com/BurntSushi/bstr"
resolver = "2"
[package.metadata.docs.rs]
all-features = true
rustdoc-args = [
"--cfg",
"docsrs",
]
[features]
alloc = [
"memchr/alloc",
"serde_core?/alloc",
]
default = [
"std",
"unicode",
]
serde = ["dep:serde_core"]
std = [
"alloc",
"memchr/std",
"serde_core?/std",
]
unicode = ["dep:regex-automata"]
[lib]
name = "bstr"
path = "src/lib.rs"
bench = false
[[example]]
name = "graphemes"
path = "examples/graphemes.rs"
required-features = [
"std",
"unicode",
]
[[example]]
name = "graphemes-std"
path = "examples/graphemes-std.rs"
[[example]]
name = "lines"
path = "examples/lines.rs"
required-features = ["std"]
[[example]]
name = "lines-std"
path = "examples/lines-std.rs"
[[example]]
name = "uppercase"
path = "examples/uppercase.rs"
required-features = [
"std",
"unicode",
]
[[example]]
name = "uppercase-std"
path = "examples/uppercase-std.rs"
[[example]]
name = "words"
path = "examples/words.rs"
required-features = [
"std",
"unicode",
]
[[example]]
name = "words-std"
path = "examples/words-std.rs"
[dependencies.memchr]
version = "2.7.1"
default-features = false
[dependencies.regex-automata]
version = "0.4.1"
features = ["dfa-search"]
optional = true
default-features = false
[dependencies.serde_core]
version = "1.0.85"
optional = true
default-features = false
[dev-dependencies.quickcheck]
version = "1"
default-features = false
[dev-dependencies.ucd-parse]
version = "0.1.3"
[dev-dependencies.unicode-segmentation]
version = "1.2.1"
[profile.release]
debug = 2
+76
View File
@@ -0,0 +1,76 @@
[package]
name = "bstr"
version = "1.13.0" #:version
authors = ["Andrew Gallant <jamslam@gmail.com>"]
description = "A string type that is not required to be valid UTF-8."
documentation = "https://docs.rs/bstr"
homepage = "https://github.com/BurntSushi/bstr"
repository = "https://github.com/BurntSushi/bstr"
readme = "README.md"
keywords = ["string", "str", "byte", "bytes", "text"]
license = "MIT OR Apache-2.0"
categories = ["text-processing", "encoding"]
exclude = ["/.github", "/scripts", "/src/unicode/data"]
edition = "2021"
rust-version = "1.65"
resolver = "2"
[workspace]
members = ["bench"]
[lib]
bench = false
[features]
default = ["std", "unicode"]
std = ["alloc", "memchr/std", "serde_core?/std"]
alloc = ["memchr/alloc", "serde_core?/alloc"]
unicode = ["dep:regex-automata"]
serde = ["dep:serde_core"]
[dependencies]
memchr = { version = "2.7.1", default-features = false }
serde_core = { version = "1.0.85", default-features = false, optional = true }
[dependencies.regex-automata]
version = "0.4.1"
default-features = false
features = ["dfa-search"]
optional = true
[dev-dependencies]
quickcheck = { version = "1", default-features = false }
ucd-parse = "0.1.3"
unicode-segmentation = "1.2.1"
[package.metadata.docs.rs]
# We want to document all features.
all-features = true
# Since this crate's feature setup is pretty complicated, it is worth opting
# into a nightly unstable option to show the features that need to be enabled
# for public API items. To do that, we set 'docsrs', and when that's enabled,
# we enable the 'doc_cfg' feature.
#
# To test this locally, run:
#
# RUSTDOCFLAGS="--cfg docsrs" cargo +nightly doc --all-features
rustdoc-args = ["--cfg", "docsrs"]
[profile.release]
debug = true
[[example]]
name = "graphemes"
required-features = ["std", "unicode"]
[[example]]
name = "lines"
required-features = ["std"]
[[example]]
name = "uppercase"
required-features = ["std", "unicode"]
[[example]]
name = "words"
required-features = ["std", "unicode"]
+201
View File
@@ -0,0 +1,201 @@
Apache License
Version 2.0, January 2004
http://www.apache.org/licenses/
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
1. Definitions.
"License" shall mean the terms and conditions for use, reproduction,
and distribution as defined by Sections 1 through 9 of this document.
"Licensor" shall mean the copyright owner or entity authorized by
the copyright owner that is granting the License.
"Legal Entity" shall mean the union of the acting entity and all
other entities that control, are controlled by, or are under common
control with that entity. For the purposes of this definition,
"control" means (i) the power, direct or indirect, to cause the
direction or management of such entity, whether by contract or
otherwise, or (ii) ownership of fifty percent (50%) or more of the
outstanding shares, or (iii) beneficial ownership of such entity.
"You" (or "Your") shall mean an individual or Legal Entity
exercising permissions granted by this License.
"Source" form shall mean the preferred form for making modifications,
including but not limited to software source code, documentation
source, and configuration files.
"Object" form shall mean any form resulting from mechanical
transformation or translation of a Source form, including but
not limited to compiled object code, generated documentation,
and conversions to other media types.
"Work" shall mean the work of authorship, whether in Source or
Object form, made available under the License, as indicated by a
copyright notice that is included in or attached to the work
(an example is provided in the Appendix below).
"Derivative Works" shall mean any work, whether in Source or Object
form, that is based on (or derived from) the Work and for which the
editorial revisions, annotations, elaborations, or other modifications
represent, as a whole, an original work of authorship. For the purposes
of this License, Derivative Works shall not include works that remain
separable from, or merely link (or bind by name) to the interfaces of,
the Work and Derivative Works thereof.
"Contribution" shall mean any work of authorship, including
the original version of the Work and any modifications or additions
to that Work or Derivative Works thereof, that is intentionally
submitted to Licensor for inclusion in the Work by the copyright owner
or by an individual or Legal Entity authorized to submit on behalf of
the copyright owner. For the purposes of this definition, "submitted"
means any form of electronic, verbal, or written communication sent
to the Licensor or its representatives, including but not limited to
communication on electronic mailing lists, source code control systems,
and issue tracking systems that are managed by, or on behalf of, the
Licensor for the purpose of discussing and improving the Work, but
excluding communication that is conspicuously marked or otherwise
designated in writing by the copyright owner as "Not a Contribution."
"Contributor" shall mean Licensor and any individual or Legal Entity
on behalf of whom a Contribution has been received by Licensor and
subsequently incorporated within the Work.
2. Grant of Copyright License. Subject to the terms and conditions of
this License, each Contributor hereby grants to You a perpetual,
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
copyright license to reproduce, prepare Derivative Works of,
publicly display, publicly perform, sublicense, and distribute the
Work and such Derivative Works in Source or Object form.
3. Grant of Patent License. Subject to the terms and conditions of
this License, each Contributor hereby grants to You a perpetual,
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
(except as stated in this section) patent license to make, have made,
use, offer to sell, sell, import, and otherwise transfer the Work,
where such license applies only to those patent claims licensable
by such Contributor that are necessarily infringed by their
Contribution(s) alone or by combination of their Contribution(s)
with the Work to which such Contribution(s) was submitted. If You
institute patent litigation against any entity (including a
cross-claim or counterclaim in a lawsuit) alleging that the Work
or a Contribution incorporated within the Work constitutes direct
or contributory patent infringement, then any patent licenses
granted to You under this License for that Work shall terminate
as of the date such litigation is filed.
4. Redistribution. You may reproduce and distribute copies of the
Work or Derivative Works thereof in any medium, with or without
modifications, and in Source or Object form, provided that You
meet the following conditions:
(a) You must give any other recipients of the Work or
Derivative Works a copy of this License; and
(b) You must cause any modified files to carry prominent notices
stating that You changed the files; and
(c) You must retain, in the Source form of any Derivative Works
that You distribute, all copyright, patent, trademark, and
attribution notices from the Source form of the Work,
excluding those notices that do not pertain to any part of
the Derivative Works; and
(d) If the Work includes a "NOTICE" text file as part of its
distribution, then any Derivative Works that You distribute must
include a readable copy of the attribution notices contained
within such NOTICE file, excluding those notices that do not
pertain to any part of the Derivative Works, in at least one
of the following places: within a NOTICE text file distributed
as part of the Derivative Works; within the Source form or
documentation, if provided along with the Derivative Works; or,
within a display generated by the Derivative Works, if and
wherever such third-party notices normally appear. The contents
of the NOTICE file are for informational purposes only and
do not modify the License. You may add Your own attribution
notices within Derivative Works that You distribute, alongside
or as an addendum to the NOTICE text from the Work, provided
that such additional attribution notices cannot be construed
as modifying the License.
You may add Your own copyright statement to Your modifications and
may provide additional or different license terms and conditions
for use, reproduction, or distribution of Your modifications, or
for any such Derivative Works as a whole, provided Your use,
reproduction, and distribution of the Work otherwise complies with
the conditions stated in this License.
5. Submission of Contributions. Unless You explicitly state otherwise,
any Contribution intentionally submitted for inclusion in the Work
by You to the Licensor shall be under the terms and conditions of
this License, without any additional terms or conditions.
Notwithstanding the above, nothing herein shall supersede or modify
the terms of any separate license agreement you may have executed
with Licensor regarding such Contributions.
6. Trademarks. This License does not grant permission to use the trade
names, trademarks, service marks, or product names of the Licensor,
except as required for reasonable and customary use in describing the
origin of the Work and reproducing the content of the NOTICE file.
7. Disclaimer of Warranty. Unless required by applicable law or
agreed to in writing, Licensor provides the Work (and each
Contributor provides its Contributions) on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
implied, including, without limitation, any warranties or conditions
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
PARTICULAR PURPOSE. You are solely responsible for determining the
appropriateness of using or redistributing the Work and assume any
risks associated with Your exercise of permissions under this License.
8. Limitation of Liability. In no event and under no legal theory,
whether in tort (including negligence), contract, or otherwise,
unless required by applicable law (such as deliberate and grossly
negligent acts) or agreed to in writing, shall any Contributor be
liable to You for damages, including any direct, indirect, special,
incidental, or consequential damages of any character arising as a
result of this License or out of the use or inability to use the
Work (including but not limited to damages for loss of goodwill,
work stoppage, computer failure or malfunction, or any and all
other commercial damages or losses), even if such Contributor
has been advised of the possibility of such damages.
9. Accepting Warranty or Additional Liability. While redistributing
the Work or Derivative Works thereof, You may choose to offer,
and charge a fee for, acceptance of support, warranty, indemnity,
or other liability obligations and/or rights consistent with this
License. However, in accepting such obligations, You may act only
on Your own behalf and on Your sole responsibility, not on behalf
of any other Contributor, and only if You agree to indemnify,
defend, and hold each Contributor harmless for any liability
incurred by, or claims asserted against, such Contributor by reason
of your accepting any such warranty or additional liability.
END OF TERMS AND CONDITIONS
APPENDIX: How to apply the Apache License to your work.
To apply the Apache License to your work, attach the following
boilerplate notice, with the fields enclosed by brackets "[]"
replaced with your own identifying information. (Don't include
the brackets!) The text should be enclosed in the appropriate
comment syntax for the file format. We also recommend that a
file or class name and description of purpose be included on the
same "printed page" as the copyright notice for easier
identification within third-party archives.
Copyright [yyyy] [name of copyright owner]
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
+21
View File
@@ -0,0 +1,21 @@
The MIT License (MIT)
Copyright (c) 2018-2019 Andrew Gallant
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
+239
View File
@@ -0,0 +1,239 @@
bstr
====
This crate provides extension traits for `&[u8]` and `Vec<u8>` that enable
their use as byte strings, where byte strings are _conventionally_ UTF-8. This
differs from the standard library's `String` and `str` types in that they are
not required to be valid UTF-8, but may be fully or partially valid UTF-8.
[![Build status](https://github.com/BurntSushi/bstr/workflows/ci/badge.svg)](https://github.com/BurntSushi/bstr/actions)
[![crates.io](https://img.shields.io/crates/v/bstr.svg)](https://crates.io/crates/bstr)
### Documentation
https://docs.rs/bstr
### When should I use byte strings?
See this part of the documentation for more details:
<https://docs.rs/bstr/1.*/bstr/#when-should-i-use-byte-strings>.
The short story is that byte strings are useful when it is inconvenient or
incorrect to require valid UTF-8.
### Usage
`cargo add bstr`
### Examples
The following two examples exhibit both the API features of byte strings and
the I/O convenience functions provided for reading line-by-line quickly.
This first example simply shows how to efficiently iterate over lines in stdin,
and print out lines containing a particular substring:
```rust
use std::{error::Error, io::{self, Write}};
use bstr::{ByteSlice, io::BufReadExt};
fn main() -> Result<(), Box<dyn Error>> {
let stdin = io::stdin();
let mut stdout = io::BufWriter::new(io::stdout());
stdin.lock().for_byte_line_with_terminator(|line| {
if line.contains_str("Dimension") {
stdout.write_all(line)?;
}
Ok(true)
})?;
Ok(())
}
```
This example shows how to count all of the words (Unicode-aware) in stdin,
line-by-line:
```rust
use std::{error::Error, io};
use bstr::{ByteSlice, io::BufReadExt};
fn main() -> Result<(), Box<dyn Error>> {
let stdin = io::stdin();
let mut words = 0;
stdin.lock().for_byte_line_with_terminator(|line| {
words += line.words().count();
Ok(true)
})?;
println!("{}", words);
Ok(())
}
```
This example shows how to convert a stream on stdin to uppercase without
performing UTF-8 validation _and_ amortizing allocation. On standard ASCII
text, this is quite a bit faster than what you can (easily) do with standard
library APIs. (N.B. Any invalid UTF-8 bytes are passed through unchanged.)
```rust
use std::{error::Error, io::{self, Write}};
use bstr::{ByteSlice, io::BufReadExt};
fn main() -> Result<(), Box<dyn Error>> {
let stdin = io::stdin();
let mut stdout = io::BufWriter::new(io::stdout());
let mut upper = vec![];
stdin.lock().for_byte_line_with_terminator(|line| {
upper.clear();
line.to_uppercase_into(&mut upper);
stdout.write_all(&upper)?;
Ok(true)
})?;
Ok(())
}
```
This example shows how to extract the first 10 visual characters (as grapheme
clusters) from each line, where invalid UTF-8 sequences are generally treated
as a single character and are passed through correctly:
```rust
use std::{error::Error, io::{self, Write}};
use bstr::{ByteSlice, io::BufReadExt};
fn main() -> Result<(), Box<dyn Error>> {
let stdin = io::stdin();
let mut stdout = io::BufWriter::new(io::stdout());
stdin.lock().for_byte_line_with_terminator(|line| {
let end = line
.grapheme_indices()
.map(|(_, end, _)| end)
.take(10)
.last()
.unwrap_or(line.len());
stdout.write_all(line[..end].trim_end())?;
stdout.write_all(b"\n")?;
Ok(true)
})?;
Ok(())
}
```
### Cargo features
This crates comes with a few features that control standard library, serde and
Unicode support.
* `std` - **Enabled** by default. This provides APIs that require the standard
library, such as `Vec<u8>` and `PathBuf`. Enabling this feature also enables
the `alloc` feature.
* `alloc` - **Enabled** by default. This provides APIs that require allocations
via the `alloc` crate, such as `Vec<u8>`.
* `unicode` - **Enabled** by default. This provides APIs that require sizable
Unicode data compiled into the binary. This includes, but is not limited to,
grapheme/word/sentence segmenters. When this is disabled, basic support such
as UTF-8 decoding is still included. Note that currently, enabling this
feature also requires enabling the `std` feature. It is expected that this
limitation will be lifted at some point.
* `serde` - Enables implementations of serde traits for `BStr`, and also
`BString` when `alloc` is enabled.
### Minimum Rust version policy
This crate's minimum supported `rustc` version (MSRV) is `1.65`.
In general, this crate will be conservative with respect to the minimum
supported version of Rust. MSRV may be bumped in minor version releases.
### Future work
Since it is plausible that some of the types in this crate might end up in your
public API (e.g., `BStr` and `BString`), we will commit to being very
conservative with respect to new major version releases. It's difficult to say
precisely how conservative, but unless there is a major issue with the `1.0`
release, I wouldn't expect a `2.0` release to come out any sooner than some
period of years.
A large part of the API surface area was taken from the standard library, so
from an API design perspective, a good portion of this crate should be on solid
ground. The main differences from the standard library are in how the various
substring search routines work. The standard library provides generic
infrastructure for supporting different types of searches with a single method,
where as this library prefers to define new methods for each type of search and
drop the generic infrastructure.
Some _probable_ future considerations for APIs include, but are not limited to:
* Unicode normalization.
* More sophisticated support for dealing with Unicode case, perhaps by
combining the use cases supported by [`caseless`](https://docs.rs/caseless)
and [`unicase`](https://docs.rs/unicase).
Here are some examples that are _probably_ out of scope for this crate:
* Regular expressions.
* Unicode collation.
The exact scope isn't quite clear, but I expect we can iterate on it.
In general, as stated below, this crate brings lots of related APIs together
into a single crate while simultaneously attempting to keep the total number of
dependencies low. Indeed, every dependency of `bstr`, except for `memchr`, is
optional.
### High level motivation
Strictly speaking, the `bstr` crate provides very little that can't already be
achieved with the standard library `Vec<u8>`/`&[u8]` APIs and the ecosystem of
library crates. For example:
* The standard library's
[`Utf8Error`](https://doc.rust-lang.org/std/str/struct.Utf8Error.html) can be
used for incremental lossy decoding of `&[u8]`.
* The
[`unicode-segmentation`](https://unicode-rs.github.io/unicode-segmentation/unicode_segmentation/index.html)
crate can be used for iterating over graphemes (or words), but is only
implemented for `&str` types. One could use `Utf8Error` above to implement
grapheme iteration with the same semantics as what `bstr` provides (automatic
Unicode replacement codepoint substitution).
* The [`twoway`](https://docs.rs/twoway) crate can be used for fast substring
searching on `&[u8]`.
So why create `bstr`? Part of the point of the `bstr` crate is to provide a
uniform API of coupled components instead of relying on users to piece together
loosely coupled components from the crate ecosystem. For example, if you wanted
to perform a search and replace in a `Vec<u8>`, then writing the code to do
that with the `twoway` crate is not that difficult, but it's still additional
glue code you have to write. This work adds up depending on what you're doing.
Consider, for example, trimming and splitting, along with their different
variants.
In other words, `bstr` is partially a way of pushing back against the
micro-crate ecosystem that appears to be evolving. Namely, it is a goal of
`bstr` to keep its dependency list lightweight. For example, `serde` is an
optional dependency because there is no feasible alternative. In service of
this philosophy, currently, the only required dependency of `bstr` is `memchr`.
### License
This project is licensed under either of
* Apache License, Version 2.0, ([LICENSE-APACHE](LICENSE-APACHE) or
https://www.apache.org/licenses/LICENSE-2.0)
* MIT license ([LICENSE-MIT](LICENSE-MIT) or
https://opensource.org/licenses/MIT)
at your option.
The data in `src/unicode/data/` is licensed under the Unicode License Agreement
([LICENSE-UNICODE](https://www.unicode.org/copyright.html#License)), although
this data is only used in tests.
+25
View File
@@ -0,0 +1,25 @@
use std::error::Error;
use std::io::{self, BufRead, Write};
use unicode_segmentation::UnicodeSegmentation;
fn main() -> Result<(), Box<dyn Error>> {
let stdin = io::stdin();
let mut stdin = stdin.lock();
let mut stdout = io::BufWriter::new(io::stdout());
let mut line = String::new();
while stdin.read_line(&mut line)? > 0 {
let end = line
.grapheme_indices(true)
.map(|(start, g)| start + g.len())
.take(10)
.last()
.unwrap_or(line.len());
stdout.write_all(line[..end].trim_end().as_bytes())?;
stdout.write_all(b"\n")?;
line.clear();
}
Ok(())
}
+22
View File
@@ -0,0 +1,22 @@
use std::error::Error;
use std::io::{self, Write};
use bstr::{io::BufReadExt, ByteSlice};
fn main() -> Result<(), Box<dyn Error>> {
let stdin = io::stdin();
let mut stdout = io::BufWriter::new(io::stdout());
stdin.lock().for_byte_line_with_terminator(|line| {
let end = line
.grapheme_indices()
.map(|(_, end, _)| end)
.take(10)
.last()
.unwrap_or(line.len());
stdout.write_all(line[..end].trim_end())?;
stdout.write_all(b"\n")?;
Ok(true)
})?;
Ok(())
}
+17
View File
@@ -0,0 +1,17 @@
use std::error::Error;
use std::io::{self, BufRead, Write};
fn main() -> Result<(), Box<dyn Error>> {
let stdin = io::stdin();
let mut stdin = stdin.lock();
let mut stdout = io::BufWriter::new(io::stdout());
let mut line = String::new();
while stdin.read_line(&mut line)? > 0 {
if line.contains("Dimension") {
stdout.write_all(line.as_bytes())?;
}
line.clear();
}
Ok(())
}
+17
View File
@@ -0,0 +1,17 @@
use std::error::Error;
use std::io::{self, Write};
use bstr::{io::BufReadExt, ByteSlice};
fn main() -> Result<(), Box<dyn Error>> {
let stdin = io::stdin();
let mut stdout = io::BufWriter::new(io::stdout());
stdin.lock().for_byte_line_with_terminator(|line| {
if line.contains_str("Dimension") {
stdout.write_all(line)?;
}
Ok(true)
})?;
Ok(())
}
+15
View File
@@ -0,0 +1,15 @@
use std::error::Error;
use std::io::{self, BufRead, Write};
fn main() -> Result<(), Box<dyn Error>> {
let stdin = io::stdin();
let mut stdin = stdin.lock();
let mut stdout = io::BufWriter::new(io::stdout());
let mut line = String::new();
while stdin.read_line(&mut line)? > 0 {
stdout.write_all(line.to_uppercase().as_bytes())?;
line.clear();
}
Ok(())
}
+18
View File
@@ -0,0 +1,18 @@
use std::error::Error;
use std::io::{self, Write};
use bstr::{io::BufReadExt, ByteSlice};
fn main() -> Result<(), Box<dyn Error>> {
let stdin = io::stdin();
let mut stdout = io::BufWriter::new(io::stdout());
let mut upper = vec![];
stdin.lock().for_byte_line_with_terminator(|line| {
upper.clear();
line.to_uppercase_into(&mut upper);
stdout.write_all(&upper)?;
Ok(true)
})?;
Ok(())
}
+18
View File
@@ -0,0 +1,18 @@
use std::error::Error;
use std::io::{self, BufRead};
use unicode_segmentation::UnicodeSegmentation;
fn main() -> Result<(), Box<dyn Error>> {
let stdin = io::stdin();
let mut stdin = stdin.lock();
let mut words = 0;
let mut line = String::new();
while stdin.read_line(&mut line)? > 0 {
words += line.unicode_words().count();
line.clear();
}
println!("{}", words);
Ok(())
}
+15
View File
@@ -0,0 +1,15 @@
use std::error::Error;
use std::io;
use bstr::{io::BufReadExt, ByteSlice};
fn main() -> Result<(), Box<dyn Error>> {
let stdin = io::stdin();
let mut words = 0;
stdin.lock().for_byte_line_with_terminator(|line| {
words += line.words().count();
Ok(true)
})?;
println!("{}", words);
Ok(())
}
+2
View File
@@ -0,0 +1,2 @@
max_width = 79
use_small_heuristics = "max"
+336
View File
@@ -0,0 +1,336 @@
// The following ~400 lines of code exists for exactly one purpose, which is
// to optimize this code:
//
// byte_slice.iter().position(|&b| b > 0x7F).unwrap_or(byte_slice.len())
//
// Yes... Overengineered is a word that comes to mind, but this is effectively
// a very similar problem to memchr, and virtually nobody has been able to
// resist optimizing the crap out of that (except for perhaps the BSD and MUSL
// folks). In particular, this routine makes a very common case (ASCII) very
// fast, which seems worth it. We do stop short of adding AVX variants of the
// code below in order to retain our sanity and also to avoid needing to deal
// with runtime target feature detection. RESIST!
//
// In order to understand the SIMD version below, it would be good to read this
// comment describing how my memchr routine works:
// https://github.com/BurntSushi/rust-memchr/blob/b0a29f267f4a7fad8ffcc8fe8377a06498202883/src/x86/sse2.rs#L19-L106
//
// The primary difference with memchr is that for ASCII, we can do a bit less
// work. In particular, we don't need to detect the presence of a specific
// byte, but rather, whether any byte has its most significant bit set. That
// means we can effectively skip the _mm_cmpeq_epi8 step and jump straight to
// _mm_movemask_epi8.
#[cfg(any(test, miri, not(target_arch = "x86_64")))]
const USIZE_BYTES: usize = core::mem::size_of::<usize>();
#[cfg(any(test, miri, not(target_arch = "x86_64")))]
const ALIGN_MASK: usize = core::mem::align_of::<usize>() - 1;
#[cfg(any(test, miri, not(target_arch = "x86_64")))]
const FALLBACK_LOOP_SIZE: usize = 2 * USIZE_BYTES;
// This is a mask where the most significant bit of each byte in the usize
// is set. We test this bit to determine whether a character is ASCII or not.
// Namely, a single byte is regarded as an ASCII codepoint if and only if it's
// most significant bit is not set.
#[cfg(any(test, miri, not(target_arch = "x86_64")))]
const ASCII_MASK_U64: u64 = 0x8080808080808080;
#[cfg(any(test, miri, not(target_arch = "x86_64")))]
const ASCII_MASK: usize = ASCII_MASK_U64 as usize;
/// Returns the index of the first non ASCII byte in the given slice.
///
/// If slice only contains ASCII bytes, then the length of the slice is
/// returned.
pub fn first_non_ascii_byte(slice: &[u8]) -> usize {
#[cfg(any(miri, not(target_arch = "x86_64")))]
{
first_non_ascii_byte_fallback(slice)
}
#[cfg(all(not(miri), target_arch = "x86_64"))]
{
first_non_ascii_byte_sse2(slice)
}
}
#[cfg(any(test, miri, not(target_arch = "x86_64")))]
fn first_non_ascii_byte_fallback(slice: &[u8]) -> usize {
let start_ptr = slice.as_ptr();
let end_ptr = slice[slice.len()..].as_ptr();
let mut ptr = start_ptr;
unsafe {
if slice.len() < USIZE_BYTES {
return first_non_ascii_byte_slow(start_ptr, end_ptr, ptr);
}
let chunk = read_unaligned_usize(ptr);
let mask = chunk & ASCII_MASK;
if mask != 0 {
return first_non_ascii_byte_mask(mask);
}
ptr = ptr_add(ptr, USIZE_BYTES - (start_ptr as usize & ALIGN_MASK));
debug_assert!(ptr > start_ptr);
debug_assert!(ptr_sub(end_ptr, USIZE_BYTES) >= start_ptr);
if slice.len() >= FALLBACK_LOOP_SIZE {
while ptr <= ptr_sub(end_ptr, FALLBACK_LOOP_SIZE) {
debug_assert_eq!(0, (ptr as usize) % USIZE_BYTES);
let a = *(ptr as *const usize);
let b = *(ptr_add(ptr, USIZE_BYTES) as *const usize);
if (a | b) & ASCII_MASK != 0 {
// What a kludge. We wrap the position finding code into
// a non-inlineable function, which makes the codegen in
// the tight loop above a bit better by avoiding a
// couple extra movs. We pay for it by two additional
// stores, but only in the case of finding a non-ASCII
// byte.
#[inline(never)]
unsafe fn findpos(
start_ptr: *const u8,
ptr: *const u8,
) -> usize {
let a = *(ptr as *const usize);
let b = *(ptr_add(ptr, USIZE_BYTES) as *const usize);
let mut at = sub(ptr, start_ptr);
let maska = a & ASCII_MASK;
if maska != 0 {
return at + first_non_ascii_byte_mask(maska);
}
at += USIZE_BYTES;
let maskb = b & ASCII_MASK;
debug_assert!(maskb != 0);
return at + first_non_ascii_byte_mask(maskb);
}
return findpos(start_ptr, ptr);
}
ptr = ptr_add(ptr, FALLBACK_LOOP_SIZE);
}
}
first_non_ascii_byte_slow(start_ptr, end_ptr, ptr)
}
}
#[cfg(all(not(miri), target_arch = "x86_64"))]
fn first_non_ascii_byte_sse2(slice: &[u8]) -> usize {
use core::arch::x86_64::*;
const VECTOR_SIZE: usize = core::mem::size_of::<__m128i>();
const VECTOR_ALIGN: usize = VECTOR_SIZE - 1;
const VECTOR_LOOP_SIZE: usize = 4 * VECTOR_SIZE;
let start_ptr = slice.as_ptr();
let end_ptr = slice[slice.len()..].as_ptr();
let mut ptr = start_ptr;
unsafe {
if slice.len() < VECTOR_SIZE {
return first_non_ascii_byte_slow(start_ptr, end_ptr, ptr);
}
let chunk = _mm_loadu_si128(ptr as *const __m128i);
let mask = _mm_movemask_epi8(chunk);
if mask != 0 {
return mask.trailing_zeros() as usize;
}
ptr = ptr.add(VECTOR_SIZE - (start_ptr as usize & VECTOR_ALIGN));
debug_assert!(ptr > start_ptr);
debug_assert!(end_ptr.sub(VECTOR_SIZE) >= start_ptr);
if slice.len() >= VECTOR_LOOP_SIZE {
while ptr <= ptr_sub(end_ptr, VECTOR_LOOP_SIZE) {
debug_assert_eq!(0, (ptr as usize) % VECTOR_SIZE);
let a = _mm_load_si128(ptr as *const __m128i);
let b = _mm_load_si128(ptr.add(VECTOR_SIZE) as *const __m128i);
let c =
_mm_load_si128(ptr.add(2 * VECTOR_SIZE) as *const __m128i);
let d =
_mm_load_si128(ptr.add(3 * VECTOR_SIZE) as *const __m128i);
let or1 = _mm_or_si128(a, b);
let or2 = _mm_or_si128(c, d);
let or3 = _mm_or_si128(or1, or2);
if _mm_movemask_epi8(or3) != 0 {
let mut at = sub(ptr, start_ptr);
let mask = _mm_movemask_epi8(a);
if mask != 0 {
return at + mask.trailing_zeros() as usize;
}
at += VECTOR_SIZE;
let mask = _mm_movemask_epi8(b);
if mask != 0 {
return at + mask.trailing_zeros() as usize;
}
at += VECTOR_SIZE;
let mask = _mm_movemask_epi8(c);
if mask != 0 {
return at + mask.trailing_zeros() as usize;
}
at += VECTOR_SIZE;
let mask = _mm_movemask_epi8(d);
debug_assert!(mask != 0);
return at + mask.trailing_zeros() as usize;
}
ptr = ptr_add(ptr, VECTOR_LOOP_SIZE);
}
}
while ptr <= end_ptr.sub(VECTOR_SIZE) {
debug_assert!(sub(end_ptr, ptr) >= VECTOR_SIZE);
let chunk = _mm_loadu_si128(ptr as *const __m128i);
let mask = _mm_movemask_epi8(chunk);
if mask != 0 {
return sub(ptr, start_ptr) + mask.trailing_zeros() as usize;
}
ptr = ptr.add(VECTOR_SIZE);
}
first_non_ascii_byte_slow(start_ptr, end_ptr, ptr)
}
}
#[inline(always)]
unsafe fn first_non_ascii_byte_slow(
start_ptr: *const u8,
end_ptr: *const u8,
mut ptr: *const u8,
) -> usize {
debug_assert!(start_ptr <= ptr);
debug_assert!(ptr <= end_ptr);
while ptr < end_ptr {
if *ptr > 0x7F {
return sub(ptr, start_ptr);
}
ptr = ptr.offset(1);
}
sub(end_ptr, start_ptr)
}
/// Compute the position of the first ASCII byte in the given mask.
///
/// The mask should be computed by `chunk & ASCII_MASK`, where `chunk` is
/// 8 contiguous bytes of the slice being checked where *at least* one of those
/// bytes is not an ASCII byte.
///
/// The position returned is always in the inclusive range [0, 7].
#[cfg(any(test, miri, not(target_arch = "x86_64")))]
fn first_non_ascii_byte_mask(mask: usize) -> usize {
#[cfg(target_endian = "little")]
{
mask.trailing_zeros() as usize / 8
}
#[cfg(target_endian = "big")]
{
mask.leading_zeros() as usize / 8
}
}
/// Increment the given pointer by the given amount.
unsafe fn ptr_add(ptr: *const u8, amt: usize) -> *const u8 {
ptr.add(amt)
}
/// Decrement the given pointer by the given amount.
unsafe fn ptr_sub(ptr: *const u8, amt: usize) -> *const u8 {
ptr.sub(amt)
}
#[cfg(any(test, miri, not(target_arch = "x86_64")))]
unsafe fn read_unaligned_usize(ptr: *const u8) -> usize {
use core::ptr;
let mut n: usize = 0;
ptr::copy_nonoverlapping(ptr, &mut n as *mut _ as *mut u8, USIZE_BYTES);
n
}
/// Subtract `b` from `a` and return the difference. `a` should be greater than
/// or equal to `b`.
fn sub(a: *const u8, b: *const u8) -> usize {
debug_assert!(a >= b);
(a as usize) - (b as usize)
}
#[cfg(test)]
mod tests {
use super::*;
// Our testing approach here is to try and exhaustively test every case.
// This includes the position at which a non-ASCII byte occurs in addition
// to the alignment of the slice that we're searching.
#[test]
fn positive_fallback_forward() {
for i in 0..517 {
let s = "a".repeat(i);
assert_eq!(
i,
first_non_ascii_byte_fallback(s.as_bytes()),
"i: {:?}, len: {:?}, s: {:?}",
i,
s.len(),
s
);
}
}
#[test]
#[cfg(target_arch = "x86_64")]
#[cfg(not(miri))]
fn positive_sse2_forward() {
for i in 0..517 {
let b = "a".repeat(i).into_bytes();
assert_eq!(b.len(), first_non_ascii_byte_sse2(&b));
}
}
#[test]
#[cfg(not(miri))]
fn negative_fallback_forward() {
for i in 0..517 {
for align in 0..65 {
let mut s = "a".repeat(i);
s.push_str("☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃");
let s = s.get(align..).unwrap_or("");
assert_eq!(
i.saturating_sub(align),
first_non_ascii_byte_fallback(s.as_bytes()),
"i: {:?}, align: {:?}, len: {:?}, s: {:?}",
i,
align,
s.len(),
s
);
}
}
}
#[test]
#[cfg(target_arch = "x86_64")]
#[cfg(not(miri))]
fn negative_sse2_forward() {
for i in 0..517 {
for align in 0..65 {
let mut s = "a".repeat(i);
s.push_str("☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃☃");
let s = s.get(align..).unwrap_or("");
assert_eq!(
i.saturating_sub(align),
first_non_ascii_byte_sse2(s.as_bytes()),
"i: {:?}, align: {:?}, len: {:?}, s: {:?}",
i,
align,
s.len(),
s
);
}
}
}
}
+97
View File
@@ -0,0 +1,97 @@
#[cfg(feature = "alloc")]
use alloc::boxed::Box;
/// A wrapper for `&[u8]` that provides convenient string oriented trait impls.
///
/// If you need ownership or a growable byte string buffer, then use
/// [`BString`](struct.BString.html).
///
/// Using a `&BStr` is just like using a `&[u8]`, since `BStr`
/// implements `Deref` to `[u8]`. So all methods available on `[u8]`
/// are also available on `BStr`.
///
/// # Representation
///
/// A `&BStr` has the same representation as a `&str`. That is, a `&BStr` is
/// a fat pointer which consists of a pointer to some bytes and a length.
///
/// # Trait implementations
///
/// The `BStr` type has a number of trait implementations, and in particular,
/// defines equality and ordinal comparisons between `&BStr`, `&str` and
/// `&[u8]` for convenience.
///
/// The `Debug` implementation for `BStr` shows its bytes as a normal string.
/// For invalid UTF-8, hex escape sequences are used.
///
/// The `Display` implementation behaves as if `BStr` were first lossily
/// converted to a `str`. Invalid UTF-8 bytes are substituted with the Unicode
/// replacement codepoint, which looks like this: �.
#[repr(transparent)]
pub struct BStr {
pub(crate) bytes: [u8],
}
impl BStr {
/// Directly creates a `BStr` slice from anything that can be converted
/// to a byte slice.
///
/// This is very similar to the [`B`](crate::B) function, except this
/// returns a `&BStr` instead of a `&[u8]`.
///
/// This is a cost-free conversion.
///
/// # Example
///
/// You can create `BStr`'s from byte arrays, byte slices or even string
/// slices:
///
/// ```
/// use bstr::BStr;
///
/// let a = BStr::new(b"abc");
/// let b = BStr::new(&b"abc"[..]);
/// let c = BStr::new("abc");
///
/// assert_eq!(a, b);
/// assert_eq!(a, c);
/// ```
#[inline]
pub fn new<B: ?Sized + AsRef<[u8]>>(bytes: &B) -> &BStr {
BStr::from_bytes(bytes.as_ref())
}
#[inline]
pub(crate) fn new_mut<B: ?Sized + AsMut<[u8]>>(
bytes: &mut B,
) -> &mut BStr {
BStr::from_bytes_mut(bytes.as_mut())
}
#[inline]
pub(crate) fn from_bytes(slice: &[u8]) -> &BStr {
unsafe { &*(slice as *const [u8] as *const BStr) }
}
#[inline]
pub(crate) fn from_bytes_mut(slice: &mut [u8]) -> &mut BStr {
unsafe { &mut *(slice as *mut [u8] as *mut BStr) }
}
#[inline]
#[cfg(feature = "alloc")]
pub(crate) fn from_boxed_bytes(slice: Box<[u8]>) -> Box<BStr> {
unsafe { Box::from_raw(Box::into_raw(slice) as _) }
}
#[inline]
#[cfg(feature = "alloc")]
pub(crate) fn into_boxed_bytes(slice: Box<BStr>) -> Box<[u8]> {
unsafe { Box::from_raw(Box::into_raw(slice) as _) }
}
#[inline]
pub(crate) fn as_bytes(&self) -> &[u8] {
&self.bytes
}
}
+122
View File
@@ -0,0 +1,122 @@
use alloc::vec::Vec;
use crate::bstr::BStr;
/// A wrapper for `Vec<u8>` that provides convenient string oriented trait
/// impls.
///
/// A `BString` has ownership over its contents and corresponds to
/// a growable or shrinkable buffer. Its borrowed counterpart is a
/// [`BStr`](struct.BStr.html), called a byte string slice.
///
/// Using a `BString` is just like using a `Vec<u8>`, since `BString`
/// implements `Deref` to `Vec<u8>`. So all methods available on `Vec<u8>`
/// are also available on `BString`.
///
/// # Examples
///
/// You can create a new `BString` from a `Vec<u8>` via a `From` impl:
///
/// ```
/// use bstr::BString;
///
/// let s = BString::from("Hello, world!");
/// ```
///
/// # Deref
///
/// The `BString` type implements `Deref` and `DerefMut`, where the target
/// types are `&Vec<u8>` and `&mut Vec<u8>`, respectively. `Deref` permits all of the
/// methods defined on `Vec<u8>` to be implicitly callable on any `BString`.
///
/// For more information about how deref works, see the documentation for the
/// [`std::ops::Deref`](https://doc.rust-lang.org/std/ops/trait.Deref.html)
/// trait.
///
/// # Representation
///
/// A `BString` has the same representation as a `Vec<u8>` and a `String`.
/// That is, it is made up of three word sized components: a pointer to a
/// region of memory containing the bytes, a length and a capacity.
#[derive(Clone)]
#[repr(transparent)]
pub struct BString {
bytes: Vec<u8>,
}
impl BString {
/// Constructs a new `BString` from the given [`Vec`].
///
/// # Examples
///
/// ```
/// use bstr::BString;
///
/// let mut b = BString::new(Vec::with_capacity(10));
/// ```
///
/// This function is `const`:
///
/// ```
/// use bstr::BString;
///
/// const B: BString = BString::new(vec![]);
/// ```
#[inline]
pub const fn new(bytes: Vec<u8>) -> BString {
BString { bytes }
}
#[inline]
pub(crate) fn as_bytes(&self) -> &[u8] {
&self.bytes
}
#[inline]
pub(crate) fn as_bytes_mut(&mut self) -> &mut [u8] {
&mut self.bytes
}
#[inline]
pub(crate) fn as_bstr(&self) -> &BStr {
BStr::new(&self.bytes)
}
#[inline]
pub(crate) fn as_mut_bstr(&mut self) -> &mut BStr {
BStr::new_mut(&mut self.bytes)
}
#[inline]
pub(crate) fn as_vec(&self) -> &Vec<u8> {
&self.bytes
}
#[inline]
pub(crate) fn as_vec_mut(&mut self) -> &mut Vec<u8> {
&mut self.bytes
}
#[inline]
pub(crate) fn into_vec(self) -> Vec<u8> {
self.bytes
}
#[inline]
pub(crate) fn from_vec_ref(v: &Vec<u8>) -> &BString {
// SAFETY: `BString` is a `repr(transparent)` wrapper around `Vec<u8>`, and accepts any
// `Vec<u8>` without validation.
//
// MSRV: Switch this to use `ptr::from_ref` once bstr can require at least Rust 1.72.
unsafe { &*(v as *const Vec<u8> as *const BString) }
}
#[inline]
pub(crate) fn from_vec_mut(v: &mut Vec<u8>) -> &mut BString {
// SAFETY: `BString` is a `repr(transparent)` wrapper around `Vec<u8>`, and accepts any
// `Vec<u8>` without validation.
//
// MSRV: Switch this to use `ptr::from_mut` once bstr can require at least Rust 1.72.
unsafe { &mut *(v as *mut Vec<u8> as *mut BString) }
}
}
+117
View File
@@ -0,0 +1,117 @@
use memchr::{memchr, memchr2, memchr3, memrchr, memrchr2, memrchr3};
mod scalar;
#[inline]
fn build_table(byteset: &[u8]) -> [u8; 256] {
let mut table = [0u8; 256];
for &b in byteset {
table[b as usize] = 1;
}
table
}
#[inline]
pub(crate) fn find(haystack: &[u8], byteset: &[u8]) -> Option<usize> {
match byteset.len() {
0 => None,
1 => memchr(byteset[0], haystack),
2 => memchr2(byteset[0], byteset[1], haystack),
3 => memchr3(byteset[0], byteset[1], byteset[2], haystack),
_ => {
let table = build_table(byteset);
scalar::forward_search_bytes(haystack, |b| table[b as usize] != 0)
}
}
}
#[inline]
pub(crate) fn rfind(haystack: &[u8], byteset: &[u8]) -> Option<usize> {
match byteset.len() {
0 => None,
1 => memrchr(byteset[0], haystack),
2 => memrchr2(byteset[0], byteset[1], haystack),
3 => memrchr3(byteset[0], byteset[1], byteset[2], haystack),
_ => {
let table = build_table(byteset);
scalar::reverse_search_bytes(haystack, |b| table[b as usize] != 0)
}
}
}
#[inline]
pub(crate) fn find_not(haystack: &[u8], byteset: &[u8]) -> Option<usize> {
if haystack.is_empty() {
return None;
}
match byteset.len() {
0 => Some(0),
1 => scalar::inv_memchr(byteset[0], haystack),
2 => scalar::forward_search_bytes(haystack, |b| {
b != byteset[0] && b != byteset[1]
}),
3 => scalar::forward_search_bytes(haystack, |b| {
b != byteset[0] && b != byteset[1] && b != byteset[2]
}),
_ => {
let table = build_table(byteset);
scalar::forward_search_bytes(haystack, |b| table[b as usize] == 0)
}
}
}
#[inline]
pub(crate) fn rfind_not(haystack: &[u8], byteset: &[u8]) -> Option<usize> {
if haystack.is_empty() {
return None;
}
match byteset.len() {
0 => Some(haystack.len() - 1),
1 => scalar::inv_memrchr(byteset[0], haystack),
2 => scalar::reverse_search_bytes(haystack, |b| {
b != byteset[0] && b != byteset[1]
}),
3 => scalar::reverse_search_bytes(haystack, |b| {
b != byteset[0] && b != byteset[1] && b != byteset[2]
}),
_ => {
let table = build_table(byteset);
scalar::reverse_search_bytes(haystack, |b| table[b as usize] == 0)
}
}
}
#[cfg(all(test, feature = "std", not(miri)))]
mod tests {
use alloc::vec::Vec;
quickcheck::quickcheck! {
fn qc_byteset_forward_matches_naive(
haystack: Vec<u8>,
needles: Vec<u8>
) -> bool {
super::find(&haystack, &needles)
== haystack.iter().position(|b| needles.contains(b))
}
fn qc_byteset_backwards_matches_naive(
haystack: Vec<u8>,
needles: Vec<u8>
) -> bool {
super::rfind(&haystack, &needles)
== haystack.iter().rposition(|b| needles.contains(b))
}
fn qc_byteset_forward_not_matches_naive(
haystack: Vec<u8>,
needles: Vec<u8>
) -> bool {
super::find_not(&haystack, &needles)
== haystack.iter().position(|b| !needles.contains(b))
}
fn qc_byteset_backwards_not_matches_naive(
haystack: Vec<u8>,
needles: Vec<u8>
) -> bool {
super::rfind_not(&haystack, &needles)
== haystack.iter().rposition(|b| !needles.contains(b))
}
}
}
+306
View File
@@ -0,0 +1,306 @@
// This is adapted from `fallback.rs` from rust-memchr. It's modified to return
// the 'inverse' query of memchr, e.g. finding the first byte not in the
// provided set. This is simple for the 1-byte case.
use core::{cmp, usize};
const USIZE_BYTES: usize = core::mem::size_of::<usize>();
const ALIGN_MASK: usize = core::mem::align_of::<usize>() - 1;
// The number of bytes to loop at in one iteration of memchr/memrchr.
const LOOP_SIZE: usize = 2 * USIZE_BYTES;
/// Repeat the given byte into a word size number. That is, every 8 bits
/// is equivalent to the given byte. For example, if `b` is `\x4E` or
/// `01001110` in binary, then the returned value on a 32-bit system would be:
/// `01001110_01001110_01001110_01001110`.
#[inline(always)]
fn repeat_byte(b: u8) -> usize {
(b as usize) * (usize::MAX / 255)
}
pub fn inv_memchr(n1: u8, haystack: &[u8]) -> Option<usize> {
let vn1 = repeat_byte(n1);
let confirm = |byte| byte != n1;
let loop_size = cmp::min(LOOP_SIZE, haystack.len());
let start_ptr = haystack.as_ptr();
unsafe {
let end_ptr = haystack.as_ptr().add(haystack.len());
let mut ptr = start_ptr;
if haystack.len() < USIZE_BYTES {
return forward_search(start_ptr, end_ptr, ptr, confirm);
}
let chunk = read_unaligned_usize(ptr);
if (chunk ^ vn1) != 0 {
return forward_search(start_ptr, end_ptr, ptr, confirm);
}
ptr = ptr.add(USIZE_BYTES - (start_ptr as usize & ALIGN_MASK));
debug_assert!(ptr > start_ptr);
debug_assert!(end_ptr.sub(USIZE_BYTES) >= start_ptr);
while loop_size == LOOP_SIZE && ptr <= end_ptr.sub(loop_size) {
debug_assert_eq!(0, (ptr as usize) % USIZE_BYTES);
let a = *(ptr as *const usize);
let b = *(ptr.add(USIZE_BYTES) as *const usize);
let eqa = (a ^ vn1) != 0;
let eqb = (b ^ vn1) != 0;
if eqa || eqb {
break;
}
ptr = ptr.add(LOOP_SIZE);
}
forward_search(start_ptr, end_ptr, ptr, confirm)
}
}
/// Return the last index not matching the byte `x` in `text`.
pub fn inv_memrchr(n1: u8, haystack: &[u8]) -> Option<usize> {
let vn1 = repeat_byte(n1);
let confirm = |byte| byte != n1;
let loop_size = cmp::min(LOOP_SIZE, haystack.len());
let start_ptr = haystack.as_ptr();
unsafe {
let end_ptr = haystack.as_ptr().add(haystack.len());
let mut ptr = end_ptr;
if haystack.len() < USIZE_BYTES {
return reverse_search(start_ptr, end_ptr, ptr, confirm);
}
let chunk = read_unaligned_usize(ptr.sub(USIZE_BYTES));
if (chunk ^ vn1) != 0 {
return reverse_search(start_ptr, end_ptr, ptr, confirm);
}
ptr = ptr.sub(end_ptr as usize & ALIGN_MASK);
debug_assert!(start_ptr <= ptr && ptr <= end_ptr);
while loop_size == LOOP_SIZE && ptr >= start_ptr.add(loop_size) {
debug_assert_eq!(0, (ptr as usize) % USIZE_BYTES);
let a = *(ptr.sub(2 * USIZE_BYTES) as *const usize);
let b = *(ptr.sub(1 * USIZE_BYTES) as *const usize);
let eqa = (a ^ vn1) != 0;
let eqb = (b ^ vn1) != 0;
if eqa || eqb {
break;
}
ptr = ptr.sub(loop_size);
}
reverse_search(start_ptr, end_ptr, ptr, confirm)
}
}
#[inline(always)]
unsafe fn forward_search<F: Fn(u8) -> bool>(
start_ptr: *const u8,
end_ptr: *const u8,
mut ptr: *const u8,
confirm: F,
) -> Option<usize> {
debug_assert!(start_ptr <= ptr);
debug_assert!(ptr <= end_ptr);
while ptr < end_ptr {
if confirm(*ptr) {
return Some(sub(ptr, start_ptr));
}
ptr = ptr.offset(1);
}
None
}
#[inline(always)]
unsafe fn reverse_search<F: Fn(u8) -> bool>(
start_ptr: *const u8,
end_ptr: *const u8,
mut ptr: *const u8,
confirm: F,
) -> Option<usize> {
debug_assert!(start_ptr <= ptr);
debug_assert!(ptr <= end_ptr);
while ptr > start_ptr {
ptr = ptr.offset(-1);
if confirm(*ptr) {
return Some(sub(ptr, start_ptr));
}
}
None
}
unsafe fn read_unaligned_usize(ptr: *const u8) -> usize {
(ptr as *const usize).read_unaligned()
}
/// Subtract `b` from `a` and return the difference. `a` should be greater than
/// or equal to `b`.
fn sub(a: *const u8, b: *const u8) -> usize {
debug_assert!(a >= b);
(a as usize) - (b as usize)
}
/// Safe wrapper around `forward_search`
#[inline]
pub(crate) fn forward_search_bytes<F: Fn(u8) -> bool>(
s: &[u8],
confirm: F,
) -> Option<usize> {
unsafe {
let start = s.as_ptr();
let end = start.add(s.len());
forward_search(start, end, start, confirm)
}
}
/// Safe wrapper around `reverse_search`
#[inline]
pub(crate) fn reverse_search_bytes<F: Fn(u8) -> bool>(
s: &[u8],
confirm: F,
) -> Option<usize> {
unsafe {
let start = s.as_ptr();
let end = start.add(s.len());
reverse_search(start, end, end, confirm)
}
}
#[cfg(all(test, feature = "std"))]
mod tests {
use alloc::{vec, vec::Vec};
use super::{inv_memchr, inv_memrchr};
// search string, search byte, inv_memchr result, inv_memrchr result.
// these are expanded into a much larger set of tests in build_tests
const TESTS: &[(&[u8], u8, usize, usize)] = &[
(b"z", b'a', 0, 0),
(b"zz", b'a', 0, 1),
(b"aza", b'a', 1, 1),
(b"zaz", b'a', 0, 2),
(b"zza", b'a', 0, 1),
(b"zaa", b'a', 0, 0),
(b"zzz", b'a', 0, 2),
];
type TestCase = (Vec<u8>, u8, Option<(usize, usize)>);
fn build_tests() -> Vec<TestCase> {
#[cfg(not(miri))]
const MAX_PER: usize = 515;
#[cfg(miri)]
const MAX_PER: usize = 10;
let mut result = vec![];
for &(search, byte, fwd_pos, rev_pos) in TESTS {
result.push((search.to_vec(), byte, Some((fwd_pos, rev_pos))));
for i in 1..MAX_PER {
// add a bunch of copies of the search byte to the end.
let mut suffixed: Vec<u8> = search.into();
suffixed.extend(std::iter::repeat(byte).take(i));
result.push((suffixed, byte, Some((fwd_pos, rev_pos))));
// add a bunch of copies of the search byte to the start.
let mut prefixed: Vec<u8> =
std::iter::repeat(byte).take(i).collect();
prefixed.extend(search);
result.push((
prefixed,
byte,
Some((fwd_pos + i, rev_pos + i)),
));
// add a bunch of copies of the search byte to both ends.
let mut surrounded: Vec<u8> =
std::iter::repeat(byte).take(i).collect();
surrounded.extend(search);
surrounded.extend(std::iter::repeat(byte).take(i));
result.push((
surrounded,
byte,
Some((fwd_pos + i, rev_pos + i)),
));
}
}
// build non-matching tests for several sizes
for i in 0..MAX_PER {
result.push((
std::iter::repeat(b'\0').take(i).collect(),
b'\0',
None,
));
}
result
}
#[test]
fn test_inv_memchr() {
use crate::{ByteSlice, B};
#[cfg(not(miri))]
const MAX_OFFSET: usize = 130;
#[cfg(miri)]
const MAX_OFFSET: usize = 13;
for (search, byte, matching) in build_tests() {
assert_eq!(
inv_memchr(byte, &search),
matching.map(|m| m.0),
"inv_memchr when searching for {:?} in {:?}",
byte as char,
// better printing
B(&search).as_bstr(),
);
assert_eq!(
inv_memrchr(byte, &search),
matching.map(|m| m.1),
"inv_memrchr when searching for {:?} in {:?}",
byte as char,
// better printing
B(&search).as_bstr(),
);
// Test a rather large number off offsets for potential alignment
// issues.
for offset in 1..MAX_OFFSET {
if offset >= search.len() {
break;
}
// If this would cause us to shift the results off the end,
// skip it so that we don't have to recompute them.
if let Some((f, r)) = matching {
if offset > f || offset > r {
break;
}
}
let realigned = &search[offset..];
let forward_pos = matching.map(|m| m.0 - offset);
let reverse_pos = matching.map(|m| m.1 - offset);
assert_eq!(
inv_memchr(byte, &realigned),
forward_pos,
"inv_memchr when searching (realigned by {}) for {:?} in {:?}",
offset,
byte as char,
realigned.as_bstr(),
);
assert_eq!(
inv_memrchr(byte, &realigned),
reverse_pos,
"inv_memrchr when searching (realigned by {}) for {:?} in {:?}",
offset,
byte as char,
realigned.as_bstr(),
);
}
}
}
}
+453
View File
@@ -0,0 +1,453 @@
/// An iterator of `char` values that represent an escaping of arbitrary bytes.
///
/// The lifetime parameter `'a` refers to the lifetime of the bytes being
/// escaped.
///
/// This iterator is created by the
/// [`ByteSlice::escape_bytes`](crate::ByteSlice::escape_bytes) method.
#[derive(Clone, Debug)]
pub struct EscapeBytes<'a> {
remaining: &'a [u8],
state: EscapeState,
}
impl<'a> EscapeBytes<'a> {
pub(crate) fn new(bytes: &'a [u8]) -> EscapeBytes<'a> {
EscapeBytes { remaining: bytes, state: EscapeState::Start }
}
}
impl<'a> Iterator for EscapeBytes<'a> {
type Item = char;
#[inline]
fn next(&mut self) -> Option<char> {
use self::EscapeState::*;
match self.state {
Start => {
let byte = match crate::decode_utf8(self.remaining) {
(None, 0) => return None,
// If we see invalid UTF-8 or ASCII, then we always just
// peel one byte off. If it's printable ASCII, we'll pass
// it through as-is below. Otherwise, below, it will get
// escaped in some way.
(None, _) | (Some(_), 1) => {
let byte = self.remaining[0];
self.remaining = &self.remaining[1..];
byte
}
// For any valid UTF-8 that is not ASCII, we pass it
// through as-is. We don't do any Unicode escaping.
(Some(ch), size) => {
self.remaining = &self.remaining[size..];
return Some(ch);
}
};
self.state = match byte {
0x21..=0x5B | 0x5D..=0x7E => {
return Some(char::from(byte))
}
b'\0' => SpecialEscape('0'),
b'\n' => SpecialEscape('n'),
b'\r' => SpecialEscape('r'),
b'\t' => SpecialEscape('t'),
b'\\' => SpecialEscape('\\'),
_ => HexEscapeX(byte),
};
Some('\\')
}
SpecialEscape(ch) => {
self.state = Start;
Some(ch)
}
HexEscapeX(byte) => {
self.state = HexEscapeHighNybble(byte);
Some('x')
}
HexEscapeHighNybble(byte) => {
self.state = HexEscapeLowNybble(byte);
let nybble = byte >> 4;
Some(hexdigit_to_char(nybble))
}
HexEscapeLowNybble(byte) => {
self.state = Start;
let nybble = byte & 0xF;
Some(hexdigit_to_char(nybble))
}
}
}
}
impl<'a> core::fmt::Display for EscapeBytes<'a> {
fn fmt(&self, f: &mut core::fmt::Formatter) -> core::fmt::Result {
use core::fmt::Write;
for ch in self.clone() {
f.write_char(ch)?;
}
Ok(())
}
}
/// The state used by the FSM in the escaping iterator.
#[derive(Clone, Debug)]
enum EscapeState {
/// Read and remove the next byte from 'remaining'. If 'remaining' is
/// empty, then return None. Otherwise, escape the byte according to the
/// following rules or emit it as-is.
///
/// If it's \n, \r, \t, \\ or \0, then emit a '\' and set the current
/// state to 'SpecialEscape(n | r | t | \ | 0)'. Otherwise, if the 'byte'
/// is not in [\x21-\x5B\x5D-\x7E], then emit a '\' and set the state to
/// to 'HexEscapeX(byte)'.
Start,
/// Emit the given codepoint as is. This assumes '\' has just been emitted.
/// Then set the state to 'Start'.
SpecialEscape(char),
/// Emit the 'x' part of a hex escape. This assumes '\' has just been
/// emitted. Then set the state to 'HexEscapeHighNybble(byte)'.
HexEscapeX(u8),
/// Emit the high nybble of the byte as a hexadecimal digit. This
/// assumes '\x' has just been emitted. Then set the state to
/// 'HexEscapeLowNybble(byte)'.
HexEscapeHighNybble(u8),
/// Emit the low nybble of the byte as a hexadecimal digit. This assume
/// '\xZ' has just been emitted, where 'Z' is the high nybble of this byte.
/// Then set the state to 'Start'.
HexEscapeLowNybble(u8),
}
/// An iterator of `u8` values that represent an unescaping of a sequence of
/// codepoints.
///
/// The type parameter `I` refers to the iterator of codepoints that is
/// unescaped.
///
/// Currently this iterator is not exposed in the crate API, and instead all
/// we expose is a `ByteVec::unescape` method. Which of course requires an
/// alloc. That's the most convenient form of this, but in theory, we could
/// expose this for core-only use cases too. I'm just not quite sure what the
/// API should be.
#[derive(Clone, Debug)]
#[cfg(feature = "alloc")]
pub(crate) struct UnescapeBytes<I> {
it: I,
state: UnescapeState,
}
#[cfg(feature = "alloc")]
impl<I: Iterator<Item = char>> UnescapeBytes<I> {
pub(crate) fn new<T: IntoIterator<IntoIter = I>>(
t: T,
) -> UnescapeBytes<I> {
UnescapeBytes { it: t.into_iter(), state: UnescapeState::Start }
}
}
#[cfg(feature = "alloc")]
impl<I: Iterator<Item = char>> Iterator for UnescapeBytes<I> {
type Item = u8;
fn next(&mut self) -> Option<u8> {
use self::UnescapeState::*;
loop {
match self.state {
Start => {
let ch = self.it.next()?;
match ch {
'\\' => {
self.state = Escape;
}
ch => {
self.state = UnescapeState::bytes(&[], ch);
}
}
}
Bytes { buf, mut cur, len } => {
let byte = buf[cur];
cur += 1;
if cur >= len {
self.state = Start;
} else {
self.state = Bytes { buf, cur, len };
}
return Some(byte);
}
Escape => {
let ch = match self.it.next() {
Some(ch) => ch,
None => {
self.state = Start;
// Incomplete escape sequences unescape as
// themselves.
return Some(b'\\');
}
};
match ch {
'0' => {
self.state = Start;
return Some(b'\x00');
}
'\\' => {
self.state = Start;
return Some(b'\\');
}
'r' => {
self.state = Start;
return Some(b'\r');
}
'n' => {
self.state = Start;
return Some(b'\n');
}
't' => {
self.state = Start;
return Some(b'\t');
}
'x' => {
self.state = HexFirst;
}
ch => {
// An invalid escape sequence unescapes as itself.
self.state = UnescapeState::bytes(&[b'\\'], ch);
}
}
}
HexFirst => {
let ch = match self.it.next() {
Some(ch) => ch,
None => {
// An incomplete escape sequence unescapes as
// itself.
self.state = UnescapeState::bytes_raw(&[b'x']);
return Some(b'\\');
}
};
match ch {
'0'..='9' | 'A'..='F' | 'a'..='f' => {
self.state = HexSecond(ch);
}
ch => {
// An invalid escape sequence unescapes as itself.
self.state = UnescapeState::bytes(&[b'x'], ch);
return Some(b'\\');
}
}
}
HexSecond(first) => {
let second = match self.it.next() {
Some(ch) => ch,
None => {
// An incomplete escape sequence unescapes as
// itself.
self.state = UnescapeState::bytes(&[b'x'], first);
return Some(b'\\');
}
};
match second {
'0'..='9' | 'A'..='F' | 'a'..='f' => {
self.state = Start;
let hinybble = char_to_hexdigit(first);
let lonybble = char_to_hexdigit(second);
let byte = hinybble << 4 | lonybble;
return Some(byte);
}
ch => {
// An invalid escape sequence unescapes as itself.
self.state =
UnescapeState::bytes2(&[b'x'], first, ch);
return Some(b'\\');
}
}
}
}
}
}
}
/// The state used by the FSM in the unescaping iterator.
#[derive(Clone, Debug)]
#[cfg(feature = "alloc")]
enum UnescapeState {
/// The start state. Look for an escape sequence, otherwise emit the next
/// codepoint as-is.
Start,
/// Emit the byte at `buf[cur]`.
///
/// This state should never be created when `cur >= len`. That is, when
/// this state is visited, it is assumed that `cur < len`.
Bytes { buf: [u8; 11], cur: usize, len: usize },
/// This state is entered after a `\` is seen.
Escape,
/// This state is entered after a `\x` is seen.
HexFirst,
/// This state is entered after a `\xN` is seen, where `N` is in
/// `[0-9A-Fa-f]`. The given codepoint corresponds to `N`.
HexSecond(char),
}
#[cfg(feature = "alloc")]
impl UnescapeState {
/// Create a new `Bytes` variant with the given slice.
///
/// # Panics
///
/// Panics if `bytes.len() > 11`.
fn bytes_raw(bytes: &[u8]) -> UnescapeState {
// This can be increased, you just need to make sure 'buf' in the
// 'Bytes' state has enough room.
assert!(bytes.len() <= 11, "no more than 11 bytes allowed");
let mut buf = [0; 11];
buf[..bytes.len()].copy_from_slice(bytes);
UnescapeState::Bytes { buf, cur: 0, len: bytes.len() }
}
/// Create a new `Bytes` variant with the prefix byte slice, followed by
/// the UTF-8 encoding of the given char.
///
/// # Panics
///
/// Panics if `prefix.len() > 3`.
fn bytes(prefix: &[u8], ch: char) -> UnescapeState {
// This can be increased, you just need to make sure 'buf' in the
// 'Bytes' state has enough room.
assert!(prefix.len() <= 3, "no more than 3 bytes allowed");
let mut buf = [0; 11];
buf[..prefix.len()].copy_from_slice(prefix);
let chlen = ch.encode_utf8(&mut buf[prefix.len()..]).len();
UnescapeState::Bytes { buf, cur: 0, len: prefix.len() + chlen }
}
/// Create a new `Bytes` variant with the prefix byte slice, followed by
/// the UTF-8 encoding of `ch1` and then `ch2`.
///
/// # Panics
///
/// Panics if `prefix.len() > 3`.
fn bytes2(prefix: &[u8], ch1: char, ch2: char) -> UnescapeState {
// This can be increased, you just need to make sure 'buf' in the
// 'Bytes' state has enough room.
assert!(prefix.len() <= 3, "no more than 3 bytes allowed");
let mut buf = [0; 11];
buf[..prefix.len()].copy_from_slice(prefix);
let len1 = ch1.encode_utf8(&mut buf[prefix.len()..]).len();
let len2 = ch2.encode_utf8(&mut buf[prefix.len() + len1..]).len();
UnescapeState::Bytes { buf, cur: 0, len: prefix.len() + len1 + len2 }
}
}
/// Convert the given codepoint to its corresponding hexadecimal digit.
///
/// # Panics
///
/// This panics if `ch` is not in `[0-9A-Fa-f]`.
#[cfg(feature = "alloc")]
fn char_to_hexdigit(ch: char) -> u8 {
u8::try_from(ch.to_digit(16).unwrap()).unwrap()
}
/// Convert the given hexadecimal digit to its corresponding codepoint.
///
/// # Panics
///
/// This panics when `digit > 15`.
fn hexdigit_to_char(digit: u8) -> char {
char::from_digit(u32::from(digit), 16).unwrap().to_ascii_uppercase()
}
#[cfg(all(test, feature = "std"))]
mod tests {
use alloc::string::{String, ToString};
use crate::BString;
use super::*;
#[allow(non_snake_case)]
fn B<B: AsRef<[u8]>>(bytes: B) -> BString {
BString::from(bytes.as_ref())
}
fn e<B: AsRef<[u8]>>(bytes: B) -> String {
EscapeBytes::new(bytes.as_ref()).to_string()
}
fn u(string: &str) -> BString {
UnescapeBytes::new(string.chars()).collect()
}
#[test]
fn escape() {
assert_eq!(r"a", e(br"a"));
assert_eq!(r"\\x61", e(br"\x61"));
assert_eq!(r"a", e(b"\x61"));
assert_eq!(r"~", e(b"\x7E"));
assert_eq!(r"\x7F", e(b"\x7F"));
assert_eq!(r"\n", e(b"\n"));
assert_eq!(r"\r", e(b"\r"));
assert_eq!(r"\t", e(b"\t"));
assert_eq!(r"\\", e(b"\\"));
assert_eq!(r"\0", e(b"\0"));
assert_eq!(r"\0", e(b"\x00"));
assert_eq!(r"\x88", e(b"\x88"));
assert_eq!(r"\x8F", e(b"\x8F"));
assert_eq!(r"\xF8", e(b"\xF8"));
assert_eq!(r"\xFF", e(b"\xFF"));
assert_eq!(r"\xE2", e(b"\xE2"));
assert_eq!(r"\xE2\x98", e(b"\xE2\x98"));
assert_eq!(r"☃", e(b"\xE2\x98\x83"));
assert_eq!(r"\xF0", e(b"\xF0"));
assert_eq!(r"\xF0\x9F", e(b"\xF0\x9F"));
assert_eq!(r"\xF0\x9F\x92", e(b"\xF0\x9F\x92"));
assert_eq!(r"💩", e(b"\xF0\x9F\x92\xA9"));
}
#[test]
fn unescape() {
assert_eq!(B(r"a"), u(r"a"));
assert_eq!(B(r"\x61"), u(r"\\x61"));
assert_eq!(B(r"a"), u(r"\x61"));
assert_eq!(B(r"~"), u(r"\x7E"));
assert_eq!(B(b"\x7F"), u(r"\x7F"));
assert_eq!(B(b"\n"), u(r"\n"));
assert_eq!(B(b"\r"), u(r"\r"));
assert_eq!(B(b"\t"), u(r"\t"));
assert_eq!(B(b"\\"), u(r"\\"));
assert_eq!(B(b"\0"), u(r"\0"));
assert_eq!(B(b"\0"), u(r"\x00"));
assert_eq!(B(b"\x88"), u(r"\x88"));
assert_eq!(B(b"\x8F"), u(r"\x8F"));
assert_eq!(B(b"\xF8"), u(r"\xF8"));
assert_eq!(B(b"\xFF"), u(r"\xFF"));
assert_eq!(B(b"\xE2"), u(r"\xE2"));
assert_eq!(B(b"\xE2\x98"), u(r"\xE2\x98"));
assert_eq!(B("☃"), u(r"\xE2\x98\x83"));
assert_eq!(B(b"\xF0"), u(r"\xf0"));
assert_eq!(B(b"\xF0\x9F"), u(r"\xf0\x9f"));
assert_eq!(B(b"\xF0\x9F\x92"), u(r"\xf0\x9f\x92"));
assert_eq!(B("💩"), u(r"\xf0\x9f\x92\xa9"));
}
#[test]
fn unescape_weird() {
assert_eq!(B(b"\\"), u(r"\"));
assert_eq!(B(b"\\"), u(r"\\"));
assert_eq!(B(b"\\x"), u(r"\x"));
assert_eq!(B(b"\\xA"), u(r"\xA"));
assert_eq!(B(b"\\xZ"), u(r"\xZ"));
assert_eq!(B(b"\\xZZ"), u(r"\xZZ"));
assert_eq!(B(b"\\i"), u(r"\i"));
assert_eq!(B(b"\\u"), u(r"\u"));
assert_eq!(B(b"\\u{2603}"), u(r"\u{2603}"));
}
}
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
+520
View File
@@ -0,0 +1,520 @@
/*!
Utilities for working with I/O using byte strings.
This module currently only exports a single trait, `BufReadExt`, which provides
facilities for conveniently and efficiently working with lines as byte strings.
More APIs may be added in the future.
*/
use alloc::{vec, vec::Vec};
use std::io;
use crate::{ext_slice::ByteSlice, ext_vec::ByteVec};
/// An extension trait for
/// [`std::io::BufRead`](https://doc.rust-lang.org/std/io/trait.BufRead.html)
/// which provides convenience APIs for dealing with byte strings.
pub trait BufReadExt: io::BufRead {
/// Returns an iterator over the lines of this reader, where each line
/// is represented as a byte string.
///
/// Each item yielded by this iterator is a `io::Result<Vec<u8>>`, where
/// an error is yielded if there was a problem reading from the underlying
/// reader.
///
/// On success, the next line in the iterator is returned. The line does
/// *not* contain a trailing `\n` or `\r\n`.
///
/// # Examples
///
/// Basic usage:
///
/// ```
/// use std::io;
///
/// use bstr::io::BufReadExt;
///
/// # fn example() -> Result<(), io::Error> {
/// let mut cursor = io::Cursor::new(b"lorem\nipsum\r\ndolor");
///
/// let mut lines = vec![];
/// for result in cursor.byte_lines() {
/// let line = result?;
/// lines.push(line);
/// }
/// assert_eq!(lines.len(), 3);
/// assert_eq!(lines[0], "lorem".as_bytes());
/// assert_eq!(lines[1], "ipsum".as_bytes());
/// assert_eq!(lines[2], "dolor".as_bytes());
/// # Ok(()) }; example().unwrap()
/// ```
fn byte_lines(self) -> ByteLines<Self>
where
Self: Sized,
{
ByteLines { buf: self }
}
/// Returns an iterator over byte-terminated records of this reader, where
/// each record is represented as a byte string.
///
/// Each item yielded by this iterator is a `io::Result<Vec<u8>>`, where
/// an error is yielded if there was a problem reading from the underlying
/// reader.
///
/// On success, the next record in the iterator is returned. The record
/// does *not* contain its trailing terminator.
///
/// Note that calling `byte_records(b'\n')` differs from `byte_lines()` in
/// that it has no special handling for `\r`.
///
/// # Examples
///
/// Basic usage:
///
/// ```
/// use std::io;
///
/// use bstr::io::BufReadExt;
///
/// # fn example() -> Result<(), io::Error> {
/// let mut cursor = io::Cursor::new(b"lorem\x00ipsum\x00dolor");
///
/// let mut records = vec![];
/// for result in cursor.byte_records(b'\x00') {
/// let record = result?;
/// records.push(record);
/// }
/// assert_eq!(records.len(), 3);
/// assert_eq!(records[0], "lorem".as_bytes());
/// assert_eq!(records[1], "ipsum".as_bytes());
/// assert_eq!(records[2], "dolor".as_bytes());
/// # Ok(()) }; example().unwrap()
/// ```
fn byte_records(self, terminator: u8) -> ByteRecords<Self>
where
Self: Sized,
{
ByteRecords { terminator, buf: self }
}
/// Executes the given closure on each line in the underlying reader.
///
/// If the closure returns an error (or if the underlying reader returns an
/// error), then iteration is stopped and the error is returned. If false
/// is returned, then iteration is stopped and no error is returned.
///
/// The closure given is called on exactly the same values as yielded by
/// the [`byte_lines`](trait.BufReadExt.html#method.byte_lines)
/// iterator. Namely, lines do _not_ contain trailing `\n` or `\r\n` bytes.
///
/// This routine is useful for iterating over lines as quickly as
/// possible. Namely, a single allocation is reused for each line.
///
/// # Examples
///
/// Basic usage:
///
/// ```
/// use std::io;
///
/// use bstr::io::BufReadExt;
///
/// # fn example() -> Result<(), io::Error> {
/// let mut cursor = io::Cursor::new(b"lorem\nipsum\r\ndolor");
///
/// let mut lines = vec![];
/// cursor.for_byte_line(|line| {
/// lines.push(line.to_vec());
/// Ok(true)
/// })?;
/// assert_eq!(lines.len(), 3);
/// assert_eq!(lines[0], "lorem".as_bytes());
/// assert_eq!(lines[1], "ipsum".as_bytes());
/// assert_eq!(lines[2], "dolor".as_bytes());
/// # Ok(()) }; example().unwrap()
/// ```
fn for_byte_line<F>(&mut self, mut for_each_line: F) -> io::Result<()>
where
Self: Sized,
F: FnMut(&[u8]) -> io::Result<bool>,
{
self.for_byte_line_with_terminator(|line| {
for_each_line(trim_line_slice(line))
})
}
/// Executes the given closure on each byte-terminated record in the
/// underlying reader.
///
/// If the closure returns an error (or if the underlying reader returns an
/// error), then iteration is stopped and the error is returned. If false
/// is returned, then iteration is stopped and no error is returned.
///
/// The closure given is called on exactly the same values as yielded by
/// the [`byte_records`](trait.BufReadExt.html#method.byte_records)
/// iterator. Namely, records do _not_ contain a trailing terminator byte.
///
/// This routine is useful for iterating over records as quickly as
/// possible. Namely, a single allocation is reused for each record.
///
/// # Examples
///
/// Basic usage:
///
/// ```
/// use std::io;
///
/// use bstr::io::BufReadExt;
///
/// # fn example() -> Result<(), io::Error> {
/// let mut cursor = io::Cursor::new(b"lorem\x00ipsum\x00dolor");
///
/// let mut records = vec![];
/// cursor.for_byte_record(b'\x00', |record| {
/// records.push(record.to_vec());
/// Ok(true)
/// })?;
/// assert_eq!(records.len(), 3);
/// assert_eq!(records[0], "lorem".as_bytes());
/// assert_eq!(records[1], "ipsum".as_bytes());
/// assert_eq!(records[2], "dolor".as_bytes());
/// # Ok(()) }; example().unwrap()
/// ```
fn for_byte_record<F>(
&mut self,
terminator: u8,
mut for_each_record: F,
) -> io::Result<()>
where
Self: Sized,
F: FnMut(&[u8]) -> io::Result<bool>,
{
self.for_byte_record_with_terminator(terminator, |chunk| {
for_each_record(trim_record_slice(chunk, terminator))
})
}
/// Executes the given closure on each line in the underlying reader.
///
/// If the closure returns an error (or if the underlying reader returns an
/// error), then iteration is stopped and the error is returned. If false
/// is returned, then iteration is stopped and no error is returned.
///
/// Unlike
/// [`for_byte_line`](trait.BufReadExt.html#method.for_byte_line),
/// the lines given to the closure *do* include the line terminator, if one
/// exists.
///
/// This routine is useful for iterating over lines as quickly as
/// possible. Namely, a single allocation is reused for each line.
///
/// This is identical to `for_byte_record_with_terminator` with a
/// terminator of `\n`.
///
/// # Examples
///
/// Basic usage:
///
/// ```
/// use std::io;
///
/// use bstr::io::BufReadExt;
///
/// # fn example() -> Result<(), io::Error> {
/// let mut cursor = io::Cursor::new(b"lorem\nipsum\r\ndolor");
///
/// let mut lines = vec![];
/// cursor.for_byte_line_with_terminator(|line| {
/// lines.push(line.to_vec());
/// Ok(true)
/// })?;
/// assert_eq!(lines.len(), 3);
/// assert_eq!(lines[0], "lorem\n".as_bytes());
/// assert_eq!(lines[1], "ipsum\r\n".as_bytes());
/// assert_eq!(lines[2], "dolor".as_bytes());
/// # Ok(()) }; example().unwrap()
/// ```
fn for_byte_line_with_terminator<F>(
&mut self,
for_each_line: F,
) -> io::Result<()>
where
Self: Sized,
F: FnMut(&[u8]) -> io::Result<bool>,
{
self.for_byte_record_with_terminator(b'\n', for_each_line)
}
/// Executes the given closure on each byte-terminated record in the
/// underlying reader.
///
/// If the closure returns an error (or if the underlying reader returns an
/// error), then iteration is stopped and the error is returned. If false
/// is returned, then iteration is stopped and no error is returned.
///
/// Unlike
/// [`for_byte_record`](trait.BufReadExt.html#method.for_byte_record),
/// the lines given to the closure *do* include the record terminator, if
/// one exists.
///
/// This routine is useful for iterating over records as quickly as
/// possible. Namely, a single allocation is reused for each record.
///
/// # Examples
///
/// Basic usage:
///
/// ```
/// use std::io;
///
/// use bstr::{io::BufReadExt, B};
///
/// # fn example() -> Result<(), io::Error> {
/// let mut cursor = io::Cursor::new(b"lorem\x00ipsum\x00dolor");
///
/// let mut records = vec![];
/// cursor.for_byte_record_with_terminator(b'\x00', |record| {
/// records.push(record.to_vec());
/// Ok(true)
/// })?;
/// assert_eq!(records.len(), 3);
/// assert_eq!(records[0], B(b"lorem\x00"));
/// assert_eq!(records[1], B("ipsum\x00"));
/// assert_eq!(records[2], B("dolor"));
/// # Ok(()) }; example().unwrap()
/// ```
fn for_byte_record_with_terminator<F>(
&mut self,
terminator: u8,
mut for_each_record: F,
) -> io::Result<()>
where
Self: Sized,
F: FnMut(&[u8]) -> io::Result<bool>,
{
let mut bytes = vec![];
let mut res = Ok(());
let mut consumed = 0;
'outer: loop {
// Lend out complete record slices from our buffer
{
let mut buf = self.fill_buf()?;
if buf.is_empty() {
break;
}
while let Some(index) = buf.find_byte(terminator) {
let (record, rest) = buf.split_at(index + 1);
buf = rest;
consumed += record.len();
match for_each_record(record) {
Ok(false) => break 'outer,
Err(err) => {
res = Err(err);
break 'outer;
}
_ => (),
}
}
// Copy the final record fragment to our local buffer. This
// saves read_until() from re-scanning a buffer we know
// contains no remaining terminators.
bytes.extend_from_slice(buf);
consumed += buf.len();
}
self.consume(consumed);
consumed = 0;
// N.B. read_until uses a different version of memchr that may
// be slower than the memchr crate that bstr uses. However, this
// should only run for a fairly small number of records, assuming a
// decent buffer size.
self.read_until(terminator, &mut bytes)?;
if bytes.is_empty() || !for_each_record(&bytes)? {
break;
}
bytes.clear();
}
self.consume(consumed);
res
}
}
impl<B: io::BufRead> BufReadExt for B {}
/// An iterator over lines from an instance of
/// [`std::io::BufRead`](https://doc.rust-lang.org/std/io/trait.BufRead.html).
///
/// This iterator is generally created by calling the
/// [`byte_lines`](trait.BufReadExt.html#method.byte_lines)
/// method on the
/// [`BufReadExt`](trait.BufReadExt.html)
/// trait.
#[derive(Debug)]
pub struct ByteLines<B> {
buf: B,
}
/// An iterator over records from an instance of
/// [`std::io::BufRead`](https://doc.rust-lang.org/std/io/trait.BufRead.html).
///
/// A byte record is any sequence of bytes terminated by a particular byte
/// chosen by the caller. For example, NUL separated byte strings are said to
/// be NUL-terminated byte records.
///
/// This iterator is generally created by calling the
/// [`byte_records`](trait.BufReadExt.html#method.byte_records)
/// method on the
/// [`BufReadExt`](trait.BufReadExt.html)
/// trait.
#[derive(Debug)]
pub struct ByteRecords<B> {
buf: B,
terminator: u8,
}
impl<B: io::BufRead> Iterator for ByteLines<B> {
type Item = io::Result<Vec<u8>>;
fn next(&mut self) -> Option<io::Result<Vec<u8>>> {
let mut bytes = vec![];
match self.buf.read_until(b'\n', &mut bytes) {
Err(e) => Some(Err(e)),
Ok(0) => None,
Ok(_) => {
trim_line(&mut bytes);
Some(Ok(bytes))
}
}
}
}
impl<B: io::BufRead> Iterator for ByteRecords<B> {
type Item = io::Result<Vec<u8>>;
fn next(&mut self) -> Option<io::Result<Vec<u8>>> {
let mut bytes = vec![];
match self.buf.read_until(self.terminator, &mut bytes) {
Err(e) => Some(Err(e)),
Ok(0) => None,
Ok(_) => {
trim_record(&mut bytes, self.terminator);
Some(Ok(bytes))
}
}
}
}
fn trim_line(line: &mut Vec<u8>) {
if line.last_byte() == Some(b'\n') {
line.pop_byte();
if line.last_byte() == Some(b'\r') {
line.pop_byte();
}
}
}
fn trim_line_slice(mut line: &[u8]) -> &[u8] {
if line.last_byte() == Some(b'\n') {
line = &line[..line.len() - 1];
if line.last_byte() == Some(b'\r') {
line = &line[..line.len() - 1];
}
}
line
}
fn trim_record(record: &mut Vec<u8>, terminator: u8) {
if record.last_byte() == Some(terminator) {
record.pop_byte();
}
}
fn trim_record_slice(mut record: &[u8], terminator: u8) -> &[u8] {
if record.last_byte() == Some(terminator) {
record = &record[..record.len() - 1];
}
record
}
#[cfg(all(test, feature = "std"))]
mod tests {
use alloc::{vec, vec::Vec};
use crate::bstring::BString;
use super::BufReadExt;
fn collect_lines<B: AsRef<[u8]>>(slice: B) -> Vec<BString> {
let mut lines = vec![];
slice
.as_ref()
.for_byte_line(|line| {
lines.push(BString::from(line.to_vec()));
Ok(true)
})
.unwrap();
lines
}
fn collect_lines_term<B: AsRef<[u8]>>(slice: B) -> Vec<BString> {
let mut lines = vec![];
slice
.as_ref()
.for_byte_line_with_terminator(|line| {
lines.push(BString::from(line.to_vec()));
Ok(true)
})
.unwrap();
lines
}
#[test]
fn lines_without_terminator() {
assert_eq!(collect_lines(""), Vec::<BString>::new());
assert_eq!(collect_lines("\n"), vec![""]);
assert_eq!(collect_lines("\n\n"), vec!["", ""]);
assert_eq!(collect_lines("a\nb\n"), vec!["a", "b"]);
assert_eq!(collect_lines("a\nb"), vec!["a", "b"]);
assert_eq!(collect_lines("abc\nxyz\n"), vec!["abc", "xyz"]);
assert_eq!(collect_lines("abc\nxyz"), vec!["abc", "xyz"]);
assert_eq!(collect_lines("\r\n"), vec![""]);
assert_eq!(collect_lines("\r\n\r\n"), vec!["", ""]);
assert_eq!(collect_lines("a\r\nb\r\n"), vec!["a", "b"]);
assert_eq!(collect_lines("a\r\nb"), vec!["a", "b"]);
assert_eq!(collect_lines("abc\r\nxyz\r\n"), vec!["abc", "xyz"]);
assert_eq!(collect_lines("abc\r\nxyz"), vec!["abc", "xyz"]);
assert_eq!(collect_lines("abc\rxyz"), vec!["abc\rxyz"]);
}
#[test]
fn lines_with_terminator() {
assert_eq!(collect_lines_term(""), Vec::<BString>::new());
assert_eq!(collect_lines_term("\n"), vec!["\n"]);
assert_eq!(collect_lines_term("\n\n"), vec!["\n", "\n"]);
assert_eq!(collect_lines_term("a\nb\n"), vec!["a\n", "b\n"]);
assert_eq!(collect_lines_term("a\nb"), vec!["a\n", "b"]);
assert_eq!(collect_lines_term("abc\nxyz\n"), vec!["abc\n", "xyz\n"]);
assert_eq!(collect_lines_term("abc\nxyz"), vec!["abc\n", "xyz"]);
assert_eq!(collect_lines_term("\r\n"), vec!["\r\n"]);
assert_eq!(collect_lines_term("\r\n\r\n"), vec!["\r\n", "\r\n"]);
assert_eq!(collect_lines_term("a\r\nb\r\n"), vec!["a\r\n", "b\r\n"]);
assert_eq!(collect_lines_term("a\r\nb"), vec!["a\r\n", "b"]);
assert_eq!(
collect_lines_term("abc\r\nxyz\r\n"),
vec!["abc\r\n", "xyz\r\n"]
);
assert_eq!(collect_lines_term("abc\r\nxyz"), vec!["abc\r\n", "xyz"]);
assert_eq!(collect_lines_term("abc\rxyz"), vec!["abc\rxyz"]);
}
}
+474
View File
@@ -0,0 +1,474 @@
/*!
A byte string library.
Byte strings are just like standard Unicode strings with one very important
difference: byte strings are only *conventionally* UTF-8 while Rust's standard
Unicode strings are *guaranteed* to be valid UTF-8. The primary motivation for
byte strings is for handling arbitrary bytes that are mostly UTF-8.
# Overview
This crate provides two important traits that provide string oriented methods
on `&[u8]` and `Vec<u8>` types:
* [`ByteSlice`](trait.ByteSlice.html) extends the `[u8]` type with additional
string oriented methods.
* [`ByteVec`](trait.ByteVec.html) extends the `Vec<u8>` type with additional
string oriented methods.
Additionally, this crate provides two concrete byte string types that deref to
`[u8]` and `Vec<u8>`. These are useful for storing byte string types, and come
with convenient `std::fmt::Debug` implementations:
* [`BStr`](struct.BStr.html) is a byte string slice, analogous to `str`.
* [`BString`](struct.BString.html) is an owned growable byte string buffer,
analogous to `String`.
Additionally, the free function [`B`](fn.B.html) serves as a convenient short
hand for writing byte string literals.
# Quick examples
Byte strings build on the existing APIs for `Vec<u8>` and `&[u8]`, with
additional string oriented methods. Operations such as iterating over
graphemes, searching for substrings, replacing substrings, trimming and case
conversion are examples of things not provided on the standard library `&[u8]`
APIs but are provided by this crate. For example, this code iterates over all
of occurrences of a substring:
```
use bstr::ByteSlice;
let s = b"foo bar foo foo quux foo";
let mut matches = vec![];
for start in s.find_iter("foo") {
matches.push(start);
}
assert_eq!(matches, [0, 8, 12, 21]);
```
Here's another example showing how to do a search and replace (and also showing
use of the `B` function):
```
# #[cfg(feature = "alloc")] {
use bstr::{B, ByteSlice};
let old = B("foo ☃☃☃ foo foo quux foo");
let new = old.replace("foo", "hello");
assert_eq!(new, B("hello ☃☃☃ hello hello quux hello"));
# }
```
And here's an example that shows case conversion, even in the presence of
invalid UTF-8:
```
# #[cfg(all(feature = "alloc", feature = "unicode"))] {
use bstr::{ByteSlice, ByteVec};
let mut lower = Vec::from("hello β");
lower[0] = b'\xFF';
// lowercase β is uppercased to Β
assert_eq!(lower.to_uppercase(), b"\xFFELLO \xCE\x92");
# }
```
# Convenient debug representation
When working with byte strings, it is often useful to be able to print them
as if they were byte strings and not sequences of integers. While this crate
cannot affect the `std::fmt::Debug` implementations for `[u8]` and `Vec<u8>`,
this crate does provide the `BStr` and `BString` types which have convenient
`std::fmt::Debug` implementations.
For example, this
```
use bstr::ByteSlice;
let mut bytes = Vec::from("hello β");
bytes[0] = b'\xFF';
println!("{:?}", bytes.as_bstr());
```
will output `"\xFFello β"`.
This example works because the
[`ByteSlice::as_bstr`](trait.ByteSlice.html#method.as_bstr)
method converts any `&[u8]` to a `&BStr`.
# When should I use byte strings?
This library reflects my belief that UTF-8 by convention is a better trade
off in some circumstances than guaranteed UTF-8.
The first time this idea hit me was in the implementation of Rust's regex
engine. In particular, very little of the internal implementation cares at all
about searching valid UTF-8 encoded strings. Indeed, internally, the
implementation converts `&str` from the API to `&[u8]` fairly quickly and
just deals with raw bytes. UTF-8 match boundaries are then guaranteed by the
finite state machine itself rather than any specific string type. This makes it
possible to not only run regexes on `&str` values, but also on `&[u8]` values.
Why would you ever want to run a regex on a `&[u8]` though? Well, `&[u8]` is
the fundamental way at which one reads data from all sorts of streams, via the
standard library's [`Read`](https://doc.rust-lang.org/std/io/trait.Read.html)
trait. In particular, there is no platform independent way to determine whether
what you're reading from is some binary file or a human readable text file.
Therefore, if you're writing a program to search files, you probably need to
deal with `&[u8]` directly unless you're okay with first converting it to a
`&str` and dropping any bytes that aren't valid UTF-8. (Or otherwise determine
the encoding---which is often impractical---and perform a transcoding step.)
Often, the simplest and most robust way to approach this is to simply treat the
contents of a file as if it were mostly valid UTF-8 and pass through invalid
UTF-8 untouched. This may not be the most correct approach though!
One case in particular exacerbates these issues, and that's memory mapping
a file. When you memory map a file, that file may be gigabytes big, but all
you get is a `&[u8]`. Converting that to a `&str` all in one go is generally
not a good idea because of the costs associated with doing so, and also
because it generally causes one to do two passes over the data instead of
one, which is quite undesirable. It is of course usually possible to do it an
incremental way by only parsing chunks at a time, but this is often complex to
do or impractical. For example, many regex engines only accept one contiguous
sequence of bytes at a time with no way to perform incremental matching.
# `bstr` in public APIs
This library is past version `1` and is expected to remain at version `1` for
the foreseeable future. Therefore, it is encouraged to put types from `bstr`
(like `BStr` and `BString`) in your public API if that makes sense for your
crate.
With that said, in general, it should be possible to avoid putting anything
in this crate into your public APIs. Namely, you should never need to use the
`ByteSlice` or `ByteVec` traits as bounds on public APIs, since their only
purpose is to extend the methods on the concrete types `[u8]` and `Vec<u8>`,
respectively. Similarly, it should not be necessary to put either the `BStr` or
`BString` types into public APIs. If you want to use them internally, then they
can be converted to/from `[u8]`/`Vec<u8>` as needed. The conversions are free.
So while it shouldn't ever be 100% necessary to make `bstr` a public
dependency, there may be cases where it is convenient to do so. This is an
explicitly supported use case of `bstr`, and as such, major version releases
should be exceptionally rare.
# Differences with standard strings
The primary difference between `[u8]` and `str` is that the former is
conventionally UTF-8 while the latter is guaranteed to be UTF-8. The phrase
"conventionally UTF-8" means that a `[u8]` may contain bytes that do not form
a valid UTF-8 sequence, but operations defined on the type in this crate are
generally most useful on valid UTF-8 sequences. For example, iterating over
Unicode codepoints or grapheme clusters is an operation that is only defined
on valid UTF-8. Therefore, when invalid UTF-8 is encountered, the Unicode
replacement codepoint is substituted. Thus, a byte string that is not UTF-8 at
all is of limited utility when using these crate.
However, not all operations on byte strings are specifically Unicode aware. For
example, substring search has no specific Unicode semantics ascribed to it. It
works just as well for byte strings that are completely valid UTF-8 as for byte
strings that contain no valid UTF-8 at all. Similarly for replacements and
various other operations that do not need any Unicode specific tailoring.
Aside from the difference in how UTF-8 is handled, the APIs between `[u8]` and
`str` (and `Vec<u8>` and `String`) are intentionally very similar, including
maintaining the same behavior for corner cases in things like substring
splitting. There are, however, some differences:
* Substring search is not done with `matches`, but instead, `find_iter`.
In general, this crate does not define any generic
[`Pattern`](https://doc.rust-lang.org/std/str/pattern/trait.Pattern.html)
infrastructure, and instead prefers adding new methods for different
argument types. For example, `matches` can search by a `char` or a `&str`,
where as `find_iter` can only search by a byte string. `find_char` can be
used for searching by a `char`.
* Since `SliceConcatExt` in the standard library is unstable, it is not
possible to reuse that to implement `join` and `concat` methods. Instead,
[`join`](fn.join.html) and [`concat`](fn.concat.html) are provided as free
functions that perform a similar task.
* This library bundles in a few more Unicode operations, such as grapheme,
word and sentence iterators. More operations, such as normalization and
case folding, may be provided in the future.
* Some `String`/`str` APIs will panic if a particular index was not on a valid
UTF-8 code unit sequence boundary. Conversely, no such checking is performed
in this crate, as is consistent with treating byte strings as a sequence of
bytes. This means callers are responsible for maintaining a UTF-8 invariant
if that's important.
* Some routines provided by this crate, such as `starts_with_str`, have a
`_str` suffix to differentiate them from similar routines already defined
on the `[u8]` type. The difference is that `starts_with` requires its
parameter to be a `&[u8]`, where as `starts_with_str` permits its parameter
to by anything that implements `AsRef<[u8]>`, which is more flexible. This
means you can write `bytes.starts_with_str("☃")` instead of
`bytes.starts_with("☃".as_bytes())`.
Otherwise, you should find most of the APIs between this crate and the standard
library string APIs to be very similar, if not identical.
# Handling of invalid UTF-8
Since byte strings are only *conventionally* UTF-8, there is no guarantee
that byte strings contain valid UTF-8. Indeed, it is perfectly legal for a
byte string to contain arbitrary bytes. However, since this library defines
a *string* type, it provides many operations specified by Unicode. These
operations are typically only defined over codepoints, and thus have no real
meaning on bytes that are invalid UTF-8 because they do not map to a particular
codepoint.
For this reason, whenever operations defined only on codepoints are used, this
library will automatically convert invalid UTF-8 to the Unicode replacement
codepoint, `U+FFFD`, which looks like this: `�`. For example, an
[iterator over codepoints](struct.Chars.html) will yield a Unicode
replacement codepoint whenever it comes across bytes that are not valid UTF-8:
```
use bstr::ByteSlice;
let bs = b"a\xFF\xFFz";
let chars: Vec<char> = bs.chars().collect();
assert_eq!(vec!['a', '\u{FFFD}', '\u{FFFD}', 'z'], chars);
```
There are a few ways in which invalid bytes can be substituted with a Unicode
replacement codepoint. One way, not used by this crate, is to replace every
individual invalid byte with a single replacement codepoint. In contrast, the
approach this crate uses is called the "substitution of maximal subparts," as
specified by the Unicode Standard (Chapter 3, Section 9). (This approach is
also used by [W3C's Encoding Standard](https://www.w3.org/TR/encoding/).) In
this strategy, a replacement codepoint is inserted whenever a byte is found
that cannot possibly lead to a valid UTF-8 code unit sequence. If there were
previous bytes that represented a *prefix* of a well-formed UTF-8 code unit
sequence, then all of those bytes (up to 3) are substituted with a single
replacement codepoint. For example:
```
use bstr::ByteSlice;
let bs = b"a\xF0\x9F\x87z";
let chars: Vec<char> = bs.chars().collect();
// The bytes \xF0\x9F\x87 could lead to a valid UTF-8 sequence, but 3 of them
// on their own are invalid. Only one replacement codepoint is substituted,
// which demonstrates the "substitution of maximal subparts" strategy.
assert_eq!(vec!['a', '\u{FFFD}', 'z'], chars);
```
If you do need to access the raw bytes for some reason in an iterator like
`Chars`, then you should use the iterator's "indices" variant, which gives
the byte offsets containing the invalid UTF-8 bytes that were substituted with
the replacement codepoint. For example:
```
use bstr::{B, ByteSlice};
let bs = b"a\xE2\x98z";
let chars: Vec<(usize, usize, char)> = bs.char_indices().collect();
// Even though the replacement codepoint is encoded as 3 bytes itself, the
// byte range given here is only two bytes, corresponding to the original
// raw bytes.
assert_eq!(vec![(0, 1, 'a'), (1, 3, '\u{FFFD}'), (3, 4, 'z')], chars);
// Thus, getting the original raw bytes is as simple as slicing the original
// byte string:
let chars: Vec<&[u8]> = bs.char_indices().map(|(s, e, _)| &bs[s..e]).collect();
assert_eq!(vec![B("a"), B(b"\xE2\x98"), B("z")], chars);
```
# File paths and OS strings
One of the premiere features of Rust's standard library is how it handles file
paths. In particular, it makes it very hard to write incorrect code while
simultaneously providing a correct cross platform abstraction for manipulating
file paths. The key challenge that one faces with file paths across platforms
is derived from the following observations:
* On most Unix-like systems, file paths are an arbitrary sequence of bytes.
* On Windows, file paths are an arbitrary sequence of 16-bit integers.
(In both cases, certain sequences aren't allowed. For example a `NUL` byte is
not allowed in either case. But we can ignore this for the purposes of this
section.)
Byte strings, like the ones provided in this crate, line up really well with
file paths on Unix like systems, which are themselves just arbitrary sequences
of bytes. It turns out that if you treat them as "mostly UTF-8," then things
work out pretty well. On the contrary, byte strings _don't_ really work
that well on Windows because it's not possible to correctly roundtrip file
paths between 16-bit integers and something that looks like UTF-8 _without_
explicitly defining an encoding to do this for you, which is anathema to byte
strings, which are just bytes.
Rust's standard library elegantly solves this problem by specifying an
internal encoding for file paths that's only used on Windows called
[WTF-8](https://simonsapin.github.io/wtf-8/). Its key properties are that they
permit losslessly roundtripping file paths on Windows by extending UTF-8 to
support an encoding of surrogate codepoints, while simultaneously supporting
zero-cost conversion from Rust's Unicode strings to file paths. (Since UTF-8 is
a proper subset of WTF-8.)
The fundamental point at which the above strategy fails is when you want to
treat file paths as things that look like strings in a zero cost way. In most
cases, this is actually the wrong thing to do, but some cases call for it,
for example, glob or regex matching on file paths. This is because WTF-8 is
treated as an internal implementation detail, and there is no way to access
those bytes via a public API. Therefore, such consumers are limited in what
they can do:
1. One could re-implement WTF-8 and re-encode file paths on Windows to WTF-8
by accessing their underlying 16-bit integer representation. Unfortunately,
this isn't zero cost (it introduces a second WTF-8 decoding step) and it's
not clear this is a good thing to do, since WTF-8 should ideally remain an
internal implementation detail. This is roughly the approach taken by the
[`os_str_bytes`](https://crates.io/crates/os_str_bytes) crate.
2. One could instead declare that they will not handle paths on Windows that
are not valid UTF-16, and return an error when one is encountered.
3. Like (2), but instead of returning an error, lossily decode the file path
on Windows that isn't valid UTF-16 into UTF-16 by replacing invalid bytes
with the Unicode replacement codepoint.
While this library may provide facilities for (1) in the future, currently,
this library only provides facilities for (2) and (3). In particular, a suite
of conversion functions are provided that permit converting between byte
strings, OS strings and file paths. For owned byte strings, they are:
* [`ByteVec::from_os_string`](trait.ByteVec.html#method.from_os_string)
* [`ByteVec::from_os_str_lossy`](trait.ByteVec.html#method.from_os_str_lossy)
* [`ByteVec::from_path_buf`](trait.ByteVec.html#method.from_path_buf)
* [`ByteVec::from_path_lossy`](trait.ByteVec.html#method.from_path_lossy)
* [`ByteVec::into_os_string`](trait.ByteVec.html#method.into_os_string)
* [`ByteVec::into_os_string_lossy`](trait.ByteVec.html#method.into_os_string_lossy)
* [`ByteVec::into_path_buf`](trait.ByteVec.html#method.into_path_buf)
* [`ByteVec::into_path_buf_lossy`](trait.ByteVec.html#method.into_path_buf_lossy)
For byte string slices, they are:
* [`ByteSlice::from_os_str`](trait.ByteSlice.html#method.from_os_str)
* [`ByteSlice::from_path`](trait.ByteSlice.html#method.from_path)
* [`ByteSlice::to_os_str`](trait.ByteSlice.html#method.to_os_str)
* [`ByteSlice::to_os_str_lossy`](trait.ByteSlice.html#method.to_os_str_lossy)
* [`ByteSlice::to_path`](trait.ByteSlice.html#method.to_path)
* [`ByteSlice::to_path_lossy`](trait.ByteSlice.html#method.to_path_lossy)
On Unix, all of these conversions are rigorously zero cost, which gives one
a way to ergonomically deal with raw file paths exactly as they are using
normal string-related functions. On Windows, these conversion routines perform
a UTF-8 check and either return an error or lossily decode the file path
into valid UTF-8, depending on which function you use. This means that you
cannot roundtrip all file paths on Windows correctly using these conversion
routines. However, this may be an acceptable downside since such file paths
are exceptionally rare. Moreover, roundtripping isn't always necessary, for
example, if all you're doing is filtering based on file paths.
The reason why using byte strings for this is potentially superior than the
standard library's approach is that a lot of Rust code is already lossily
converting file paths to Rust's Unicode strings, which are required to be valid
UTF-8, and thus contain latent bugs on Unix where paths with invalid UTF-8 are
not terribly uncommon. If you instead use byte strings, then you're guaranteed
to write correct code for Unix, at the cost of getting a corner case wrong on
Windows.
# Cargo features
This crates comes with a few features that control standard library, serde
and Unicode support.
* `std` - **Enabled** by default. This provides APIs that require the standard
library, such as `Vec<u8>` and `PathBuf`. Enabling this feature also enables
the `alloc` feature and any other relevant `std` features for dependencies.
* `alloc` - **Enabled** by default. This provides APIs that require allocations
via the `alloc` crate, such as `Vec<u8>`.
* `unicode` - **Enabled** by default. This provides APIs that require sizable
Unicode data compiled into the binary. This includes, but is not limited to,
grapheme/word/sentence segmenters. When this is disabled, basic support such
as UTF-8 decoding is still included. Note that currently, enabling this
feature also requires enabling the `std` feature. It is expected that this
limitation will be lifted at some point.
* `serde` - Enables implementations of serde traits for `BStr`, and also
`BString` when `alloc` is enabled.
*/
// #![cfg_attr(not(any(feature = "std", test)), no_std)]
#![no_std]
#![cfg_attr(docsrs, feature(doc_cfg))]
#[cfg(any(test, feature = "std"))]
extern crate std;
#[cfg(any(test, feature = "alloc"))]
extern crate alloc;
pub use crate::bstr::BStr;
#[cfg(feature = "alloc")]
pub use crate::bstring::BString;
pub use crate::escape_bytes::EscapeBytes;
#[cfg(feature = "unicode")]
pub use crate::ext_slice::Fields;
pub use crate::ext_slice::{
ByteSlice, Bytes, FieldsWith, Find, FindReverse, Finder, FinderReverse,
Lines, LinesWithTerminator, Split, SplitN, SplitNReverse, SplitReverse, B,
};
#[cfg(feature = "alloc")]
pub use crate::ext_vec::{concat, join, ByteVec, DrainBytes, FromUtf8Error};
#[cfg(feature = "unicode")]
pub use crate::unicode::{
GraphemeIndices, Graphemes, SentenceIndices, Sentences, WordIndices,
Words, WordsWithBreakIndices, WordsWithBreaks,
};
pub use crate::utf8::{
decode as decode_utf8, decode_last as decode_last_utf8, CharIndices,
Chars, Utf8Chunk, Utf8Chunks, Utf8Error,
};
mod ascii;
mod bstr;
#[cfg(feature = "alloc")]
mod bstring;
mod byteset;
mod escape_bytes;
mod ext_slice;
#[cfg(feature = "alloc")]
mod ext_vec;
mod impls;
#[cfg(feature = "std")]
pub mod io;
#[cfg(all(test, feature = "std"))]
mod tests;
#[cfg(feature = "unicode")]
mod unicode;
mod utf8;
#[cfg(all(test, feature = "std"))]
mod apitests {
use crate::{
bstr::BStr,
bstring::BString,
ext_slice::{Finder, FinderReverse},
};
#[test]
fn oibits() {
use std::panic::{RefUnwindSafe, UnwindSafe};
fn assert_send<T: Send>() {}
fn assert_sync<T: Sync>() {}
fn assert_unwind_safe<T: RefUnwindSafe + UnwindSafe>() {}
assert_send::<&BStr>();
assert_sync::<&BStr>();
assert_unwind_safe::<&BStr>();
assert_send::<BString>();
assert_sync::<BString>();
assert_unwind_safe::<BString>();
assert_send::<Finder<'_>>();
assert_sync::<Finder<'_>>();
assert_unwind_safe::<Finder<'_>>();
assert_send::<FinderReverse<'_>>();
assert_sync::<FinderReverse<'_>>();
assert_unwind_safe::<FinderReverse<'_>>();
}
}
+32
View File
@@ -0,0 +1,32 @@
/// A sequence of tests for checking whether lossy decoding uses the maximal
/// subpart strategy correctly. Namely, if a sequence of otherwise invalid
/// UTF-8 bytes is a valid prefix of a valid UTF-8 sequence, then the entire
/// prefix is replaced by a single replacement codepoint. In all other cases,
/// each invalid byte is replaced by a single replacement codepoint.
///
/// The first element in each tuple is the expected result of lossy decoding,
/// while the second element is the input given.
pub(crate) const LOSSY_TESTS: &[(&str, &[u8])] = &[
("a", b"a"),
("\u{FFFD}", b"\xFF"),
("\u{FFFD}\u{FFFD}", b"\xFF\xFF"),
("β\u{FFFD}", b"\xCE\xB2\xFF"),
("☃\u{FFFD}", b"\xE2\x98\x83\xFF"),
("𝝱\u{FFFD}", b"\xF0\x9D\x9D\xB1\xFF"),
("\u{FFFD}\u{FFFD}", b"\xCE\xF0"),
("\u{FFFD}\u{FFFD}", b"\xCE\xFF"),
("\u{FFFD}\u{FFFD}", b"\xE2\x98\xF0"),
("\u{FFFD}\u{FFFD}", b"\xE2\x98\xFF"),
("\u{FFFD}", b"\xF0\x9D\x9D"),
("\u{FFFD}\u{FFFD}", b"\xF0\x9D\x9D\xF0"),
("\u{FFFD}\u{FFFD}", b"\xF0\x9D\x9D\xFF"),
("\u{FFFD}", b"\xCE"),
("a\u{FFFD}", b"a\xCE"),
("\u{FFFD}", b"\xE2\x98"),
("a\u{FFFD}", b"a\xE2\x98"),
("\u{FFFD}", b"\xF0\x9D\x9C"),
("a\u{FFFD}", b"a\xF0\x9D\x9C"),
("a\u{FFFD}\u{FFFD}\u{FFFD}z", b"a\xED\xA0\x80z"),
("☃βツ\u{FFFD}", b"\xe2\x98\x83\xce\xb2\xe3\x83\x84\xFF"),
("a\u{FFFD}\u{FFFD}\u{FFFD}b", b"\x61\xF1\x80\x80\xE1\x80\xC2\x62"),
];
@@ -0,0 +1,19 @@
// DO NOT EDIT THIS FILE. IT WAS AUTOMATICALLY GENERATED BY:
//
// regex-cli generate serialize sparse dfa --minimize --start-kind anchored --shrink --rustfmt --safe GRAPHEME_BREAK_FWD src/unicode/fsm/ <snip: arg too long>
//
// regex-cli 0.0.1 is available on crates.io.
use regex_automata::{dfa::sparse::DFA, util::lazy::Lazy};
pub static GRAPHEME_BREAK_FWD: Lazy<DFA<&'static [u8]>> = Lazy::new(|| {
#[cfg(target_endian = "big")]
static BYTES: &'static [u8] =
include_bytes!("grapheme_break_fwd.bigendian.dfa");
#[cfg(target_endian = "little")]
static BYTES: &'static [u8] =
include_bytes!("grapheme_break_fwd.littleendian.dfa");
let (dfa, _) =
DFA::from_bytes(BYTES).expect("serialized DFA should be valid");
dfa
});
@@ -0,0 +1,19 @@
// DO NOT EDIT THIS FILE. IT WAS AUTOMATICALLY GENERATED BY:
//
// regex-cli generate serialize sparse dfa --minimize --start-kind anchored --reverse --match-kind all --no-captures --shrink --rustfmt --safe GRAPHEME_BREAK_REV src/unicode/fsm/ <snip: arg too long>
//
// regex-cli 0.0.1 is available on crates.io.
use regex_automata::{dfa::sparse::DFA, util::lazy::Lazy};
pub static GRAPHEME_BREAK_REV: Lazy<DFA<&'static [u8]>> = Lazy::new(|| {
#[cfg(target_endian = "big")]
static BYTES: &'static [u8] =
include_bytes!("grapheme_break_rev.bigendian.dfa");
#[cfg(target_endian = "little")]
static BYTES: &'static [u8] =
include_bytes!("grapheme_break_rev.littleendian.dfa");
let (dfa, _) =
DFA::from_bytes(BYTES).expect("serialized DFA should be valid");
dfa
});
+8
View File
@@ -0,0 +1,8 @@
pub mod grapheme_break_fwd;
pub mod grapheme_break_rev;
pub mod regional_indicator_rev;
pub mod sentence_break_fwd;
pub mod simple_word_fwd;
pub mod whitespace_anchored_fwd;
pub mod whitespace_anchored_rev;
pub mod word_break_fwd;
@@ -0,0 +1,24 @@
// DO NOT EDIT THIS FILE. IT WAS AUTOMATICALLY GENERATED BY:
//
// regex-cli generate serialize dense dfa --minimize --start-kind anchored --reverse --no-captures --shrink --rustfmt --safe REGIONAL_INDICATOR_REV src/unicode/fsm/ \p{gcb=Regional_Indicator}
//
// regex-cli 0.0.1 is available on crates.io.
use regex_automata::{
dfa::dense::DFA,
util::{lazy::Lazy, wire::AlignAs},
};
pub static REGIONAL_INDICATOR_REV: Lazy<DFA<&'static [u32]>> =
Lazy::new(|| {
static ALIGNED: &AlignAs<[u8], u32> = &AlignAs {
_align: [],
#[cfg(target_endian = "big")]
bytes: *include_bytes!("regional_indicator_rev.bigendian.dfa"),
#[cfg(target_endian = "little")]
bytes: *include_bytes!("regional_indicator_rev.littleendian.dfa"),
};
let (dfa, _) = DFA::from_bytes(&ALIGNED.bytes)
.expect("serialized DFA should be valid");
dfa
});
@@ -0,0 +1,19 @@
// DO NOT EDIT THIS FILE. IT WAS AUTOMATICALLY GENERATED BY:
//
// regex-cli generate serialize sparse dfa --minimize --start-kind anchored --shrink --rustfmt --safe SENTENCE_BREAK_FWD src/unicode/fsm/ <snip: arg too long>
//
// regex-cli 0.0.1 is available on crates.io.
use regex_automata::{dfa::sparse::DFA, util::lazy::Lazy};
pub static SENTENCE_BREAK_FWD: Lazy<DFA<&'static [u8]>> = Lazy::new(|| {
#[cfg(target_endian = "big")]
static BYTES: &'static [u8] =
include_bytes!("sentence_break_fwd.bigendian.dfa");
#[cfg(target_endian = "little")]
static BYTES: &'static [u8] =
include_bytes!("sentence_break_fwd.littleendian.dfa");
let (dfa, _) =
DFA::from_bytes(BYTES).expect("serialized DFA should be valid");
dfa
});
@@ -0,0 +1,19 @@
// DO NOT EDIT THIS FILE. IT WAS AUTOMATICALLY GENERATED BY:
//
// regex-cli generate serialize sparse dfa --minimize --start-kind anchored --shrink --rustfmt --safe SIMPLE_WORD_FWD src/unicode/fsm/ \w
//
// regex-cli 0.0.1 is available on crates.io.
use regex_automata::{dfa::sparse::DFA, util::lazy::Lazy};
pub static SIMPLE_WORD_FWD: Lazy<DFA<&'static [u8]>> = Lazy::new(|| {
#[cfg(target_endian = "big")]
static BYTES: &'static [u8] =
include_bytes!("simple_word_fwd.bigendian.dfa");
#[cfg(target_endian = "little")]
static BYTES: &'static [u8] =
include_bytes!("simple_word_fwd.littleendian.dfa");
let (dfa, _) =
DFA::from_bytes(BYTES).expect("serialized DFA should be valid");
dfa
});
@@ -0,0 +1,24 @@
// DO NOT EDIT THIS FILE. IT WAS AUTOMATICALLY GENERATED BY:
//
// regex-cli generate serialize dense dfa --minimize --start-kind anchored --shrink --rustfmt --safe WHITESPACE_ANCHORED_FWD src/unicode/fsm/ \s+
//
// regex-cli 0.0.1 is available on crates.io.
use regex_automata::{
dfa::dense::DFA,
util::{lazy::Lazy, wire::AlignAs},
};
pub static WHITESPACE_ANCHORED_FWD: Lazy<DFA<&'static [u32]>> =
Lazy::new(|| {
static ALIGNED: &AlignAs<[u8], u32> = &AlignAs {
_align: [],
#[cfg(target_endian = "big")]
bytes: *include_bytes!("whitespace_anchored_fwd.bigendian.dfa"),
#[cfg(target_endian = "little")]
bytes: *include_bytes!("whitespace_anchored_fwd.littleendian.dfa"),
};
let (dfa, _) = DFA::from_bytes(&ALIGNED.bytes)
.expect("serialized DFA should be valid");
dfa
});
@@ -0,0 +1,24 @@
// DO NOT EDIT THIS FILE. IT WAS AUTOMATICALLY GENERATED BY:
//
// regex-cli generate serialize dense dfa --minimize --start-kind anchored --reverse --no-captures --shrink --rustfmt --safe WHITESPACE_ANCHORED_REV src/unicode/fsm/ \s+
//
// regex-cli 0.0.1 is available on crates.io.
use regex_automata::{
dfa::dense::DFA,
util::{lazy::Lazy, wire::AlignAs},
};
pub static WHITESPACE_ANCHORED_REV: Lazy<DFA<&'static [u32]>> =
Lazy::new(|| {
static ALIGNED: &AlignAs<[u8], u32> = &AlignAs {
_align: [],
#[cfg(target_endian = "big")]
bytes: *include_bytes!("whitespace_anchored_rev.bigendian.dfa"),
#[cfg(target_endian = "little")]
bytes: *include_bytes!("whitespace_anchored_rev.littleendian.dfa"),
};
let (dfa, _) = DFA::from_bytes(&ALIGNED.bytes)
.expect("serialized DFA should be valid");
dfa
});
@@ -0,0 +1,19 @@
// DO NOT EDIT THIS FILE. IT WAS AUTOMATICALLY GENERATED BY:
//
// regex-cli generate serialize sparse dfa --minimize --start-kind anchored --shrink --rustfmt --safe WORD_BREAK_FWD src/unicode/fsm/ <snip: arg too long>
//
// regex-cli 0.0.1 is available on crates.io.
use regex_automata::{dfa::sparse::DFA, util::lazy::Lazy};
pub static WORD_BREAK_FWD: Lazy<DFA<&'static [u8]>> = Lazy::new(|| {
#[cfg(target_endian = "big")]
static BYTES: &'static [u8] =
include_bytes!("word_break_fwd.bigendian.dfa");
#[cfg(target_endian = "little")]
static BYTES: &'static [u8] =
include_bytes!("word_break_fwd.littleendian.dfa");
let (dfa, _) =
DFA::from_bytes(BYTES).expect("serialized DFA should be valid");
dfa
});
+395
View File
@@ -0,0 +1,395 @@
use regex_automata::{dfa::Automaton, Anchored, Input};
use crate::{
ext_slice::ByteSlice,
unicode::fsm::{
grapheme_break_fwd::GRAPHEME_BREAK_FWD,
grapheme_break_rev::GRAPHEME_BREAK_REV,
regional_indicator_rev::REGIONAL_INDICATOR_REV,
},
utf8,
};
/// An iterator over grapheme clusters in a byte string.
///
/// This iterator is typically constructed by
/// [`ByteSlice::graphemes`](trait.ByteSlice.html#method.graphemes).
///
/// Unicode defines a grapheme cluster as an *approximation* to a single user
/// visible character. A grapheme cluster, or just "grapheme," is made up of
/// one or more codepoints. For end user oriented tasks, one should generally
/// prefer using graphemes instead of [`Chars`](struct.Chars.html), which
/// always yields one codepoint at a time.
///
/// Since graphemes are made up of one or more codepoints, this iterator yields
/// `&str` elements. When invalid UTF-8 is encountered, replacement codepoints
/// are [substituted](index.html#handling-of-invalid-utf-8).
///
/// This iterator can be used in reverse. When reversed, exactly the same
/// set of grapheme clusters are yielded, but in reverse order.
///
/// This iterator only yields *extended* grapheme clusters, in accordance with
/// [UAX #29](https://www.unicode.org/reports/tr29/tr29-33.html#Grapheme_Cluster_Boundaries).
#[derive(Clone, Debug)]
pub struct Graphemes<'a> {
bs: &'a [u8],
}
impl<'a> Graphemes<'a> {
pub(crate) fn new(bs: &'a [u8]) -> Graphemes<'a> {
Graphemes { bs }
}
/// View the underlying data as a subslice of the original data.
///
/// The slice returned has the same lifetime as the original slice, and so
/// the iterator can continue to be used while this exists.
///
/// # Examples
///
/// ```
/// use bstr::ByteSlice;
///
/// let mut it = b"abc".graphemes();
///
/// assert_eq!(b"abc", it.as_bytes());
/// it.next();
/// assert_eq!(b"bc", it.as_bytes());
/// it.next();
/// it.next();
/// assert_eq!(b"", it.as_bytes());
/// ```
#[inline]
pub fn as_bytes(&self) -> &'a [u8] {
self.bs
}
}
impl<'a> Iterator for Graphemes<'a> {
type Item = &'a str;
#[inline]
fn next(&mut self) -> Option<&'a str> {
let (grapheme, size) = decode_grapheme(self.bs);
if size == 0 {
return None;
}
self.bs = &self.bs[size..];
Some(grapheme)
}
}
impl<'a> DoubleEndedIterator for Graphemes<'a> {
#[inline]
fn next_back(&mut self) -> Option<&'a str> {
let (grapheme, size) = decode_last_grapheme(self.bs);
if size == 0 {
return None;
}
self.bs = &self.bs[..self.bs.len() - size];
Some(grapheme)
}
}
/// An iterator over grapheme clusters in a byte string and their byte index
/// positions.
///
/// This iterator is typically constructed by
/// [`ByteSlice::grapheme_indices`](trait.ByteSlice.html#method.grapheme_indices).
///
/// Unicode defines a grapheme cluster as an *approximation* to a single user
/// visible character. A grapheme cluster, or just "grapheme," is made up of
/// one or more codepoints. For end user oriented tasks, one should generally
/// prefer using graphemes instead of [`Chars`](struct.Chars.html), which
/// always yields one codepoint at a time.
///
/// Since graphemes are made up of one or more codepoints, this iterator
/// yields `&str` elements (along with their start and end byte offsets).
/// When invalid UTF-8 is encountered, replacement codepoints are
/// [substituted](index.html#handling-of-invalid-utf-8). Because of this, the
/// indices yielded by this iterator may not correspond to the length of the
/// grapheme cluster yielded with those indices. For example, when this
/// iterator encounters `\xFF` in the byte string, then it will yield a pair
/// of indices ranging over a single byte, but will provide an `&str`
/// equivalent to `"\u{FFFD}"`, which is three bytes in length. However, when
/// given only valid UTF-8, then all indices are in exact correspondence with
/// their paired grapheme cluster.
///
/// This iterator can be used in reverse. When reversed, exactly the same
/// set of grapheme clusters are yielded, but in reverse order.
///
/// This iterator only yields *extended* grapheme clusters, in accordance with
/// [UAX #29](https://www.unicode.org/reports/tr29/tr29-33.html#Grapheme_Cluster_Boundaries).
#[derive(Clone, Debug)]
pub struct GraphemeIndices<'a> {
bs: &'a [u8],
forward_index: usize,
reverse_index: usize,
}
impl<'a> GraphemeIndices<'a> {
pub(crate) fn new(bs: &'a [u8]) -> GraphemeIndices<'a> {
GraphemeIndices { bs, forward_index: 0, reverse_index: bs.len() }
}
/// View the underlying data as a subslice of the original data.
///
/// The slice returned has the same lifetime as the original slice, and so
/// the iterator can continue to be used while this exists.
///
/// # Examples
///
/// ```
/// use bstr::ByteSlice;
///
/// let mut it = b"abc".grapheme_indices();
///
/// assert_eq!(b"abc", it.as_bytes());
/// it.next();
/// assert_eq!(b"bc", it.as_bytes());
/// it.next();
/// it.next();
/// assert_eq!(b"", it.as_bytes());
/// ```
#[inline]
pub fn as_bytes(&self) -> &'a [u8] {
self.bs
}
}
impl<'a> Iterator for GraphemeIndices<'a> {
type Item = (usize, usize, &'a str);
#[inline]
fn next(&mut self) -> Option<(usize, usize, &'a str)> {
let index = self.forward_index;
let (grapheme, size) = decode_grapheme(self.bs);
if size == 0 {
return None;
}
self.bs = &self.bs[size..];
self.forward_index += size;
Some((index, index + size, grapheme))
}
}
impl<'a> DoubleEndedIterator for GraphemeIndices<'a> {
#[inline]
fn next_back(&mut self) -> Option<(usize, usize, &'a str)> {
let (grapheme, size) = decode_last_grapheme(self.bs);
if size == 0 {
return None;
}
self.bs = &self.bs[..self.bs.len() - size];
self.reverse_index -= size;
Some((self.reverse_index, self.reverse_index + size, grapheme))
}
}
/// Decode a grapheme from the given byte string.
///
/// This returns the resulting grapheme (which may be a Unicode replacement
/// codepoint if invalid UTF-8 was found), along with the number of bytes
/// decoded in the byte string. The number of bytes decoded may not be the
/// same as the length of grapheme in the case where invalid UTF-8 is found.
pub fn decode_grapheme(bs: &[u8]) -> (&str, usize) {
if bs.is_empty() {
("", 0)
} else if bs.len() >= 2
&& bs[0].is_ascii()
&& bs[1].is_ascii()
&& !bs[0].is_ascii_whitespace()
{
// FIXME: It is somewhat sad that we have to special case this, but it
// leads to a significant speed up in predominantly ASCII text. The
// issue here is that the DFA has a bit of overhead, and running it for
// every byte in mostly ASCII text results in a bit slowdown. We should
// re-litigate this once regex-automata 0.3 is out, but it might be
// hard to avoid the special case. A DFA is always going to at least
// require some memory access.
// Safe because all ASCII bytes are valid UTF-8.
let grapheme = unsafe { bs[..1].to_str_unchecked() };
(grapheme, 1)
} else if let Some(hm) = {
let input = Input::new(bs).anchored(Anchored::Yes);
GRAPHEME_BREAK_FWD.try_search_fwd(&input).unwrap()
} {
// Safe because a match can only occur for valid UTF-8.
let grapheme = unsafe { bs[..hm.offset()].to_str_unchecked() };
(grapheme, grapheme.len())
} else {
const INVALID: &str = "\u{FFFD}";
// No match on non-empty bytes implies we found invalid UTF-8.
let (_, size) = utf8::decode_lossy(bs);
(INVALID, size)
}
}
fn decode_last_grapheme(bs: &[u8]) -> (&str, usize) {
if bs.is_empty() {
("", 0)
} else if let Some(hm) = {
let input = Input::new(bs).anchored(Anchored::Yes);
GRAPHEME_BREAK_REV.try_search_rev(&input).unwrap()
} {
let start = adjust_rev_for_regional_indicator(bs, hm.offset());
// Safe because a match can only occur for valid UTF-8.
let grapheme = unsafe { bs[start..].to_str_unchecked() };
(grapheme, grapheme.len())
} else {
const INVALID: &str = "\u{FFFD}";
// No match on non-empty bytes implies we found invalid UTF-8.
let (_, size) = utf8::decode_last_lossy(bs);
(INVALID, size)
}
}
/// Return the correct offset for the next grapheme decoded at the end of the
/// given byte string, where `i` is the initial guess. In particular,
/// `&bs[i..]` represents the candidate grapheme.
///
/// `i` is returned by this function in all cases except when `&bs[i..]` is
/// a pair of regional indicator codepoints. In that case, if an odd number of
/// additional regional indicator codepoints precedes `i`, then `i` is
/// adjusted such that it points to only a single regional indicator.
///
/// This "fixing" is necessary to handle the requirement that a break cannot
/// occur between regional indicators where it would cause an odd number of
/// regional indicators to exist before the break from the *start* of the
/// string. A reverse regex cannot detect this case easily without look-around.
fn adjust_rev_for_regional_indicator(mut bs: &[u8], i: usize) -> usize {
// All regional indicators use a 4 byte encoding, and we only care about
// the case where we found a pair of regional indicators.
if bs.len() - i != 8 {
return i;
}
// Count all contiguous occurrences of regional indicators. If there's an
// even number of them, then we can accept the pair we found. Otherwise,
// we can only take one of them.
//
// FIXME: This is quadratic in the worst case, e.g., a string of just
// regional indicator codepoints. A fix probably requires refactoring this
// code a bit such that we don't rescan regional indicators.
let mut count = 0;
while let Some(hm) = {
let input = Input::new(bs).anchored(Anchored::Yes);
REGIONAL_INDICATOR_REV.try_search_rev(&input).unwrap()
} {
bs = &bs[..hm.offset()];
count += 1;
}
if count % 2 == 0 {
i
} else {
i + 4
}
}
#[cfg(all(test, feature = "std"))]
mod tests {
use alloc::{
string::{String, ToString},
vec,
vec::Vec,
};
#[cfg(not(miri))]
use ucd_parse::GraphemeClusterBreakTest;
use crate::tests::LOSSY_TESTS;
use super::*;
#[test]
#[cfg(not(miri))]
fn forward_ucd() {
for (i, test) in ucdtests().into_iter().enumerate() {
let given = test.grapheme_clusters.concat();
let got: Vec<String> = Graphemes::new(given.as_bytes())
.map(|cluster| cluster.to_string())
.collect();
assert_eq!(
test.grapheme_clusters,
got,
"\ngrapheme forward break test {} failed:\n\
given: {:?}\n\
expected: {:?}\n\
got: {:?}\n",
i,
uniescape(&given),
uniescape_vec(&test.grapheme_clusters),
uniescape_vec(&got),
);
}
}
#[test]
#[cfg(not(miri))]
fn reverse_ucd() {
for (i, test) in ucdtests().into_iter().enumerate() {
let given = test.grapheme_clusters.concat();
let mut got: Vec<String> = Graphemes::new(given.as_bytes())
.rev()
.map(|cluster| cluster.to_string())
.collect();
got.reverse();
assert_eq!(
test.grapheme_clusters,
got,
"\n\ngrapheme reverse break test {} failed:\n\
given: {:?}\n\
expected: {:?}\n\
got: {:?}\n",
i,
uniescape(&given),
uniescape_vec(&test.grapheme_clusters),
uniescape_vec(&got),
);
}
}
#[test]
fn forward_lossy() {
for &(expected, input) in LOSSY_TESTS {
let got = Graphemes::new(input.as_bytes()).collect::<String>();
assert_eq!(expected, got);
}
}
#[test]
fn reverse_lossy() {
for &(expected, input) in LOSSY_TESTS {
let expected: String = expected.chars().rev().collect();
let got =
Graphemes::new(input.as_bytes()).rev().collect::<String>();
assert_eq!(expected, got);
}
}
#[cfg(not(miri))]
fn uniescape(s: &str) -> String {
s.chars().flat_map(|c| c.escape_unicode()).collect::<String>()
}
#[cfg(not(miri))]
fn uniescape_vec(strs: &[String]) -> Vec<String> {
strs.iter().map(|s| uniescape(s)).collect()
}
/// Return all of the UCD for grapheme breaks.
#[cfg(not(miri))]
fn ucdtests() -> Vec<GraphemeClusterBreakTest> {
const TESTDATA: &str = include_str!("data/GraphemeBreakTest.txt");
let mut tests = vec![];
for mut line in TESTDATA.lines() {
line = line.trim();
if line.starts_with("#") || line.contains("surrogate") {
continue;
}
tests.push(line.parse().unwrap());
}
tests
}
}
+12
View File
@@ -0,0 +1,12 @@
pub use self::{
grapheme::{decode_grapheme, GraphemeIndices, Graphemes},
sentence::{SentenceIndices, Sentences},
whitespace::{whitespace_len_fwd, whitespace_len_rev},
word::{WordIndices, Words, WordsWithBreakIndices, WordsWithBreaks},
};
mod fsm;
mod grapheme;
mod sentence;
mod whitespace;
mod word;
+229
View File
@@ -0,0 +1,229 @@
use regex_automata::{dfa::Automaton, Anchored, Input};
use crate::{
ext_slice::ByteSlice,
unicode::fsm::sentence_break_fwd::SENTENCE_BREAK_FWD, utf8,
};
/// An iterator over sentences in a byte string.
///
/// This iterator is typically constructed by
/// [`ByteSlice::sentences`](trait.ByteSlice.html#method.sentences).
///
/// Sentences typically include their trailing punctuation and whitespace.
///
/// Since sentences are made up of one or more codepoints, this iterator yields
/// `&str` elements. When invalid UTF-8 is encountered, replacement codepoints
/// are [substituted](index.html#handling-of-invalid-utf-8).
///
/// This iterator yields words in accordance with the default sentence boundary
/// rules specified in
/// [UAX #29](https://www.unicode.org/reports/tr29/tr29-33.html#Sentence_Boundaries).
#[derive(Clone, Debug)]
pub struct Sentences<'a> {
bs: &'a [u8],
}
impl<'a> Sentences<'a> {
pub(crate) fn new(bs: &'a [u8]) -> Sentences<'a> {
Sentences { bs }
}
/// View the underlying data as a subslice of the original data.
///
/// The slice returned has the same lifetime as the original slice, and so
/// the iterator can continue to be used while this exists.
///
/// # Examples
///
/// ```
/// use bstr::ByteSlice;
///
/// let mut it = b"I want this. Not that. Right now.".sentences();
///
/// assert_eq!(&b"I want this. Not that. Right now."[..], it.as_bytes());
/// it.next();
/// assert_eq!(b"Not that. Right now.", it.as_bytes());
/// it.next();
/// it.next();
/// assert_eq!(b"", it.as_bytes());
/// ```
#[inline]
pub fn as_bytes(&self) -> &'a [u8] {
self.bs
}
}
impl<'a> Iterator for Sentences<'a> {
type Item = &'a str;
#[inline]
fn next(&mut self) -> Option<&'a str> {
let (sentence, size) = decode_sentence(self.bs);
if size == 0 {
return None;
}
self.bs = &self.bs[size..];
Some(sentence)
}
}
/// An iterator over sentences in a byte string, along with their byte offsets.
///
/// This iterator is typically constructed by
/// [`ByteSlice::sentence_indices`](trait.ByteSlice.html#method.sentence_indices).
///
/// Sentences typically include their trailing punctuation and whitespace.
///
/// Since sentences are made up of one or more codepoints, this iterator
/// yields `&str` elements (along with their start and end byte offsets).
/// When invalid UTF-8 is encountered, replacement codepoints are
/// [substituted](index.html#handling-of-invalid-utf-8). Because of this, the
/// indices yielded by this iterator may not correspond to the length of the
/// sentence yielded with those indices. For example, when this iterator
/// encounters `\xFF` in the byte string, then it will yield a pair of indices
/// ranging over a single byte, but will provide an `&str` equivalent to
/// `"\u{FFFD}"`, which is three bytes in length. However, when given only
/// valid UTF-8, then all indices are in exact correspondence with their paired
/// word.
///
/// This iterator yields words in accordance with the default sentence boundary
/// rules specified in
/// [UAX #29](https://www.unicode.org/reports/tr29/tr29-33.html#Sentence_Boundaries).
#[derive(Clone, Debug)]
pub struct SentenceIndices<'a> {
bs: &'a [u8],
forward_index: usize,
}
impl<'a> SentenceIndices<'a> {
pub(crate) fn new(bs: &'a [u8]) -> SentenceIndices<'a> {
SentenceIndices { bs, forward_index: 0 }
}
/// View the underlying data as a subslice of the original data.
///
/// The slice returned has the same lifetime as the original slice, and so
/// the iterator can continue to be used while this exists.
///
/// # Examples
///
/// ```
/// use bstr::ByteSlice;
///
/// let mut it = b"I want this. Not that. Right now.".sentence_indices();
///
/// assert_eq!(&b"I want this. Not that. Right now."[..], it.as_bytes());
/// it.next();
/// assert_eq!(b"Not that. Right now.", it.as_bytes());
/// it.next();
/// it.next();
/// assert_eq!(b"", it.as_bytes());
/// ```
#[inline]
pub fn as_bytes(&self) -> &'a [u8] {
self.bs
}
}
impl<'a> Iterator for SentenceIndices<'a> {
type Item = (usize, usize, &'a str);
#[inline]
fn next(&mut self) -> Option<(usize, usize, &'a str)> {
let index = self.forward_index;
let (word, size) = decode_sentence(self.bs);
if size == 0 {
return None;
}
self.bs = &self.bs[size..];
self.forward_index += size;
Some((index, index + size, word))
}
}
fn decode_sentence(bs: &[u8]) -> (&str, usize) {
if bs.is_empty() {
("", 0)
} else if let Some(hm) = {
let input = Input::new(bs).anchored(Anchored::Yes);
SENTENCE_BREAK_FWD.try_search_fwd(&input).unwrap()
} {
// Safe because a match can only occur for valid UTF-8.
let sentence = unsafe { bs[..hm.offset()].to_str_unchecked() };
(sentence, sentence.len())
} else {
const INVALID: &str = "\u{FFFD}";
// No match on non-empty bytes implies we found invalid UTF-8.
let (_, size) = utf8::decode_lossy(bs);
(INVALID, size)
}
}
#[cfg(all(test, feature = "std"))]
mod tests {
use alloc::{vec, vec::Vec};
#[cfg(not(miri))]
use ucd_parse::SentenceBreakTest;
use crate::ext_slice::ByteSlice;
#[test]
#[cfg(not(miri))]
fn forward_ucd() {
for (i, test) in ucdtests().into_iter().enumerate() {
let given = test.sentences.concat();
let got = sentences(given.as_bytes());
assert_eq!(
test.sentences,
got,
"\n\nsentence forward break test {} failed:\n\
given: {:?}\n\
expected: {:?}\n\
got: {:?}\n",
i,
given,
strs_to_bstrs(&test.sentences),
strs_to_bstrs(&got),
);
}
}
// Some additional tests that don't seem to be covered by the UCD tests.
#[test]
fn forward_additional() {
assert_eq!(vec!["a.. ", "A"], sentences(b"a.. A"));
assert_eq!(vec!["a.. a"], sentences(b"a.. a"));
assert_eq!(vec!["a... ", "A"], sentences(b"a... A"));
assert_eq!(vec!["a... a"], sentences(b"a... a"));
assert_eq!(vec!["a...,..., a"], sentences(b"a...,..., a"));
}
fn sentences(bytes: &[u8]) -> Vec<&str> {
bytes.sentences().collect()
}
#[cfg(not(miri))]
fn strs_to_bstrs<S: AsRef<str>>(strs: &[S]) -> Vec<&[u8]> {
strs.iter().map(|s| s.as_ref().as_bytes()).collect()
}
/// Return all of the UCD for sentence breaks.
#[cfg(not(miri))]
fn ucdtests() -> Vec<SentenceBreakTest> {
const TESTDATA: &str = include_str!("data/SentenceBreakTest.txt");
let mut tests = vec![];
for mut line in TESTDATA.lines() {
line = line.trim();
if line.starts_with("#") || line.contains("surrogate") {
continue;
}
tests.push(line.parse().unwrap());
}
tests
}
}
+24
View File
@@ -0,0 +1,24 @@
use regex_automata::{dfa::Automaton, Anchored, Input};
use crate::unicode::fsm::{
whitespace_anchored_fwd::WHITESPACE_ANCHORED_FWD,
whitespace_anchored_rev::WHITESPACE_ANCHORED_REV,
};
/// Return the first position of a non-whitespace character.
pub fn whitespace_len_fwd(slice: &[u8]) -> usize {
let input = Input::new(slice).anchored(Anchored::Yes);
WHITESPACE_ANCHORED_FWD
.try_search_fwd(&input)
.unwrap()
.map_or(0, |hm| hm.offset())
}
/// Return the last position of a non-whitespace character.
pub fn whitespace_len_rev(slice: &[u8]) -> usize {
let input = Input::new(slice).anchored(Anchored::Yes);
WHITESPACE_ANCHORED_REV
.try_search_rev(&input)
.unwrap()
.map_or(slice.len(), |hm| hm.offset())
}
+429
View File
@@ -0,0 +1,429 @@
use regex_automata::{dfa::Automaton, Anchored, Input};
use crate::{
ext_slice::ByteSlice,
unicode::fsm::{
simple_word_fwd::SIMPLE_WORD_FWD, word_break_fwd::WORD_BREAK_FWD,
},
utf8,
};
/// An iterator over words in a byte string.
///
/// This iterator is typically constructed by
/// [`ByteSlice::words`](trait.ByteSlice.html#method.words).
///
/// This is similar to the [`WordsWithBreaks`](struct.WordsWithBreaks.html)
/// iterator, except it only returns elements that contain a "word" character.
/// A word character is defined by UTS #18 (Annex C) to be the combination
/// of the `Alphabetic` and `Join_Control` properties, along with the
/// `Decimal_Number`, `Mark` and `Connector_Punctuation` general categories.
///
/// Since words are made up of one or more codepoints, this iterator yields
/// `&str` elements. When invalid UTF-8 is encountered, replacement codepoints
/// are [substituted](index.html#handling-of-invalid-utf-8).
///
/// This iterator yields words in accordance with the default word boundary
/// rules specified in
/// [UAX #29](https://www.unicode.org/reports/tr29/tr29-33.html#Word_Boundaries).
/// In particular, this may not be suitable for Japanese and Chinese scripts
/// that do not use spaces between words.
#[derive(Clone, Debug)]
pub struct Words<'a>(WordsWithBreaks<'a>);
impl<'a> Words<'a> {
pub(crate) fn new(bs: &'a [u8]) -> Words<'a> {
Words(WordsWithBreaks::new(bs))
}
/// View the underlying data as a subslice of the original data.
///
/// The slice returned has the same lifetime as the original slice, and so
/// the iterator can continue to be used while this exists.
///
/// # Examples
///
/// ```
/// use bstr::ByteSlice;
///
/// let mut it = b"foo bar baz".words();
///
/// assert_eq!(b"foo bar baz", it.as_bytes());
/// it.next();
/// it.next();
/// assert_eq!(b" baz", it.as_bytes());
/// it.next();
/// assert_eq!(b"", it.as_bytes());
/// ```
#[inline]
pub fn as_bytes(&self) -> &'a [u8] {
self.0.as_bytes()
}
}
impl<'a> Iterator for Words<'a> {
type Item = &'a str;
#[inline]
fn next(&mut self) -> Option<&'a str> {
for word in self.0.by_ref() {
let input =
Input::new(word).anchored(Anchored::Yes).earliest(true);
if SIMPLE_WORD_FWD.try_search_fwd(&input).unwrap().is_some() {
return Some(word);
}
}
None
}
}
/// An iterator over words in a byte string and their byte index positions.
///
/// This iterator is typically constructed by
/// [`ByteSlice::word_indices`](trait.ByteSlice.html#method.word_indices).
///
/// This is similar to the
/// [`WordsWithBreakIndices`](struct.WordsWithBreakIndices.html) iterator,
/// except it only returns elements that contain a "word" character. A
/// word character is defined by UTS #18 (Annex C) to be the combination
/// of the `Alphabetic` and `Join_Control` properties, along with the
/// `Decimal_Number`, `Mark` and `Connector_Punctuation` general categories.
///
/// Since words are made up of one or more codepoints, this iterator
/// yields `&str` elements (along with their start and end byte offsets).
/// When invalid UTF-8 is encountered, replacement codepoints are
/// [substituted](index.html#handling-of-invalid-utf-8). Because of this, the
/// indices yielded by this iterator may not correspond to the length of the
/// word yielded with those indices. For example, when this iterator encounters
/// `\xFF` in the byte string, then it will yield a pair of indices ranging
/// over a single byte, but will provide an `&str` equivalent to `"\u{FFFD}"`,
/// which is three bytes in length. However, when given only valid UTF-8, then
/// all indices are in exact correspondence with their paired word.
///
/// This iterator yields words in accordance with the default word boundary
/// rules specified in
/// [UAX #29](https://www.unicode.org/reports/tr29/tr29-33.html#Word_Boundaries).
/// In particular, this may not be suitable for Japanese and Chinese scripts
/// that do not use spaces between words.
#[derive(Clone, Debug)]
pub struct WordIndices<'a>(WordsWithBreakIndices<'a>);
impl<'a> WordIndices<'a> {
pub(crate) fn new(bs: &'a [u8]) -> WordIndices<'a> {
WordIndices(WordsWithBreakIndices::new(bs))
}
/// View the underlying data as a subslice of the original data.
///
/// The slice returned has the same lifetime as the original slice, and so
/// the iterator can continue to be used while this exists.
///
/// # Examples
///
/// ```
/// use bstr::ByteSlice;
///
/// let mut it = b"foo bar baz".word_indices();
///
/// assert_eq!(b"foo bar baz", it.as_bytes());
/// it.next();
/// it.next();
/// assert_eq!(b" baz", it.as_bytes());
/// it.next();
/// it.next();
/// assert_eq!(b"", it.as_bytes());
/// ```
#[inline]
pub fn as_bytes(&self) -> &'a [u8] {
self.0.as_bytes()
}
}
impl<'a> Iterator for WordIndices<'a> {
type Item = (usize, usize, &'a str);
#[inline]
fn next(&mut self) -> Option<(usize, usize, &'a str)> {
for (start, end, word) in self.0.by_ref() {
let input =
Input::new(word).anchored(Anchored::Yes).earliest(true);
if SIMPLE_WORD_FWD.try_search_fwd(&input).unwrap().is_some() {
return Some((start, end, word));
}
}
None
}
}
/// An iterator over all word breaks in a byte string.
///
/// This iterator is typically constructed by
/// [`ByteSlice::words_with_breaks`](trait.ByteSlice.html#method.words_with_breaks).
///
/// This iterator yields not only all words, but the content that comes between
/// words. In particular, if all elements yielded by this iterator are
/// concatenated, then the result is the original string (subject to Unicode
/// replacement codepoint substitutions).
///
/// Since words are made up of one or more codepoints, this iterator yields
/// `&str` elements. When invalid UTF-8 is encountered, replacement codepoints
/// are [substituted](index.html#handling-of-invalid-utf-8).
///
/// This iterator yields words in accordance with the default word boundary
/// rules specified in
/// [UAX #29](https://www.unicode.org/reports/tr29/tr29-33.html#Word_Boundaries).
/// In particular, this may not be suitable for Japanese and Chinese scripts
/// that do not use spaces between words.
#[derive(Clone, Debug)]
pub struct WordsWithBreaks<'a> {
bs: &'a [u8],
}
impl<'a> WordsWithBreaks<'a> {
pub(crate) fn new(bs: &'a [u8]) -> WordsWithBreaks<'a> {
WordsWithBreaks { bs }
}
/// View the underlying data as a subslice of the original data.
///
/// The slice returned has the same lifetime as the original slice, and so
/// the iterator can continue to be used while this exists.
///
/// # Examples
///
/// ```
/// use bstr::ByteSlice;
///
/// let mut it = b"foo bar baz".words_with_breaks();
///
/// assert_eq!(b"foo bar baz", it.as_bytes());
/// it.next();
/// assert_eq!(b" bar baz", it.as_bytes());
/// it.next();
/// it.next();
/// assert_eq!(b" baz", it.as_bytes());
/// it.next();
/// it.next();
/// assert_eq!(b"", it.as_bytes());
/// ```
#[inline]
pub fn as_bytes(&self) -> &'a [u8] {
self.bs
}
}
impl<'a> Iterator for WordsWithBreaks<'a> {
type Item = &'a str;
#[inline]
fn next(&mut self) -> Option<&'a str> {
let (word, size) = decode_word(self.bs);
if size == 0 {
return None;
}
self.bs = &self.bs[size..];
Some(word)
}
}
/// An iterator over all word breaks in a byte string, along with their byte
/// index positions.
///
/// This iterator is typically constructed by
/// [`ByteSlice::words_with_break_indices`](trait.ByteSlice.html#method.words_with_break_indices).
///
/// This iterator yields not only all words, but the content that comes between
/// words. In particular, if all elements yielded by this iterator are
/// concatenated, then the result is the original string (subject to Unicode
/// replacement codepoint substitutions).
///
/// Since words are made up of one or more codepoints, this iterator
/// yields `&str` elements (along with their start and end byte offsets).
/// When invalid UTF-8 is encountered, replacement codepoints are
/// [substituted](index.html#handling-of-invalid-utf-8). Because of this, the
/// indices yielded by this iterator may not correspond to the length of the
/// word yielded with those indices. For example, when this iterator encounters
/// `\xFF` in the byte string, then it will yield a pair of indices ranging
/// over a single byte, but will provide an `&str` equivalent to `"\u{FFFD}"`,
/// which is three bytes in length. However, when given only valid UTF-8, then
/// all indices are in exact correspondence with their paired word.
///
/// This iterator yields words in accordance with the default word boundary
/// rules specified in
/// [UAX #29](https://www.unicode.org/reports/tr29/tr29-33.html#Word_Boundaries).
/// In particular, this may not be suitable for Japanese and Chinese scripts
/// that do not use spaces between words.
#[derive(Clone, Debug)]
pub struct WordsWithBreakIndices<'a> {
bs: &'a [u8],
forward_index: usize,
}
impl<'a> WordsWithBreakIndices<'a> {
pub(crate) fn new(bs: &'a [u8]) -> WordsWithBreakIndices<'a> {
WordsWithBreakIndices { bs, forward_index: 0 }
}
/// View the underlying data as a subslice of the original data.
///
/// The slice returned has the same lifetime as the original slice, and so
/// the iterator can continue to be used while this exists.
///
/// # Examples
///
/// ```
/// use bstr::ByteSlice;
///
/// let mut it = b"foo bar baz".words_with_break_indices();
///
/// assert_eq!(b"foo bar baz", it.as_bytes());
/// it.next();
/// assert_eq!(b" bar baz", it.as_bytes());
/// it.next();
/// it.next();
/// assert_eq!(b" baz", it.as_bytes());
/// it.next();
/// it.next();
/// assert_eq!(b"", it.as_bytes());
/// ```
#[inline]
pub fn as_bytes(&self) -> &'a [u8] {
self.bs
}
}
impl<'a> Iterator for WordsWithBreakIndices<'a> {
type Item = (usize, usize, &'a str);
#[inline]
fn next(&mut self) -> Option<(usize, usize, &'a str)> {
let index = self.forward_index;
let (word, size) = decode_word(self.bs);
if size == 0 {
return None;
}
self.bs = &self.bs[size..];
self.forward_index += size;
Some((index, index + size, word))
}
}
fn decode_word(bs: &[u8]) -> (&str, usize) {
if bs.is_empty() {
("", 0)
} else if let Some(hm) = {
let input = Input::new(bs).anchored(Anchored::Yes);
WORD_BREAK_FWD.try_search_fwd(&input).unwrap()
} {
// Safe because a match can only occur for valid UTF-8.
let word = unsafe { bs[..hm.offset()].to_str_unchecked() };
(word, word.len())
} else {
const INVALID: &str = "\u{FFFD}";
// No match on non-empty bytes implies we found invalid UTF-8.
let (_, size) = utf8::decode_lossy(bs);
(INVALID, size)
}
}
#[cfg(all(test, feature = "std"))]
mod tests {
use alloc::{vec, vec::Vec};
#[cfg(not(miri))]
use ucd_parse::WordBreakTest;
use crate::ext_slice::ByteSlice;
#[test]
#[cfg(not(miri))]
fn forward_ucd() {
for (i, test) in ucdtests().into_iter().enumerate() {
let given = test.words.concat();
let got = words(given.as_bytes());
assert_eq!(
test.words,
got,
"\n\nword forward break test {} failed:\n\
given: {:?}\n\
expected: {:?}\n\
got: {:?}\n",
i,
given,
strs_to_bstrs(&test.words),
strs_to_bstrs(&got),
);
}
}
// Some additional tests that don't seem to be covered by the UCD tests.
//
// It's pretty amazing that the UCD tests miss these cases. I only found
// them by running this crate's segmenter and ICU's segmenter on the same
// text and comparing the output.
#[test]
fn forward_additional() {
assert_eq!(vec!["a", ".", " ", "Y"], words(b"a. Y"));
assert_eq!(vec!["r", ".", " ", "Yo"], words(b"r. Yo"));
assert_eq!(
vec!["whatsoever", ".", " ", "You", " ", "may"],
words(b"whatsoever. You may")
);
assert_eq!(
vec!["21stcentury'syesterday"],
words(b"21stcentury'syesterday")
);
assert_eq!(vec!["Bonta_", "'", "s"], words(b"Bonta_'s"));
assert_eq!(vec!["_vhat's"], words(b"_vhat's"));
assert_eq!(vec!["__on'anima"], words(b"__on'anima"));
assert_eq!(vec!["123_", "'", "4"], words(b"123_'4"));
assert_eq!(vec!["_123'4"], words(b"_123'4"));
assert_eq!(vec!["__12'345"], words(b"__12'345"));
assert_eq!(
vec!["tomorrowat4", ":", "00", ","],
words(b"tomorrowat4:00,")
);
assert_eq!(vec!["RS1", "'", "s"], words(b"RS1's"));
assert_eq!(vec!["X38"], words(b"X38"));
assert_eq!(vec!["4abc", ":", "00", ","], words(b"4abc:00,"));
assert_eq!(vec!["12S", "'", "1"], words(b"12S'1"));
assert_eq!(vec!["1XY"], words(b"1XY"));
assert_eq!(vec!["\u{FEFF}", "Ты"], words("\u{FEFF}Ты".as_bytes()));
// Tests that Vithkuqi works, which was introduced in Unicode 14.
// This test fails prior to Unicode 14.
assert_eq!(
vec!["\u{10570}\u{10597}"],
words("\u{10570}\u{10597}".as_bytes())
);
}
fn words(bytes: &[u8]) -> Vec<&str> {
bytes.words_with_breaks().collect()
}
#[cfg(not(miri))]
fn strs_to_bstrs<S: AsRef<str>>(strs: &[S]) -> Vec<&[u8]> {
strs.iter().map(|s| s.as_ref().as_bytes()).collect()
}
/// Return all of the UCD for word breaks.
#[cfg(not(miri))]
fn ucdtests() -> Vec<WordBreakTest> {
const TESTDATA: &str = include_str!("data/WordBreakTest.txt");
let mut tests = vec![];
for mut line in TESTDATA.lines() {
line = line.trim();
if line.starts_with("#") || line.contains("surrogate") {
continue;
}
tests.push(line.parse().unwrap());
}
tests
}
}
File diff suppressed because it is too large Load Diff