feat(vendor): introduced vendored coreutils with in-process execution support
- Vendored multiple uutils coreutils packages with custom adaptations for in-process I/O and path redirection using `pi-uutils-ctx`. - Implemented consistent `run` entry points across all utilities to replace process-global termination and allow execution within the embedding application. - Developed a shared `uu-checksum-common` library to centralize CLI construction, checksum computation, and digest validation logic. - Integrated thread-local command handling and resource management to support memory-bounded stream processing and localized error reporting.
This commit is contained in:
Vendored
+18
@@ -0,0 +1,18 @@
|
||||
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/b2sum), patched to route I/O
|
||||
# and path resolution through pi-uutils-ctx so it can run in-process as a shell
|
||||
# builtin. See src/b2sum.rs for the patch markers (`pi-uutils:` comments).
|
||||
[package]
|
||||
name = "uu_b2sum"
|
||||
version = "0.8.0"
|
||||
edition = "2024"
|
||||
license = "MIT"
|
||||
description = "b2sum ~ (uutils) blake2b checksum (vendored + patched for in-process embedding)"
|
||||
|
||||
[lib]
|
||||
path = "src/b2sum.rs"
|
||||
|
||||
[dependencies]
|
||||
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
|
||||
uucore = { version = "0.8.0", features = ["checksum", "encoding", "sum", "hardware"] }
|
||||
uu_checksum_common = { path = "../uu-checksum-common" }
|
||||
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
|
||||
Vendored
+18
@@ -0,0 +1,18 @@
|
||||
Copyright (c) uutils developers
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
this software and associated documentation files (the "Software"), to deal in
|
||||
the Software without restriction, including without limitation the rights to
|
||||
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
|
||||
the Software, and to permit persons to whom the Software is furnished to do so,
|
||||
subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
|
||||
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
||||
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
Vendored
+35
@@ -0,0 +1,35 @@
|
||||
// This file is part of the uutils coreutils package.
|
||||
//
|
||||
// For the full copyright and license information, please view the LICENSE
|
||||
// file that was distributed with this source code.
|
||||
|
||||
// spell-checker:ignore (ToDO) algo
|
||||
|
||||
// pi-uutils: Patched for in-process embedding via the shared `uu-checksum-common` crate,
|
||||
// which redirects all standard stream I/O and file resolution through `pi-uutils-ctx`.
|
||||
|
||||
use clap::Command;
|
||||
use std::ffi::OsString;
|
||||
|
||||
use uucore::checksum::{AlgoKind, BlakeLength, parse_blake_length};
|
||||
|
||||
pub fn run(argv: Vec<OsString>) -> i32 {
|
||||
let calculate_blake2b_length =
|
||||
|s: &str| parse_blake_length(AlgoKind::Blake2b, BlakeLength::String(s));
|
||||
uu_checksum_common::run_standalone_with_length(
|
||||
"b2sum",
|
||||
AlgoKind::Blake2b,
|
||||
uu_app(),
|
||||
argv,
|
||||
calculate_blake2b_length,
|
||||
)
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn uu_app() -> Command {
|
||||
uu_checksum_common::standalone_checksum_app_with_length(
|
||||
"Print or check BLAKE2b (512-bit) checksums.",
|
||||
"b2sum [OPTION]... [FILE]...",
|
||||
)
|
||||
.name("b2sum")
|
||||
}
|
||||
Vendored
+17
@@ -0,0 +1,17 @@
|
||||
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/base32), patched to route I/O
|
||||
# and path resolution through pi-uutils-ctx for safe in-process embedding.
|
||||
# See src/base_common.rs and src/base32.rs for `pi-uutils:` patch markers.
|
||||
[package]
|
||||
name = "uu_base32"
|
||||
version = "0.8.0"
|
||||
edition = "2024"
|
||||
license = "MIT"
|
||||
description = "base32 ~ (uutils) decode/encode input (vendored + patched for in-process embedding)"
|
||||
|
||||
[lib]
|
||||
path = "src/base32.rs"
|
||||
|
||||
[dependencies]
|
||||
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
|
||||
uucore = { version = "0.8.0", features = ["encoding"] }
|
||||
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
|
||||
Vendored
+18
@@ -0,0 +1,18 @@
|
||||
Copyright (c) uutils developers
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
this software and associated documentation files (the "Software"), to deal in
|
||||
the Software without restriction, including without limitation the rights to
|
||||
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
|
||||
the Software, and to permit persons to whom the Software is furnished to do so,
|
||||
subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
|
||||
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
||||
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
+47
@@ -0,0 +1,47 @@
|
||||
// This file is part of the uutils coreutils package.
|
||||
//
|
||||
// For the full copyright and license information, please view the LICENSE
|
||||
// file that was distributed with this source code.
|
||||
|
||||
pub mod base_common;
|
||||
|
||||
use clap::Command;
|
||||
use std::ffi::OsString;
|
||||
use std::io::Write;
|
||||
use uucore::encoding::Format;
|
||||
|
||||
/// pi-uutils: safe in-process entry point using invocation-scoped streams.
|
||||
pub fn run(argv: Vec<OsString>) -> i32 {
|
||||
let matches = match uu_app().try_get_matches_from(argv) {
|
||||
Ok(matches) => matches,
|
||||
Err(err) => {
|
||||
let rendered = err.to_string();
|
||||
if err.use_stderr() {
|
||||
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
|
||||
return 1;
|
||||
}
|
||||
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
|
||||
return 0;
|
||||
}
|
||||
};
|
||||
let result = base_common::Config::from(&matches).and_then(|config| {
|
||||
let mut input = base_common::get_input(&config)?;
|
||||
base_common::handle_input(&mut input, Format::Base32, config)
|
||||
});
|
||||
match result {
|
||||
Ok(()) => pi_uutils_ctx::exit_code(),
|
||||
Err(err) => {
|
||||
let code = err.code();
|
||||
let _ = writeln!(pi_uutils_ctx::stderr(), "base32: {err}");
|
||||
if code == 0 { 1 } else { code }
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub fn uu_app() -> Command {
|
||||
base_common::base_app(
|
||||
"encode/decode data and print to standard output\nWith no FILE, or when FILE is -, read standard input.\n\nThe data are encoded as described for the base32 alphabet in RFC 4648.\nWhen decoding, the input may contain newlines in addition to the bytes of the formal base32 alphabet. Use --ignore-garbage to attempt to recover from any other non-alphabet bytes in the encoded stream.".into(),
|
||||
"base32 [OPTION]... [FILE]".into(),
|
||||
)
|
||||
.name("base32")
|
||||
}
|
||||
+953
@@ -0,0 +1,953 @@
|
||||
// This file is part of the uutils coreutils package.
|
||||
//
|
||||
// For the full copyright and license information, please view the LICENSE
|
||||
// file that was distributed with this source code.
|
||||
|
||||
// spell-checker:ignore hexupper lsbf msbf unpadded nopad aGVsbG8sIHdvcmxkIQ
|
||||
|
||||
use clap::{Arg, ArgAction, Command};
|
||||
use std::ffi::OsString;
|
||||
use std::fs::File;
|
||||
use std::io::{self, BufRead, BufReader, Write};
|
||||
use std::path::Path;
|
||||
use uucore::display::Quotable;
|
||||
use uucore::encoding::{
|
||||
BASE2LSBF, BASE2MSBF, Base32Wrapper, Base58Wrapper, Base64SimdWrapper, EncodingWrapper, Format,
|
||||
SupportsFastDecodeAndEncode, Z85Wrapper,
|
||||
for_base_common::{BASE32, BASE32HEX, BASE64URL, HEXUPPER_PERMISSIVE},
|
||||
};
|
||||
use uucore::error::{FromIo, UResult, USimpleError, UUsageError, strip_errno};
|
||||
use uucore::format_usage;
|
||||
|
||||
pub const BASE_CMD_PARSE_ERROR: i32 = 1;
|
||||
|
||||
/// Encoded output will be formatted in lines of this length (the last line can be shorter)
|
||||
///
|
||||
/// Other implementations default to 76
|
||||
///
|
||||
/// This default is only used if no "-w"/"--wrap" argument is passed
|
||||
pub const WRAP_DEFAULT: usize = 76;
|
||||
|
||||
// Fixed to 8 KiB (equivalent to `std::sys::io::DEFAULT_BUF_SIZE` on most targets)
|
||||
pub const DEFAULT_BUF_SIZE: usize = 8 * 1024;
|
||||
|
||||
pub struct Config {
|
||||
pub decode: bool,
|
||||
pub ignore_garbage: bool,
|
||||
pub wrap_cols: Option<usize>,
|
||||
pub to_read: Option<OsString>,
|
||||
}
|
||||
|
||||
pub mod options {
|
||||
pub static DECODE: &str = "decode";
|
||||
pub static WRAP: &str = "wrap";
|
||||
pub static IGNORE_GARBAGE: &str = "ignore-garbage";
|
||||
pub static FILE: &str = "file";
|
||||
}
|
||||
|
||||
impl Config {
|
||||
pub fn from(options: &clap::ArgMatches) -> UResult<Self> {
|
||||
let to_read = match options.get_many::<OsString>(options::FILE) {
|
||||
Some(mut values) => {
|
||||
let name = values.next().unwrap();
|
||||
|
||||
if let Some(extra_op) = values.next() {
|
||||
return Err(UUsageError::new(
|
||||
BASE_CMD_PARSE_ERROR,
|
||||
format!("extra operand {}", extra_op.quote()),
|
||||
));
|
||||
}
|
||||
|
||||
if name == "-" {
|
||||
None
|
||||
} else {
|
||||
Some(name.clone())
|
||||
}
|
||||
}
|
||||
None => None,
|
||||
};
|
||||
|
||||
let wrap_cols = options
|
||||
.get_one::<String>(options::WRAP)
|
||||
.map(|num| {
|
||||
num.parse::<usize>().map_err(|_| {
|
||||
USimpleError::new(
|
||||
BASE_CMD_PARSE_ERROR,
|
||||
format!("invalid wrap size: {}", num.quote()),
|
||||
)
|
||||
})
|
||||
})
|
||||
.transpose()?;
|
||||
|
||||
Ok(Self {
|
||||
decode: options.get_flag(options::DECODE),
|
||||
ignore_garbage: options.get_flag(options::IGNORE_GARBAGE),
|
||||
wrap_cols,
|
||||
to_read,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
pub fn base_app(about: String, usage: String) -> Command {
|
||||
let cmd = Command::new("")
|
||||
.version(uucore::crate_version!())
|
||||
.about(about)
|
||||
.override_usage(format_usage(&usage))
|
||||
.infer_long_args(true);
|
||||
uucore::clap_localization::configure_localized_command(cmd)
|
||||
// Format arguments.
|
||||
.arg(
|
||||
Arg::new(options::DECODE)
|
||||
.short('d')
|
||||
.visible_short_alias('D')
|
||||
.long(options::DECODE)
|
||||
.help("decode data")
|
||||
.action(ArgAction::SetTrue)
|
||||
.overrides_with(options::DECODE),
|
||||
)
|
||||
.arg(
|
||||
Arg::new(options::IGNORE_GARBAGE)
|
||||
.short('i')
|
||||
.long(options::IGNORE_GARBAGE)
|
||||
.help("when decoding, ignore non-alphabetic characters")
|
||||
.action(ArgAction::SetTrue)
|
||||
.overrides_with(options::IGNORE_GARBAGE),
|
||||
)
|
||||
.arg(
|
||||
Arg::new(options::WRAP)
|
||||
.short('w')
|
||||
.long(options::WRAP)
|
||||
.value_name("COLS")
|
||||
.help(format!("wrap encoded lines after COLS character (default {WRAP_DEFAULT}, 0 to disable wrapping)"))
|
||||
.overrides_with(options::WRAP),
|
||||
)
|
||||
// "multiple" arguments are used to check whether there is more than one
|
||||
// file passed in.
|
||||
.arg(
|
||||
Arg::new(options::FILE)
|
||||
.index(1)
|
||||
.action(ArgAction::Append)
|
||||
.value_parser(clap::value_parser!(OsString))
|
||||
.value_hint(clap::ValueHint::FilePath),
|
||||
)
|
||||
}
|
||||
|
||||
pub fn get_input(config: &Config) -> UResult<Box<dyn BufRead>> {
|
||||
match &config.to_read {
|
||||
Some(name) => {
|
||||
let file = File::open(pi_uutils_ctx::resolve(Path::new(name)))
|
||||
.map_err_context(|| name.maybe_quote().to_string())?;
|
||||
Ok(Box::new(BufReader::with_capacity(DEFAULT_BUF_SIZE, file)))
|
||||
}
|
||||
None => {
|
||||
// pi-uutils: stdin belongs to this invocation, never the host process.
|
||||
Ok(Box::new(BufReader::with_capacity(
|
||||
DEFAULT_BUF_SIZE,
|
||||
pi_uutils_ctx::stdin(),
|
||||
)))
|
||||
}
|
||||
}
|
||||
}
|
||||
pub fn handle_input<R: BufRead>(input: &mut R, format: Format, config: Config) -> UResult<()> {
|
||||
// Always allow padding for Base64 to avoid a full pre-scan of the input.
|
||||
let supports_fast_decode_and_encode =
|
||||
get_supports_fast_decode_and_encode(format, config.decode, true);
|
||||
|
||||
let supports_fast_decode_and_encode_ref = supports_fast_decode_and_encode.as_ref();
|
||||
// pi-uutils: all output is scoped to this invocation.
|
||||
let mut stdout_lock = pi_uutils_ctx::stdout().lock();
|
||||
let result = match (format, config.decode) {
|
||||
// Base58 must process the entire input as one big integer; keep the
|
||||
// historical behavior of buffering everything for this format only.
|
||||
(Format::Base58, _) => {
|
||||
let mut buffered = Vec::new();
|
||||
input
|
||||
.read_to_end(&mut buffered)
|
||||
.map_err(|err| USimpleError::new(1, format_read_error(&err)))?;
|
||||
if config.decode {
|
||||
fast_decode::fast_decode_buffer(
|
||||
buffered,
|
||||
&mut stdout_lock,
|
||||
supports_fast_decode_and_encode_ref,
|
||||
config.ignore_garbage,
|
||||
)
|
||||
} else {
|
||||
fast_encode::fast_encode_buffer(
|
||||
buffered,
|
||||
&mut stdout_lock,
|
||||
supports_fast_decode_and_encode_ref,
|
||||
config.wrap_cols,
|
||||
)
|
||||
}
|
||||
}
|
||||
// Streaming path for all other encodings keeps memory bounded.
|
||||
(_, true) => fast_decode::fast_decode_stream(
|
||||
input,
|
||||
&mut stdout_lock,
|
||||
supports_fast_decode_and_encode_ref,
|
||||
config.ignore_garbage,
|
||||
),
|
||||
(_, false) => fast_encode::fast_encode_stream(
|
||||
input,
|
||||
&mut stdout_lock,
|
||||
supports_fast_decode_and_encode_ref,
|
||||
config.wrap_cols,
|
||||
),
|
||||
};
|
||||
|
||||
// Ensure any pending stdout buffer is flushed even if decoding failed; GNU basenc
|
||||
// keeps already-decoded bytes visible before reporting the error.
|
||||
match (result, stdout_lock.flush()) {
|
||||
(res, Ok(())) => res,
|
||||
(Ok(_), Err(err)) => Err(err.into()),
|
||||
(Err(original), Err(_)) => Err(original),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn get_supports_fast_decode_and_encode(
|
||||
format: Format,
|
||||
decode: bool,
|
||||
has_padding: bool,
|
||||
) -> Box<dyn SupportsFastDecodeAndEncode> {
|
||||
const BASE16_VALID_DECODING_MULTIPLE: usize = 2;
|
||||
const BASE2_VALID_DECODING_MULTIPLE: usize = 8;
|
||||
const BASE32_VALID_DECODING_MULTIPLE: usize = 8;
|
||||
const BASE64_VALID_DECODING_MULTIPLE: usize = 4;
|
||||
|
||||
const BASE16_UNPADDED_MULTIPLE: usize = 1;
|
||||
const BASE2_UNPADDED_MULTIPLE: usize = 1;
|
||||
const BASE32_UNPADDED_MULTIPLE: usize = 5;
|
||||
const BASE64_UNPADDED_MULTIPLE: usize = 3;
|
||||
|
||||
match format {
|
||||
Format::Base16 => Box::from(EncodingWrapper::new(
|
||||
HEXUPPER_PERMISSIVE,
|
||||
BASE16_VALID_DECODING_MULTIPLE,
|
||||
BASE16_UNPADDED_MULTIPLE,
|
||||
// spell-checker:disable-next-line
|
||||
b"0123456789ABCDEFabcdef",
|
||||
)),
|
||||
Format::Base2Lsbf => Box::from(EncodingWrapper::new(
|
||||
BASE2LSBF,
|
||||
BASE2_VALID_DECODING_MULTIPLE,
|
||||
BASE2_UNPADDED_MULTIPLE,
|
||||
// spell-checker:disable-next-line
|
||||
b"01",
|
||||
)),
|
||||
Format::Base2Msbf => Box::from(EncodingWrapper::new(
|
||||
BASE2MSBF,
|
||||
BASE2_VALID_DECODING_MULTIPLE,
|
||||
BASE2_UNPADDED_MULTIPLE,
|
||||
// spell-checker:disable-next-line
|
||||
b"01",
|
||||
)),
|
||||
Format::Base32 => Box::from(Base32Wrapper::new(
|
||||
BASE32,
|
||||
BASE32_VALID_DECODING_MULTIPLE,
|
||||
BASE32_UNPADDED_MULTIPLE,
|
||||
// spell-checker:disable-next-line
|
||||
b"ABCDEFGHIJKLMNOPQRSTUVWXYZ234567=",
|
||||
)),
|
||||
Format::Base32Hex => Box::from(Base32Wrapper::new(
|
||||
BASE32HEX,
|
||||
BASE32_VALID_DECODING_MULTIPLE,
|
||||
BASE32_UNPADDED_MULTIPLE,
|
||||
// spell-checker:disable-next-line
|
||||
b"0123456789ABCDEFGHIJKLMNOPQRSTUV=",
|
||||
)),
|
||||
Format::Base64 => {
|
||||
let alphabet: &[u8] = if has_padding {
|
||||
&b"abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789+/="[..]
|
||||
} else {
|
||||
&b"abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789+/"[..]
|
||||
};
|
||||
let use_padding = !decode || has_padding;
|
||||
Box::from(Base64SimdWrapper::new(
|
||||
use_padding,
|
||||
BASE64_VALID_DECODING_MULTIPLE,
|
||||
BASE64_UNPADDED_MULTIPLE,
|
||||
alphabet,
|
||||
))
|
||||
}
|
||||
Format::Base64Url => Box::from(EncodingWrapper::new(
|
||||
BASE64URL,
|
||||
BASE64_VALID_DECODING_MULTIPLE,
|
||||
BASE64_UNPADDED_MULTIPLE,
|
||||
// spell-checker:disable-next-line
|
||||
b"abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789=_-",
|
||||
)),
|
||||
Format::Z85 => Box::from(Z85Wrapper {}),
|
||||
Format::Base58 => Box::from(Base58Wrapper {}),
|
||||
}
|
||||
}
|
||||
|
||||
pub mod fast_encode {
|
||||
use crate::base_common::WRAP_DEFAULT;
|
||||
use std::{
|
||||
cmp::min,
|
||||
collections::VecDeque,
|
||||
io::{self, BufRead, Write},
|
||||
num::NonZeroUsize,
|
||||
};
|
||||
use uucore::{
|
||||
encoding::SupportsFastDecodeAndEncode,
|
||||
error::{UResult, USimpleError},
|
||||
};
|
||||
|
||||
struct LineWrapping {
|
||||
line_length: NonZeroUsize,
|
||||
print_buffer: Vec<u8>,
|
||||
}
|
||||
|
||||
// Start of helper functions
|
||||
fn encode_in_chunks_to_buffer(
|
||||
supports_fast_decode_and_encode: &dyn SupportsFastDecodeAndEncode,
|
||||
read_buffer: &[u8],
|
||||
encoded_buffer: &mut VecDeque<u8>,
|
||||
) -> UResult<()> {
|
||||
supports_fast_decode_and_encode.encode_to_vec_deque(read_buffer, encoded_buffer)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn write_without_line_breaks(
|
||||
encoded_buffer: &mut VecDeque<u8>,
|
||||
output: &mut dyn Write,
|
||||
is_cleanup: bool,
|
||||
empty_wrap: bool,
|
||||
) -> io::Result<()> {
|
||||
// TODO
|
||||
// `encoded_buffer` only has to be a VecDeque if line wrapping is enabled
|
||||
// (`make_contiguous` should be a no-op here)
|
||||
// Refactoring could avoid this call
|
||||
output.write_all(encoded_buffer.make_contiguous())?;
|
||||
|
||||
if is_cleanup {
|
||||
if !empty_wrap {
|
||||
output.write_all(b"\n")?;
|
||||
}
|
||||
} else {
|
||||
encoded_buffer.clear();
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn write_with_line_breaks(
|
||||
&mut LineWrapping {
|
||||
ref line_length,
|
||||
ref mut print_buffer,
|
||||
}: &mut LineWrapping,
|
||||
encoded_buffer: &mut VecDeque<u8>,
|
||||
output: &mut dyn Write,
|
||||
is_cleanup: bool,
|
||||
) -> io::Result<()> {
|
||||
let line_length = line_length.get();
|
||||
|
||||
let make_contiguous_result = encoded_buffer.make_contiguous();
|
||||
|
||||
let chunks_exact = make_contiguous_result.chunks_exact(line_length);
|
||||
|
||||
let mut bytes_added_to_print_buffer = 0;
|
||||
|
||||
for sl in chunks_exact {
|
||||
bytes_added_to_print_buffer += sl.len();
|
||||
|
||||
print_buffer.extend_from_slice(sl);
|
||||
print_buffer.push(b'\n');
|
||||
}
|
||||
|
||||
output.write_all(print_buffer)?;
|
||||
|
||||
// Remove the bytes that were just printed from `encoded_buffer`
|
||||
drop(encoded_buffer.drain(..bytes_added_to_print_buffer));
|
||||
|
||||
if is_cleanup {
|
||||
if encoded_buffer.is_empty() {
|
||||
// Do not write a newline in this case, because two trailing newlines should never be printed
|
||||
} else {
|
||||
// Print the partial line, since this is cleanup and no more data is coming
|
||||
output.write_all(encoded_buffer.make_contiguous())?;
|
||||
output.write_all(b"\n")?;
|
||||
}
|
||||
} else {
|
||||
print_buffer.clear();
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn write_to_output(
|
||||
line_wrapping: &mut Option<LineWrapping>,
|
||||
encoded_buffer: &mut VecDeque<u8>,
|
||||
output: &mut dyn Write,
|
||||
is_cleanup: bool,
|
||||
empty_wrap: bool,
|
||||
) -> io::Result<()> {
|
||||
// Write all data in `encoded_buffer` to `output`
|
||||
if let &mut Some(ref mut li) = line_wrapping {
|
||||
write_with_line_breaks(li, encoded_buffer, output, is_cleanup)?;
|
||||
} else {
|
||||
write_without_line_breaks(encoded_buffer, output, is_cleanup, empty_wrap)?;
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
// End of helper functions
|
||||
|
||||
pub fn fast_encode_buffer(
|
||||
input: Vec<u8>,
|
||||
output: &mut dyn Write,
|
||||
supports_fast_decode_and_encode: &dyn SupportsFastDecodeAndEncode,
|
||||
wrap: Option<usize>,
|
||||
) -> UResult<()> {
|
||||
// Based on performance testing
|
||||
|
||||
const ENCODE_IN_CHUNKS_OF_SIZE_MULTIPLE: usize = 1_024;
|
||||
|
||||
let encode_in_chunks_of_size =
|
||||
supports_fast_decode_and_encode.unpadded_multiple() * ENCODE_IN_CHUNKS_OF_SIZE_MULTIPLE;
|
||||
|
||||
assert!(encode_in_chunks_of_size > 0);
|
||||
|
||||
// The "data-encoding" crate supports line wrapping, but not arbitrary line wrapping, only certain widths, so
|
||||
// line wrapping must be handled here.
|
||||
// https://github.com/ia0/data-encoding/blob/4f42ad7ef242f6d243e4de90cd1b46a57690d00e/lib/src/lib.rs#L1710
|
||||
let mut line_wrapping = match wrap {
|
||||
// Line wrapping is disabled because "-w"/"--wrap" was passed with "0"
|
||||
Some(0) => None,
|
||||
// A custom line wrapping value was passed
|
||||
Some(an) => Some(LineWrapping {
|
||||
line_length: NonZeroUsize::new(an).unwrap(),
|
||||
print_buffer: Vec::<u8>::new(),
|
||||
}),
|
||||
// Line wrapping was not set, so the default is used
|
||||
None => Some(LineWrapping {
|
||||
line_length: NonZeroUsize::new(WRAP_DEFAULT).unwrap(),
|
||||
print_buffer: Vec::<u8>::new(),
|
||||
}),
|
||||
};
|
||||
|
||||
let input_size = input.len();
|
||||
|
||||
// Start of buffers
|
||||
// Data that was read from `input` but has not been encoded yet
|
||||
let mut leftover_buffer = VecDeque::<u8>::new();
|
||||
|
||||
// Encoded data that needs to be written to `output`
|
||||
let mut encoded_buffer = VecDeque::<u8>::new();
|
||||
// End of buffers
|
||||
|
||||
input
|
||||
.iter()
|
||||
.enumerate()
|
||||
.step_by(encode_in_chunks_of_size)
|
||||
.filter_map(|(idx, _)| {
|
||||
// The part of `input_buffer` that was actually filled by the call
|
||||
// to `read`
|
||||
let buffer = &input[idx..min(input_size, idx + encode_in_chunks_of_size)];
|
||||
|
||||
if buffer.len() < encode_in_chunks_of_size {
|
||||
leftover_buffer.extend(buffer);
|
||||
assert!(leftover_buffer.len() < encode_in_chunks_of_size);
|
||||
None
|
||||
} else {
|
||||
Some(buffer)
|
||||
}
|
||||
})
|
||||
.for_each(|read_buffer| {
|
||||
// Encode data in chunks, then place it in `encoded_buffer`
|
||||
assert_eq!(read_buffer.len(), encode_in_chunks_of_size);
|
||||
encode_in_chunks_to_buffer(
|
||||
supports_fast_decode_and_encode,
|
||||
read_buffer,
|
||||
&mut encoded_buffer,
|
||||
)
|
||||
.unwrap();
|
||||
// Write all data in `encoded_buffer` to `output`
|
||||
write_to_output(
|
||||
&mut line_wrapping,
|
||||
&mut encoded_buffer,
|
||||
output,
|
||||
false,
|
||||
wrap == Some(0),
|
||||
)
|
||||
.unwrap();
|
||||
});
|
||||
|
||||
// Cleanup
|
||||
// `input` has finished producing data, so the data remaining in the buffers needs to be encoded and printed
|
||||
{
|
||||
// Encode all remaining unencoded bytes, placing them in `encoded_buffer`
|
||||
supports_fast_decode_and_encode
|
||||
.encode_to_vec_deque(leftover_buffer.make_contiguous(), &mut encoded_buffer)?;
|
||||
|
||||
// Write all data in `encoded_buffer` to output
|
||||
// `is_cleanup` triggers special cleanup-only logic
|
||||
write_to_output(
|
||||
&mut line_wrapping,
|
||||
&mut encoded_buffer,
|
||||
output,
|
||||
true,
|
||||
wrap == Some(0),
|
||||
)?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Encodes all data read from `input` into Base32 using a fast, chunked
|
||||
/// implementation and writes the result to `output`.
|
||||
///
|
||||
/// The `supports_fast_decode_and_encode` parameter supplies an optimized
|
||||
/// encoder and determines the chunk size used for bulk processing. When
|
||||
/// `wrap` is:
|
||||
/// - `Some(0)`: no line wrapping is performed,
|
||||
/// - `Some(n)`: lines are wrapped every `n` characters,
|
||||
/// - `None`: the default wrap width is applied.
|
||||
///
|
||||
/// Remaining bytes are encoded and flushed at the end. I/O or encoding
|
||||
/// failures are propagated via `UResult`.
|
||||
pub fn fast_encode_stream(
|
||||
input: &mut dyn BufRead,
|
||||
output: &mut dyn Write,
|
||||
supports_fast_decode_and_encode: &dyn SupportsFastDecodeAndEncode,
|
||||
wrap: Option<usize>,
|
||||
) -> UResult<()> {
|
||||
const ENCODE_IN_CHUNKS_OF_SIZE_MULTIPLE: usize = 1_024;
|
||||
|
||||
let encode_in_chunks_of_size =
|
||||
supports_fast_decode_and_encode.unpadded_multiple() * ENCODE_IN_CHUNKS_OF_SIZE_MULTIPLE;
|
||||
|
||||
assert!(encode_in_chunks_of_size > 0);
|
||||
|
||||
let mut line_wrapping = match wrap {
|
||||
Some(0) => None,
|
||||
Some(an) => Some(LineWrapping {
|
||||
line_length: NonZeroUsize::new(an).unwrap(),
|
||||
print_buffer: Vec::<u8>::new(),
|
||||
}),
|
||||
None => Some(LineWrapping {
|
||||
line_length: NonZeroUsize::new(WRAP_DEFAULT).unwrap(),
|
||||
print_buffer: Vec::<u8>::new(),
|
||||
}),
|
||||
};
|
||||
|
||||
// Buffers
|
||||
let mut encoded_buffer = VecDeque::<u8>::new();
|
||||
let mut leftover_buffer = Vec::<u8>::with_capacity(encode_in_chunks_of_size);
|
||||
|
||||
loop {
|
||||
let read_buffer = input
|
||||
.fill_buf()
|
||||
.map_err(|err| USimpleError::new(1, super::format_read_error(&err)))?;
|
||||
if read_buffer.is_empty() {
|
||||
break;
|
||||
}
|
||||
|
||||
let mut consumed = 0;
|
||||
|
||||
if !leftover_buffer.is_empty() {
|
||||
let needed = encode_in_chunks_of_size - leftover_buffer.len();
|
||||
let take = needed.min(read_buffer.len());
|
||||
leftover_buffer.extend_from_slice(&read_buffer[..take]);
|
||||
consumed += take;
|
||||
|
||||
if leftover_buffer.len() == encode_in_chunks_of_size {
|
||||
encode_in_chunks_to_buffer(
|
||||
supports_fast_decode_and_encode,
|
||||
leftover_buffer.as_slice(),
|
||||
&mut encoded_buffer,
|
||||
)?;
|
||||
leftover_buffer.clear();
|
||||
|
||||
write_to_output(
|
||||
&mut line_wrapping,
|
||||
&mut encoded_buffer,
|
||||
output,
|
||||
false,
|
||||
wrap == Some(0),
|
||||
)?;
|
||||
}
|
||||
}
|
||||
|
||||
let remaining = &read_buffer[consumed..];
|
||||
let full_chunk_bytes =
|
||||
(remaining.len() / encode_in_chunks_of_size) * encode_in_chunks_of_size;
|
||||
|
||||
if full_chunk_bytes > 0 {
|
||||
for chunk in remaining[..full_chunk_bytes].chunks_exact(encode_in_chunks_of_size) {
|
||||
encode_in_chunks_to_buffer(
|
||||
supports_fast_decode_and_encode,
|
||||
chunk,
|
||||
&mut encoded_buffer,
|
||||
)?;
|
||||
write_to_output(
|
||||
&mut line_wrapping,
|
||||
&mut encoded_buffer,
|
||||
output,
|
||||
false,
|
||||
wrap == Some(0),
|
||||
)?;
|
||||
}
|
||||
consumed += full_chunk_bytes;
|
||||
}
|
||||
|
||||
if consumed < read_buffer.len() {
|
||||
leftover_buffer.extend_from_slice(&read_buffer[consumed..]);
|
||||
consumed = read_buffer.len();
|
||||
}
|
||||
|
||||
input.consume(consumed);
|
||||
|
||||
// `leftover_buffer` should never exceed one partial chunk.
|
||||
debug_assert!(leftover_buffer.len() < encode_in_chunks_of_size);
|
||||
}
|
||||
|
||||
// Encode any remaining bytes and flush
|
||||
supports_fast_decode_and_encode
|
||||
.encode_to_vec_deque(&leftover_buffer, &mut encoded_buffer)?;
|
||||
|
||||
write_to_output(
|
||||
&mut line_wrapping,
|
||||
&mut encoded_buffer,
|
||||
output,
|
||||
true,
|
||||
wrap == Some(0),
|
||||
)?;
|
||||
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
pub mod fast_decode {
|
||||
use std::io::{self, BufRead, Write};
|
||||
use uucore::{
|
||||
encoding::SupportsFastDecodeAndEncode,
|
||||
error::{UResult, USimpleError},
|
||||
};
|
||||
|
||||
// Start of helper functions
|
||||
fn alphabet_lookup(alphabet: &[u8]) -> [bool; 256] {
|
||||
// Precompute O(1) membership checks so we can validate every byte before decoding.
|
||||
let mut table = [false; 256];
|
||||
|
||||
for &byte in alphabet {
|
||||
table[usize::from(byte)] = true;
|
||||
}
|
||||
|
||||
table
|
||||
}
|
||||
|
||||
fn decode_in_chunks_to_buffer(
|
||||
supports_fast_decode_and_encode: &dyn SupportsFastDecodeAndEncode,
|
||||
read_buffer_filtered: &[u8],
|
||||
decoded_buffer: &mut Vec<u8>,
|
||||
) -> UResult<()> {
|
||||
supports_fast_decode_and_encode.decode_into_vec(read_buffer_filtered, decoded_buffer)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn write_to_output(decoded_buffer: &mut Vec<u8>, output: &mut dyn Write) -> io::Result<()> {
|
||||
// Write all data in `decoded_buffer` to `output`
|
||||
output.write_all(decoded_buffer.as_slice())?;
|
||||
|
||||
decoded_buffer.clear();
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn flush_ready_chunks(
|
||||
buffer: &mut Vec<u8>,
|
||||
block_limit: usize,
|
||||
valid_multiple: usize,
|
||||
supports_fast_decode_and_encode: &dyn SupportsFastDecodeAndEncode,
|
||||
decoded_buffer: &mut Vec<u8>,
|
||||
output: &mut dyn Write,
|
||||
) -> UResult<()> {
|
||||
// While at least one full decode block is buffered, keep draining
|
||||
// it and never yield more than block_limit per chunk.
|
||||
while buffer.len() >= valid_multiple {
|
||||
let take = buffer.len().min(block_limit);
|
||||
let aligned_take = take - (take % valid_multiple);
|
||||
|
||||
if aligned_take < valid_multiple {
|
||||
break;
|
||||
}
|
||||
|
||||
decode_in_chunks_to_buffer(
|
||||
supports_fast_decode_and_encode,
|
||||
&buffer[..aligned_take],
|
||||
decoded_buffer,
|
||||
)?;
|
||||
|
||||
write_to_output(decoded_buffer, output)?;
|
||||
|
||||
buffer.drain(..aligned_take);
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
// End of helper functions
|
||||
|
||||
pub fn fast_decode_buffer(
|
||||
input: Vec<u8>,
|
||||
output: &mut dyn Write,
|
||||
supports_fast_decode_and_encode: &dyn SupportsFastDecodeAndEncode,
|
||||
ignore_garbage: bool,
|
||||
) -> UResult<()> {
|
||||
const DECODE_IN_CHUNKS_OF_SIZE_MULTIPLE: usize = 1_024;
|
||||
|
||||
let alphabet = supports_fast_decode_and_encode.alphabet();
|
||||
let alphabet_table = alphabet_lookup(alphabet);
|
||||
let valid_multiple = supports_fast_decode_and_encode.valid_decoding_multiple();
|
||||
let decode_in_chunks_of_size = valid_multiple * DECODE_IN_CHUNKS_OF_SIZE_MULTIPLE;
|
||||
|
||||
assert!(decode_in_chunks_of_size > 0);
|
||||
assert!(valid_multiple > 0);
|
||||
|
||||
// Start of buffers
|
||||
|
||||
// Decoded data that needs to be written to `output`
|
||||
let mut decoded_buffer = Vec::<u8>::new();
|
||||
|
||||
// End of buffers
|
||||
|
||||
let mut buffer = Vec::with_capacity(decode_in_chunks_of_size);
|
||||
|
||||
let supports_partial_decode = supports_fast_decode_and_encode.supports_partial_decode();
|
||||
|
||||
for &byte in &input {
|
||||
if byte == b'\n' || byte == b'\r' {
|
||||
continue;
|
||||
}
|
||||
|
||||
if alphabet_table[usize::from(byte)] {
|
||||
buffer.push(byte);
|
||||
} else if ignore_garbage {
|
||||
continue;
|
||||
} else {
|
||||
return Err(USimpleError::new(1, "error: invalid input"));
|
||||
}
|
||||
|
||||
if supports_partial_decode {
|
||||
flush_ready_chunks(
|
||||
&mut buffer,
|
||||
decode_in_chunks_of_size,
|
||||
valid_multiple,
|
||||
supports_fast_decode_and_encode,
|
||||
&mut decoded_buffer,
|
||||
output,
|
||||
)?;
|
||||
} else if buffer.len() == decode_in_chunks_of_size {
|
||||
decode_in_chunks_to_buffer(
|
||||
supports_fast_decode_and_encode,
|
||||
&buffer,
|
||||
&mut decoded_buffer,
|
||||
)?;
|
||||
write_to_output(&mut decoded_buffer, output)?;
|
||||
buffer.clear();
|
||||
}
|
||||
}
|
||||
|
||||
if supports_partial_decode {
|
||||
flush_ready_chunks(
|
||||
&mut buffer,
|
||||
decode_in_chunks_of_size,
|
||||
valid_multiple,
|
||||
supports_fast_decode_and_encode,
|
||||
&mut decoded_buffer,
|
||||
output,
|
||||
)?;
|
||||
}
|
||||
|
||||
if !buffer.is_empty() {
|
||||
let mut owned_chunk: Option<Vec<u8>> = None;
|
||||
let mut had_invalid_tail = false;
|
||||
|
||||
if let Some(pad_result) = supports_fast_decode_and_encode.pad_remainder(&buffer) {
|
||||
had_invalid_tail = pad_result.had_invalid_tail;
|
||||
owned_chunk = Some(pad_result.chunk);
|
||||
}
|
||||
|
||||
let final_chunk = owned_chunk.as_deref().unwrap_or(&buffer);
|
||||
|
||||
supports_fast_decode_and_encode.decode_into_vec(final_chunk, &mut decoded_buffer)?;
|
||||
write_to_output(&mut decoded_buffer, output)?;
|
||||
|
||||
if had_invalid_tail {
|
||||
return Err(USimpleError::new(1, "error: invalid input"));
|
||||
}
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub fn fast_decode_stream(
|
||||
input: &mut dyn BufRead,
|
||||
output: &mut dyn Write,
|
||||
supports_fast_decode_and_encode: &dyn SupportsFastDecodeAndEncode,
|
||||
ignore_garbage: bool,
|
||||
) -> UResult<()> {
|
||||
const DECODE_IN_CHUNKS_OF_SIZE_MULTIPLE: usize = 1_024;
|
||||
|
||||
let alphabet = supports_fast_decode_and_encode.alphabet();
|
||||
let alphabet_table = alphabet_lookup(alphabet);
|
||||
let valid_multiple = supports_fast_decode_and_encode.valid_decoding_multiple();
|
||||
let decode_in_chunks_of_size = valid_multiple * DECODE_IN_CHUNKS_OF_SIZE_MULTIPLE;
|
||||
|
||||
assert!(decode_in_chunks_of_size > 0);
|
||||
assert!(valid_multiple > 0);
|
||||
|
||||
let supports_partial_decode = supports_fast_decode_and_encode.supports_partial_decode();
|
||||
|
||||
let mut buffer = Vec::with_capacity(decode_in_chunks_of_size);
|
||||
let mut decoded_buffer = Vec::<u8>::new();
|
||||
|
||||
loop {
|
||||
let read_buffer = input
|
||||
.fill_buf()
|
||||
.map_err(|err| USimpleError::new(1, super::format_read_error(&err)))?;
|
||||
let read_len = read_buffer.len();
|
||||
if read_len == 0 {
|
||||
break;
|
||||
}
|
||||
|
||||
for &byte in read_buffer {
|
||||
if byte == b'\n' || byte == b'\r' {
|
||||
continue;
|
||||
}
|
||||
|
||||
if alphabet_table[usize::from(byte)] {
|
||||
buffer.push(byte);
|
||||
} else if ignore_garbage {
|
||||
continue;
|
||||
} else {
|
||||
if supports_partial_decode {
|
||||
flush_ready_chunks(
|
||||
&mut buffer,
|
||||
decode_in_chunks_of_size,
|
||||
valid_multiple,
|
||||
supports_fast_decode_and_encode,
|
||||
&mut decoded_buffer,
|
||||
output,
|
||||
)?;
|
||||
} else {
|
||||
while buffer.len() >= decode_in_chunks_of_size {
|
||||
decode_in_chunks_to_buffer(
|
||||
supports_fast_decode_and_encode,
|
||||
&buffer[..decode_in_chunks_of_size],
|
||||
&mut decoded_buffer,
|
||||
)?;
|
||||
write_to_output(&mut decoded_buffer, output)?;
|
||||
buffer.drain(..decode_in_chunks_of_size);
|
||||
}
|
||||
}
|
||||
return Err(USimpleError::new(1, "error: invalid input"));
|
||||
}
|
||||
|
||||
if supports_partial_decode {
|
||||
flush_ready_chunks(
|
||||
&mut buffer,
|
||||
decode_in_chunks_of_size,
|
||||
valid_multiple,
|
||||
supports_fast_decode_and_encode,
|
||||
&mut decoded_buffer,
|
||||
output,
|
||||
)?;
|
||||
} else if buffer.len() == decode_in_chunks_of_size {
|
||||
decode_in_chunks_to_buffer(
|
||||
supports_fast_decode_and_encode,
|
||||
&buffer,
|
||||
&mut decoded_buffer,
|
||||
)?;
|
||||
write_to_output(&mut decoded_buffer, output)?;
|
||||
buffer.clear();
|
||||
}
|
||||
}
|
||||
|
||||
input.consume(read_len);
|
||||
}
|
||||
|
||||
if supports_partial_decode {
|
||||
flush_ready_chunks(
|
||||
&mut buffer,
|
||||
decode_in_chunks_of_size,
|
||||
valid_multiple,
|
||||
supports_fast_decode_and_encode,
|
||||
&mut decoded_buffer,
|
||||
output,
|
||||
)?;
|
||||
}
|
||||
|
||||
if !buffer.is_empty() {
|
||||
let mut owned_chunk: Option<Vec<u8>> = None;
|
||||
let mut had_invalid_tail = false;
|
||||
|
||||
if let Some(pad_result) = supports_fast_decode_and_encode.pad_remainder(&buffer) {
|
||||
had_invalid_tail = pad_result.had_invalid_tail;
|
||||
owned_chunk = Some(pad_result.chunk);
|
||||
}
|
||||
|
||||
let final_chunk = owned_chunk.as_deref().unwrap_or(&buffer);
|
||||
|
||||
supports_fast_decode_and_encode.decode_into_vec(final_chunk, &mut decoded_buffer)?;
|
||||
write_to_output(&mut decoded_buffer, output)?;
|
||||
|
||||
if had_invalid_tail {
|
||||
return Err(USimpleError::new(1, "error: invalid input"));
|
||||
}
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
fn format_read_error(error: &io::Error) -> String {
|
||||
format!("read error: {}", strip_errno(error))
|
||||
}
|
||||
|
||||
/// Determines if the input buffer contains any padding ('=') ignoring trailing whitespace.
|
||||
#[cfg(test)]
|
||||
fn read_and_has_padding<R: io::Read>(input: &mut R) -> UResult<(bool, Vec<u8>)> {
|
||||
let mut buf = Vec::new();
|
||||
input
|
||||
.read_to_end(&mut buf)
|
||||
.map_err(|err| USimpleError::new(1, format_read_error(&err)))?;
|
||||
|
||||
// Treat the stream as padded if any '=' exists (GNU coreutils continues decoding
|
||||
// even when padding bytes are followed by more data).
|
||||
let has_padding = buf.contains(&b'=');
|
||||
|
||||
Ok((has_padding, buf))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::base_common::read_and_has_padding;
|
||||
use std::io::Cursor;
|
||||
|
||||
#[test]
|
||||
fn test_has_padding() {
|
||||
let test_cases = vec![
|
||||
("aGVsbG8sIHdvcmxkIQ==", true),
|
||||
("aGVsbG8sIHdvcmxkIQ== ", true),
|
||||
("aGVsbG8sIHdvcmxkIQ==\n", true),
|
||||
("aGVsbG8sIHdvcmxkIQ== \n", true),
|
||||
("aGVsbG8sIHdvcmxkIQ=", true),
|
||||
("aGVsbG8sIHdvcmxkIQ= ", true),
|
||||
("MTIzNA==MTIzNA", true),
|
||||
("MTIzNA==\nMTIzNA", true),
|
||||
("aGVsbG8sIHdvcmxkIQ \n", false),
|
||||
("aGVsbG8sIHdvcmxkIQ", false),
|
||||
];
|
||||
|
||||
for (input, expected) in test_cases {
|
||||
let mut cursor = Cursor::new(input.as_bytes());
|
||||
assert_eq!(
|
||||
read_and_has_padding(&mut cursor).unwrap().0,
|
||||
expected,
|
||||
"Failed for input: '{input}'"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
Vendored
+18
@@ -0,0 +1,18 @@
|
||||
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/base64), patched to expose
|
||||
# an in-process entrypoint using pi-uutils-ctx streams. Shared implementation is
|
||||
# in ../uu-base32; see source comments for `pi-uutils:` patch markers.
|
||||
[package]
|
||||
name = "uu_base64"
|
||||
version = "0.8.0"
|
||||
edition = "2024"
|
||||
license = "MIT"
|
||||
description = "base64 ~ (uutils) decode/encode input (vendored + patched for in-process embedding)"
|
||||
|
||||
[lib]
|
||||
path = "src/base64.rs"
|
||||
|
||||
[dependencies]
|
||||
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
|
||||
uucore = { version = "0.8.0", features = ["encoding"] }
|
||||
uu_base32 = { path = "../uu-base32" }
|
||||
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
|
||||
Vendored
+18
@@ -0,0 +1,18 @@
|
||||
Copyright (c) uutils developers
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
this software and associated documentation files (the "Software"), to deal in
|
||||
the Software without restriction, including without limitation the rights to
|
||||
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
|
||||
the Software, and to permit persons to whom the Software is furnished to do so,
|
||||
subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
|
||||
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
||||
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
+46
@@ -0,0 +1,46 @@
|
||||
// This file is part of the uutils coreutils package.
|
||||
//
|
||||
// For the full copyright and license information, please view the LICENSE
|
||||
// file that was distributed with this source code.
|
||||
|
||||
use clap::Command;
|
||||
use std::ffi::OsString;
|
||||
use std::io::Write;
|
||||
use uu_base32::base_common;
|
||||
use uucore::encoding::Format;
|
||||
|
||||
/// pi-uutils: safe in-process entry point using invocation-scoped streams.
|
||||
pub fn run(argv: Vec<OsString>) -> i32 {
|
||||
let matches = match uu_app().try_get_matches_from(argv) {
|
||||
Ok(matches) => matches,
|
||||
Err(err) => {
|
||||
let rendered = err.to_string();
|
||||
if err.use_stderr() {
|
||||
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
|
||||
return 1;
|
||||
}
|
||||
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
|
||||
return 0;
|
||||
}
|
||||
};
|
||||
let result = base_common::Config::from(&matches).and_then(|config| {
|
||||
let mut input = base_common::get_input(&config)?;
|
||||
base_common::handle_input(&mut input, Format::Base64, config)
|
||||
});
|
||||
match result {
|
||||
Ok(()) => pi_uutils_ctx::exit_code(),
|
||||
Err(err) => {
|
||||
let code = err.code();
|
||||
let _ = writeln!(pi_uutils_ctx::stderr(), "base64: {err}");
|
||||
if code == 0 { 1 } else { code }
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub fn uu_app() -> Command {
|
||||
base_common::base_app(
|
||||
"encode/decode data and print to standard output\nWith no FILE, or when FILE is -, read standard input.\n\nThe data are encoded as described for the base64 alphabet in RFC 3548.\nWhen decoding, the input may contain newlines in addition to the bytes of the formal base64 alphabet. Use --ignore-garbage to attempt to recover from any other non-alphabet bytes in the encoded stream.".into(),
|
||||
"base64 [OPTION]... [FILE]".into(),
|
||||
)
|
||||
.name("base64")
|
||||
}
|
||||
Vendored
+17
@@ -0,0 +1,17 @@
|
||||
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/basename), patched to route I/O
|
||||
# and path resolution through pi-uutils-ctx so it can run in-process as a shell
|
||||
# builtin. See src/basename.rs for the patch markers (`pi-uutils:` comments).
|
||||
[package]
|
||||
name = "uu_basename"
|
||||
version = "0.8.0"
|
||||
edition = "2024"
|
||||
license = "MIT"
|
||||
description = "basename ~ (uutils) display PATHNAME with leading directory components removed (vendored + patched for in-process embedding)"
|
||||
|
||||
[lib]
|
||||
path = "src/basename.rs"
|
||||
|
||||
[dependencies]
|
||||
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
|
||||
uucore = { version = "0.8.0" }
|
||||
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
|
||||
Vendored
+18
@@ -0,0 +1,18 @@
|
||||
Copyright (c) uutils developers
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
this software and associated documentation files (the "Software"), to deal in
|
||||
the Software without restriction, including without limitation the rights to
|
||||
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
|
||||
the Software, and to permit persons to whom the Software is furnished to do so,
|
||||
subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
|
||||
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
||||
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
+172
@@ -0,0 +1,172 @@
|
||||
// This file is part of the uutils coreutils package.
|
||||
//
|
||||
// For the full copyright and license information, please view the LICENSE
|
||||
// file that was distributed with this source code.
|
||||
|
||||
// spell-checker:ignore (ToDO) fullname
|
||||
|
||||
// pi-uutils: Patched for in-process embedding in the shell.
|
||||
// All I/O is routed through thread-local stream buffers provided by `pi-uutils-ctx`.
|
||||
// Command-line arguments are parsed and errors are mapped without process-global
|
||||
// termination or stdout/stderr pollution.
|
||||
|
||||
use clap::builder::ValueParser;
|
||||
use clap::{Arg, ArgAction, ArgMatches, Command};
|
||||
use std::ffi::OsString;
|
||||
use std::io::Write;
|
||||
use std::path::PathBuf;
|
||||
use uucore::display::Quotable;
|
||||
use uucore::error::{UResult, UUsageError};
|
||||
use pi_uutils_ctx::format_usage;
|
||||
use uucore::line_ending::LineEnding;
|
||||
|
||||
pub mod options {
|
||||
pub static MULTIPLE: &str = "multiple";
|
||||
pub static NAME: &str = "name";
|
||||
pub static SUFFIX: &str = "suffix";
|
||||
pub static ZERO: &str = "zero";
|
||||
}
|
||||
|
||||
/// In-process builtin entry point. Unlike upstream's `uumain`, this parses the
|
||||
/// arguments directly (without the uucore clap-localization helper that would
|
||||
/// terminate the process), renders clap help/usage/version to the context
|
||||
/// streams, and maps the `UResult` to an exit code, so it is safe to run inside
|
||||
/// the host shell process.
|
||||
pub fn run(argv: Vec<OsString>) -> i32 {
|
||||
let matches = match uu_app().try_get_matches_from(argv) {
|
||||
Ok(matches) => matches,
|
||||
Err(err) => {
|
||||
let rendered = err.to_string();
|
||||
if err.use_stderr() {
|
||||
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
|
||||
return 1;
|
||||
}
|
||||
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
|
||||
return 0;
|
||||
},
|
||||
};
|
||||
match basename_main(&matches) {
|
||||
Ok(()) => pi_uutils_ctx::exit_code(),
|
||||
Err(err) => {
|
||||
let code = err.code();
|
||||
let _ = writeln!(pi_uutils_ctx::stderr(), "basename: {err}");
|
||||
if code == 0 { 1 } else { code }
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
fn basename_main(matches: &ArgMatches) -> UResult<()> {
|
||||
let line_ending = LineEnding::from_zero_flag(matches.get_flag(options::ZERO));
|
||||
|
||||
let mut name_args = matches
|
||||
.get_many::<OsString>(options::NAME)
|
||||
.unwrap_or_default()
|
||||
.collect::<Vec<_>>();
|
||||
if name_args.is_empty() {
|
||||
return Err(UUsageError::new(
|
||||
1,
|
||||
"missing operand".to_string(),
|
||||
));
|
||||
}
|
||||
let multiple_paths = matches.get_one::<OsString>(options::SUFFIX).is_some()
|
||||
|| matches.get_flag(options::MULTIPLE);
|
||||
let suffix = if multiple_paths {
|
||||
matches
|
||||
.get_one::<OsString>(options::SUFFIX)
|
||||
.cloned()
|
||||
.unwrap_or_default()
|
||||
} else {
|
||||
// "simple format"
|
||||
match name_args.len() {
|
||||
0 => panic!("already checked"),
|
||||
1 => OsString::default(),
|
||||
2 => name_args.pop().unwrap().clone(),
|
||||
_ => {
|
||||
return Err(UUsageError::new(
|
||||
1,
|
||||
format!("extra operand {}", name_args[2].quote()),
|
||||
));
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
//
|
||||
// Main Program Processing
|
||||
//
|
||||
let mut out = pi_uutils_ctx::stdout();
|
||||
for path in name_args {
|
||||
out.write_all(&basename(path, &suffix)?)?;
|
||||
write!(out, "{line_ending}")?;
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub fn uu_app() -> Command {
|
||||
Command::new("basename")
|
||||
.version(uucore::crate_version!())
|
||||
.about("Print NAME with any leading directory components removed\nIf specified, also remove a trailing SUFFIX")
|
||||
.override_usage(format_usage("basename [-z] NAME [SUFFIX]\n basename OPTION... NAME..."))
|
||||
.infer_long_args(true)
|
||||
.arg(
|
||||
Arg::new(options::MULTIPLE)
|
||||
.short('a')
|
||||
.long(options::MULTIPLE)
|
||||
.help("support multiple arguments and treat each as a NAME")
|
||||
.action(ArgAction::SetTrue)
|
||||
.overrides_with(options::MULTIPLE),
|
||||
)
|
||||
.arg(
|
||||
Arg::new(options::NAME)
|
||||
.action(ArgAction::Append)
|
||||
.value_parser(ValueParser::os_string())
|
||||
.value_hint(clap::ValueHint::AnyPath)
|
||||
.hide(true)
|
||||
.trailing_var_arg(true),
|
||||
)
|
||||
.arg(
|
||||
Arg::new(options::SUFFIX)
|
||||
.short('s')
|
||||
.long(options::SUFFIX)
|
||||
.value_name("SUFFIX")
|
||||
.value_parser(ValueParser::os_string())
|
||||
.help("remove a trailing SUFFIX; implies -a")
|
||||
.overrides_with(options::SUFFIX),
|
||||
)
|
||||
.arg(
|
||||
Arg::new(options::ZERO)
|
||||
.short('z')
|
||||
.long(options::ZERO)
|
||||
.help("end each output line with NUL, not newline")
|
||||
.action(ArgAction::SetTrue)
|
||||
.overrides_with(options::ZERO),
|
||||
)
|
||||
}
|
||||
|
||||
// We return a Vec<u8>. Returning a seemingly more proper `OsString` would
|
||||
// require back and forth conversions as we need a &[u8] for printing anyway.
|
||||
fn basename(fullname: &OsString, suffix: &OsString) -> UResult<Vec<u8>> {
|
||||
let fullname_bytes = uucore::os_str_as_bytes(fullname)?;
|
||||
|
||||
// Handle special case where path ends with /.
|
||||
if fullname_bytes.ends_with(b"/.") {
|
||||
return Ok(b".".into());
|
||||
}
|
||||
|
||||
// Convert to path buffer and get last path component
|
||||
let pb = PathBuf::from(fullname);
|
||||
|
||||
pb.components().next_back().map_or(Ok([].into()), |c| {
|
||||
let name = c.as_os_str();
|
||||
let name_bytes = uucore::os_str_as_bytes(name)?;
|
||||
if name == suffix {
|
||||
Ok(name_bytes.into())
|
||||
} else {
|
||||
let suffix_bytes = uucore::os_str_as_bytes(suffix)?;
|
||||
Ok(name_bytes
|
||||
.strip_suffix(suffix_bytes)
|
||||
.unwrap_or(name_bytes)
|
||||
.into())
|
||||
}
|
||||
})
|
||||
}
|
||||
+20
@@ -0,0 +1,20 @@
|
||||
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/checksum_common plus
|
||||
# uucore checksum compute/validate), patched to route I/O and path resolution
|
||||
# through pi-uutils-ctx for safe in-process shell builtin execution.
|
||||
[package]
|
||||
name = "uu_checksum_common"
|
||||
version = "0.8.0"
|
||||
edition = "2024"
|
||||
license = "MIT"
|
||||
description = "Shared checksum implementation from uutils (vendored + patched for in-process embedding)"
|
||||
|
||||
[lib]
|
||||
path = "src/lib.rs"
|
||||
|
||||
[dependencies]
|
||||
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
|
||||
base64-simd = "0.8"
|
||||
hex = "0.4.3"
|
||||
os_display = "0.1.4"
|
||||
uucore = { version = "0.8.0", features = ["checksum", "encoding", "sum", "hardware"] }
|
||||
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
|
||||
+18
@@ -0,0 +1,18 @@
|
||||
Copyright (c) uutils developers
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
this software and associated documentation files (the "Software"), to deal in
|
||||
the Software without restriction, including without limitation the rights to
|
||||
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
|
||||
the Software, and to permit persons to whom the Software is furnished to do so,
|
||||
subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
|
||||
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
||||
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
+217
@@ -0,0 +1,217 @@
|
||||
// This file is part of the uutils coreutils package.
|
||||
//
|
||||
// For the full copyright and license information, please view the LICENSE
|
||||
// file that was distributed with this source code.
|
||||
|
||||
use clap::{Arg, ArgAction, Command};
|
||||
use uucore::checksum::SUPPORTED_ALGORITHMS;
|
||||
|
||||
/// List of all options that can be encountered in checksum utils
|
||||
pub mod options {
|
||||
// cksum-specific
|
||||
pub const ALGORITHM: &str = "algorithm";
|
||||
pub const DEBUG: &str = "debug";
|
||||
|
||||
// positional arg
|
||||
pub const FILE: &str = "file";
|
||||
|
||||
pub const UNTAGGED: &str = "untagged";
|
||||
pub const TAG: &str = "tag";
|
||||
pub const LENGTH: &str = "length";
|
||||
pub const RAW: &str = "raw";
|
||||
pub const BASE64: &str = "base64";
|
||||
pub const CHECK: &str = "check";
|
||||
pub const TEXT: &str = "text";
|
||||
pub const BINARY: &str = "binary";
|
||||
pub const ZERO: &str = "zero";
|
||||
|
||||
// check-specific
|
||||
pub const STRICT: &str = "strict";
|
||||
pub const STATUS: &str = "status";
|
||||
pub const WARN: &str = "warn";
|
||||
pub const IGNORE_MISSING: &str = "ignore-missing";
|
||||
pub const QUIET: &str = "quiet";
|
||||
}
|
||||
|
||||
/// `ChecksumCommand` is a convenience trait to more easily declare checksum
|
||||
/// CLI interfaces with
|
||||
pub trait ChecksumCommand {
|
||||
fn with_algo(self) -> Self;
|
||||
|
||||
fn with_length(self) -> Self;
|
||||
|
||||
fn with_check_and_opts(self) -> Self;
|
||||
|
||||
fn with_binary(self) -> Self;
|
||||
|
||||
fn with_text(self, is_default: bool) -> Self;
|
||||
|
||||
fn with_tag(self, is_default: bool) -> Self;
|
||||
|
||||
fn with_untagged(self) -> Self;
|
||||
|
||||
fn with_raw(self) -> Self;
|
||||
|
||||
fn with_base64(self) -> Self;
|
||||
|
||||
fn with_zero(self) -> Self;
|
||||
|
||||
fn with_debug(self) -> Self;
|
||||
}
|
||||
|
||||
impl ChecksumCommand for Command {
|
||||
fn with_algo(self) -> Self {
|
||||
self.arg(
|
||||
Arg::new(options::ALGORITHM)
|
||||
.long(options::ALGORITHM)
|
||||
.short('a')
|
||||
.help("select the digest type to use. See DIGEST below")
|
||||
.value_name("ALGORITHM")
|
||||
.value_parser(SUPPORTED_ALGORITHMS),
|
||||
)
|
||||
}
|
||||
|
||||
fn with_length(self) -> Self {
|
||||
self.arg(
|
||||
Arg::new(options::LENGTH)
|
||||
.long(options::LENGTH)
|
||||
.short('l')
|
||||
.help("digest length in bits; must not exceed the maximum and must be a multiple of 8 for BLAKE2b")
|
||||
.action(ArgAction::Set),
|
||||
)
|
||||
}
|
||||
|
||||
fn with_check_and_opts(self) -> Self {
|
||||
self.arg(
|
||||
Arg::new(options::CHECK)
|
||||
.short('c')
|
||||
.long(options::CHECK)
|
||||
.help("read checksums from the FILEs and check them")
|
||||
.action(ArgAction::SetTrue),
|
||||
)
|
||||
.arg(
|
||||
Arg::new(options::WARN)
|
||||
.short('w')
|
||||
.long("warn")
|
||||
.help("warn about improperly formatted checksum lines")
|
||||
.action(ArgAction::SetTrue)
|
||||
.overrides_with_all([options::STATUS, options::QUIET]),
|
||||
)
|
||||
.arg(
|
||||
Arg::new(options::STATUS)
|
||||
.long("status")
|
||||
.help("don't output anything, status code shows success")
|
||||
.action(ArgAction::SetTrue)
|
||||
.overrides_with_all([options::WARN, options::QUIET]),
|
||||
)
|
||||
.arg(
|
||||
Arg::new(options::QUIET)
|
||||
.long(options::QUIET)
|
||||
.help("don't print OK for each successfully verified file")
|
||||
.action(ArgAction::SetTrue)
|
||||
.overrides_with_all([options::STATUS, options::WARN]),
|
||||
)
|
||||
.arg(
|
||||
Arg::new(options::IGNORE_MISSING)
|
||||
.long(options::IGNORE_MISSING)
|
||||
.help("don't fail or report status for missing files")
|
||||
.action(ArgAction::SetTrue),
|
||||
)
|
||||
.arg(
|
||||
Arg::new(options::STRICT)
|
||||
.long(options::STRICT)
|
||||
.help("exit non-zero for improperly formatted checksum lines")
|
||||
.action(ArgAction::SetTrue),
|
||||
)
|
||||
}
|
||||
|
||||
fn with_binary(self) -> Self {
|
||||
self.arg(
|
||||
Arg::new(options::BINARY)
|
||||
.long(options::BINARY)
|
||||
.short('b')
|
||||
.hide(true)
|
||||
.overrides_with(options::TEXT)
|
||||
.action(ArgAction::SetTrue),
|
||||
)
|
||||
}
|
||||
|
||||
fn with_text(self, is_default: bool) -> Self {
|
||||
let mut arg = Arg::new(options::TEXT)
|
||||
.long(options::TEXT)
|
||||
.short('t')
|
||||
.action(ArgAction::SetTrue);
|
||||
|
||||
arg = if is_default {
|
||||
arg.help("read in text mode (default)")
|
||||
} else {
|
||||
arg.hide(true)
|
||||
};
|
||||
|
||||
self.arg(arg)
|
||||
}
|
||||
|
||||
fn with_tag(self, default: bool) -> Self {
|
||||
let mut arg = Arg::new(options::TAG)
|
||||
.long(options::TAG)
|
||||
.action(ArgAction::SetTrue);
|
||||
|
||||
arg = if default {
|
||||
arg.help("create a BSD style checksum (default)")
|
||||
} else {
|
||||
arg.help("create a BSD style checksum")
|
||||
};
|
||||
|
||||
self.arg(arg)
|
||||
}
|
||||
|
||||
fn with_untagged(self) -> Self {
|
||||
self.arg(
|
||||
Arg::new(options::UNTAGGED)
|
||||
.long(options::UNTAGGED)
|
||||
.help("create a reversed style checksum, without digest type")
|
||||
.overrides_with(options::TAG)
|
||||
.action(ArgAction::SetTrue),
|
||||
)
|
||||
}
|
||||
|
||||
fn with_raw(self) -> Self {
|
||||
self.arg(
|
||||
Arg::new(options::RAW)
|
||||
.long(options::RAW)
|
||||
.help("emit a raw binary digest, not hexadecimal")
|
||||
.action(ArgAction::SetTrue),
|
||||
)
|
||||
}
|
||||
|
||||
fn with_base64(self) -> Self {
|
||||
self.arg(
|
||||
Arg::new(options::BASE64)
|
||||
.long(options::BASE64)
|
||||
.help("emit base64-encoded digests, not hexadecimal")
|
||||
.action(ArgAction::SetTrue)
|
||||
// Even though this could easily just override an earlier '--raw',
|
||||
// GNU cksum does not permit these flags to be combined:
|
||||
.conflicts_with(options::RAW),
|
||||
)
|
||||
}
|
||||
|
||||
fn with_zero(self) -> Self {
|
||||
self.arg(
|
||||
Arg::new(options::ZERO)
|
||||
.long(options::ZERO)
|
||||
.short('z')
|
||||
.help("end each output line with NUL, not newline, and disable file name escaping")
|
||||
.action(ArgAction::SetTrue),
|
||||
)
|
||||
}
|
||||
|
||||
fn with_debug(self) -> Self {
|
||||
self.arg(
|
||||
Arg::new(options::DEBUG)
|
||||
.long(options::DEBUG)
|
||||
.help("print CPU hardware capability detection info used by cksum")
|
||||
.action(ArgAction::SetTrue),
|
||||
)
|
||||
}
|
||||
}
|
||||
+317
@@ -0,0 +1,317 @@
|
||||
// This file is part of the uutils coreutils package.
|
||||
//
|
||||
// For the full copyright and license information, please view the LICENSE
|
||||
// file that was distributed with this source code.
|
||||
|
||||
// spell-checker:ignore bitlen
|
||||
|
||||
use std::ffi::OsStr;
|
||||
use std::fs::File;
|
||||
use std::io::{BufReader, Read, Write};
|
||||
use std::path::Path;
|
||||
|
||||
use uucore::checksum::{
|
||||
AlgoKind, ChecksumError, ReadingMode, SizedAlgoKind, digest_reader, escape_filename,
|
||||
};
|
||||
use uucore::error::{FromIo, UResult, USimpleError};
|
||||
use uucore::line_ending::LineEnding;
|
||||
use uucore::sum::DigestOutput;
|
||||
use crate::report_error;
|
||||
|
||||
/// Use the same buffer size as GNU when reading a file to create a checksum
|
||||
/// from it: 32 KiB.
|
||||
const READ_BUFFER_SIZE: usize = 32 * 1024;
|
||||
|
||||
/// Necessary options when computing a checksum. Historically, these options
|
||||
/// included a `binary` field to differentiate `--binary` and `--text` modes on
|
||||
/// windows. Since the support for this feature is approximate in GNU, and it's
|
||||
/// deprecated anyway, it was decided in #9168 to ignore the difference when
|
||||
/// computing the checksum.
|
||||
pub struct ChecksumComputeOptions {
|
||||
/// Which algorithm to use to compute the digest.
|
||||
pub algo_kind: SizedAlgoKind,
|
||||
|
||||
/// Printing format to use for each checksum.
|
||||
pub output_format: OutputFormat,
|
||||
|
||||
/// Whether to finish lines with '\n' or '\0'.
|
||||
pub line_ending: LineEnding,
|
||||
}
|
||||
|
||||
/// Whether to write the digest as hexadecimal or encoded in base64.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum DigestFormat {
|
||||
Hexadecimal,
|
||||
Base64,
|
||||
}
|
||||
|
||||
impl DigestFormat {
|
||||
#[inline]
|
||||
fn is_base64(self) -> bool {
|
||||
self == Self::Base64
|
||||
}
|
||||
}
|
||||
|
||||
/// Holds the representation that shall be used for printing a checksum line
|
||||
#[derive(Debug, PartialEq, Eq)]
|
||||
pub enum OutputFormat {
|
||||
/// Raw digest
|
||||
Raw,
|
||||
|
||||
/// Selected for older algorithms which had their custom formatting
|
||||
///
|
||||
/// Default for crc, sysv, bsd
|
||||
Legacy,
|
||||
|
||||
/// `$ALGO_NAME ($FILENAME) = $DIGEST`
|
||||
Tagged(DigestFormat),
|
||||
|
||||
/// '$DIGEST $FLAG$FILENAME'
|
||||
/// where 'flag' depends on the reading mode
|
||||
///
|
||||
/// Default for standalone checksum utilities
|
||||
Untagged(DigestFormat, ReadingMode),
|
||||
}
|
||||
|
||||
impl OutputFormat {
|
||||
#[inline]
|
||||
fn is_raw(&self) -> bool {
|
||||
*self == Self::Raw
|
||||
}
|
||||
|
||||
/// Find the correct output format for cksum.
|
||||
pub fn from_cksum(algo: AlgoKind, tag: bool, binary: bool, raw: bool, base64: bool) -> Self {
|
||||
// Raw output format takes precedence over anything else.
|
||||
if raw {
|
||||
return Self::Raw;
|
||||
}
|
||||
|
||||
// Then, if the algo is legacy, takes precedence over the rest
|
||||
if algo.is_legacy() {
|
||||
return Self::Legacy;
|
||||
}
|
||||
|
||||
let digest_format = if base64 {
|
||||
DigestFormat::Base64
|
||||
} else {
|
||||
DigestFormat::Hexadecimal
|
||||
};
|
||||
|
||||
// After that, decide between tagged and untagged output
|
||||
if tag {
|
||||
Self::Tagged(digest_format)
|
||||
} else {
|
||||
let reading_mode = if binary {
|
||||
ReadingMode::Binary
|
||||
} else {
|
||||
ReadingMode::Text
|
||||
};
|
||||
Self::Untagged(digest_format, reading_mode)
|
||||
}
|
||||
}
|
||||
|
||||
/// Find the correct output format for a standalone checksum util (b2sum,
|
||||
/// md5sum, etc)
|
||||
///
|
||||
/// Since standalone utils can't use the Raw or Legacy output format, it is
|
||||
/// decided only using the --tag, --binary and --text arguments.
|
||||
pub fn from_standalone(text: bool, tag: bool) -> Self {
|
||||
if tag {
|
||||
Self::Tagged(DigestFormat::Hexadecimal)
|
||||
} else {
|
||||
Self::Untagged(
|
||||
DigestFormat::Hexadecimal,
|
||||
if text {
|
||||
ReadingMode::Text
|
||||
} else {
|
||||
ReadingMode::Binary
|
||||
},
|
||||
)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn print_legacy_checksum(
|
||||
options: &ChecksumComputeOptions,
|
||||
filename: &OsStr,
|
||||
sum: &DigestOutput,
|
||||
size: usize,
|
||||
) {
|
||||
debug_assert!(options.algo_kind.is_legacy());
|
||||
debug_assert!(matches!(sum, DigestOutput::U16(_) | DigestOutput::Crc(_)));
|
||||
|
||||
let (escaped_filename, prefix) = if options.line_ending == LineEnding::Nul {
|
||||
(filename.to_string_lossy().to_string(), "")
|
||||
} else {
|
||||
escape_filename(filename)
|
||||
};
|
||||
|
||||
// Print the sum
|
||||
match (options.algo_kind, sum) {
|
||||
(SizedAlgoKind::Sysv, DigestOutput::U16(sum)) => {
|
||||
let _ = write!(
|
||||
pi_uutils_ctx::stdout(),
|
||||
"{prefix}{sum} {}",
|
||||
size.div_ceil(options.algo_kind.bitlen()),
|
||||
);
|
||||
}
|
||||
(SizedAlgoKind::Bsd, DigestOutput::U16(sum)) => {
|
||||
// The BSD checksum output is 5 digit integer
|
||||
let bsd_width = 5;
|
||||
let _ = write!(
|
||||
pi_uutils_ctx::stdout(),
|
||||
"{prefix}{sum:0bsd_width$} {:bsd_width$}",
|
||||
size.div_ceil(options.algo_kind.bitlen()),
|
||||
);
|
||||
}
|
||||
(SizedAlgoKind::Crc | SizedAlgoKind::Crc32b, DigestOutput::Crc(sum)) => {
|
||||
let _ = write!(pi_uutils_ctx::stdout(), "{prefix}{sum} {size}");
|
||||
}
|
||||
(algo, output) => unreachable!("Bug: Invalid legacy checksum ({algo:?}, {output:?})"),
|
||||
}
|
||||
|
||||
// Print the filename after a space if not stdin
|
||||
if escaped_filename != "-" {
|
||||
let _ = write!(pi_uutils_ctx::stdout(), " ");
|
||||
let _dropped_result = pi_uutils_ctx::stdout().write_all(escaped_filename.as_bytes());
|
||||
}
|
||||
}
|
||||
|
||||
fn print_tagged_checksum(options: &ChecksumComputeOptions, filename: &OsStr, sum: &String) {
|
||||
let (escaped_filename, prefix) = if options.line_ending == LineEnding::Nul {
|
||||
(filename.to_string_lossy().to_string(), "")
|
||||
} else {
|
||||
escape_filename(filename)
|
||||
};
|
||||
|
||||
// Print algo name and opening parenthesis.
|
||||
let _ = write!(
|
||||
pi_uutils_ctx::stdout(),
|
||||
"{prefix}{} (",
|
||||
options.algo_kind.to_tag()
|
||||
);
|
||||
|
||||
// Print filename
|
||||
let _dropped_result = pi_uutils_ctx::stdout().write_all(escaped_filename.as_bytes());
|
||||
|
||||
// Print closing parenthesis and sum
|
||||
let _ = write!(pi_uutils_ctx::stdout(), ") = {sum}");
|
||||
}
|
||||
|
||||
fn print_untagged_checksum(
|
||||
options: &ChecksumComputeOptions,
|
||||
filename: &OsStr,
|
||||
sum: &String,
|
||||
reading_mode: ReadingMode,
|
||||
) {
|
||||
let (escaped_filename, prefix) = if options.line_ending == LineEnding::Nul {
|
||||
(filename.to_string_lossy().to_string(), "")
|
||||
} else {
|
||||
escape_filename(filename)
|
||||
};
|
||||
|
||||
// Print checksum and reading mode flag
|
||||
let _ = write!(
|
||||
pi_uutils_ctx::stdout(),
|
||||
"{prefix}{sum} {}",
|
||||
match reading_mode {
|
||||
ReadingMode::Binary => '*',
|
||||
ReadingMode::Text => ' ',
|
||||
}
|
||||
);
|
||||
|
||||
// Print filename
|
||||
let _dropped_result = pi_uutils_ctx::stdout().write_all(escaped_filename.as_bytes());
|
||||
}
|
||||
|
||||
/// Calculate checksum
|
||||
///
|
||||
/// # Arguments
|
||||
///
|
||||
/// * `options` - CLI options for the assigning checksum algorithm
|
||||
/// * `files` - A iterator of [`OsStr`] which is a bunch of files that are using for calculating checksum
|
||||
pub fn perform_checksum_computation<'a, I>(options: ChecksumComputeOptions, files: I) -> UResult<()>
|
||||
where
|
||||
I: Iterator<Item = &'a OsStr>,
|
||||
{
|
||||
let mut files = files.peekable();
|
||||
|
||||
while let Some(filename) = files.next() {
|
||||
// Check that in raw mode, we are not provided with several files.
|
||||
if options.output_format.is_raw() && files.peek().is_some() {
|
||||
return Err(Box::new(ChecksumError::RawMultipleFiles));
|
||||
}
|
||||
|
||||
let filepath = Path::new(filename);
|
||||
let resolved_filepath = pi_uutils_ctx::resolve(filepath);
|
||||
let stdin_buf;
|
||||
let file_buf;
|
||||
if resolved_filepath.is_dir() {
|
||||
report_error(&USimpleError::new(1, format!("{}: Is a directory", filepath.display())));
|
||||
continue;
|
||||
}
|
||||
|
||||
// Handle the file input
|
||||
let mut file = BufReader::with_capacity(
|
||||
READ_BUFFER_SIZE,
|
||||
if filename == "-" {
|
||||
stdin_buf = pi_uutils_ctx::stdin();
|
||||
Box::new(stdin_buf) as Box<dyn Read>
|
||||
} else {
|
||||
file_buf = match File::open(&resolved_filepath) {
|
||||
Ok(file) => file,
|
||||
Err(err) => {
|
||||
report_error(&err.map_err_context(|| filepath.to_string_lossy().into()));
|
||||
continue;
|
||||
}
|
||||
};
|
||||
Box::new(file_buf) as Box<dyn Read>
|
||||
},
|
||||
);
|
||||
|
||||
let mut digest = options.algo_kind.create_digest();
|
||||
|
||||
// Always compute the "binary" version of the digest, i.e. on Windows,
|
||||
// never handle CRLFs specifically.
|
||||
let (digest_output, sz) = digest_reader(&mut digest, &mut file, ReadingMode::Binary)
|
||||
.map_err_context(|| "failed to read input".to_string())?;
|
||||
|
||||
// Encodes the sum if df is Base64, leaves as-is otherwise.
|
||||
let encode_sum = |sum: DigestOutput, df: DigestFormat| {
|
||||
if df.is_base64() {
|
||||
sum.to_base64()
|
||||
} else {
|
||||
sum.to_hex()
|
||||
}
|
||||
};
|
||||
|
||||
match options.output_format {
|
||||
OutputFormat::Raw => {
|
||||
// Cannot handle multiple files anyway, output immediately.
|
||||
digest_output.write_raw(pi_uutils_ctx::stdout())?;
|
||||
return Ok(());
|
||||
}
|
||||
OutputFormat::Legacy => {
|
||||
print_legacy_checksum(&options, filename, &digest_output, sz);
|
||||
}
|
||||
OutputFormat::Tagged(digest_format) => {
|
||||
print_tagged_checksum(
|
||||
&options,
|
||||
filename,
|
||||
&encode_sum(digest_output, digest_format)?,
|
||||
);
|
||||
}
|
||||
OutputFormat::Untagged(digest_format, reading_mode) => {
|
||||
print_untagged_checksum(
|
||||
&options,
|
||||
filename,
|
||||
&encode_sum(digest_output, digest_format)?,
|
||||
reading_mode,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
let _ = write!(pi_uutils_ctx::stdout(), "{}", options.line_ending);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
+223
@@ -0,0 +1,223 @@
|
||||
// This file is part of the uutils coreutils package.
|
||||
//
|
||||
// For the full copyright and license information, please view the LICENSE
|
||||
// file that was distributed with this source code.
|
||||
//
|
||||
// pi-uutils: vendored from uutils/coreutils 0.8.0 checksum_common and patched
|
||||
// to use invocation-scoped I/O and cwd resolution for in-process builtins.
|
||||
|
||||
use std::borrow::Borrow;
|
||||
use std::cell::RefCell;
|
||||
use std::ffi::OsString;
|
||||
use std::io::Write;
|
||||
|
||||
use clap::builder::ValueParser;
|
||||
use clap::{Arg, ArgAction, ArgMatches, Command, ValueHint};
|
||||
use uucore::checksum::{AlgoKind, ChecksumError, SizedAlgoKind};
|
||||
use uucore::error::{UError, UResult};
|
||||
use uucore::line_ending::LineEnding;
|
||||
|
||||
mod cli;
|
||||
mod compute;
|
||||
mod validate;
|
||||
pub use cli::{options, ChecksumCommand};
|
||||
pub use compute::{ChecksumComputeOptions, DigestFormat, OutputFormat};
|
||||
pub use validate::{ChecksumValidateOptions, ChecksumVerbose};
|
||||
|
||||
thread_local! {
|
||||
static COMMAND_NAME: RefCell<&'static str> = const { RefCell::new("checksum") };
|
||||
}
|
||||
|
||||
pub(crate) fn command_name() -> &'static str {
|
||||
COMMAND_NAME.with(|name| *name.borrow())
|
||||
}
|
||||
|
||||
pub(crate) fn report_error(error: &dyn std::fmt::Display) {
|
||||
let _ = writeln!(pi_uutils_ctx::stderr(), "{}: {error}", command_name());
|
||||
pi_uutils_ctx::set_exit_code(1);
|
||||
}
|
||||
|
||||
pub(crate) fn report_warning(message: &str) {
|
||||
let _ = writeln!(pi_uutils_ctx::stderr(), "{}: {message}", command_name());
|
||||
}
|
||||
|
||||
/// Generate a context-safe standalone checksum wrapper.
|
||||
#[macro_export]
|
||||
macro_rules! declare_standalone {
|
||||
($bin:literal, $kind:expr) => {
|
||||
pub fn run(argv: Vec<::std::ffi::OsString>) -> i32 {
|
||||
::uu_checksum_common::run_standalone($bin, $kind, uu_app(), argv)
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn uu_app() -> ::clap::Command {
|
||||
let (about, usage) = ::uu_checksum_common::standalone_strings($bin);
|
||||
::uu_checksum_common::standalone_checksum_app(about, usage).name($bin)
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
/// English descriptions used by standalone wrappers (localization is
|
||||
/// intentionally literalized because embedded commands have no global locale).
|
||||
pub fn standalone_strings(bin: &str) -> (&'static str, &'static str) {
|
||||
match bin {
|
||||
"md5sum" => ("Print or check the MD5 checksums", "md5sum [OPTIONS] [FILE]..."),
|
||||
"sha1sum" => ("Print or check SHA1 (160-bit) checksums", "sha1sum [OPTION]... [FILE]..."),
|
||||
"sha224sum" => ("Print or check SHA224 (224-bit) checksums", "sha224sum [OPTION]... [FILE]..."),
|
||||
"sha256sum" => ("Print or check SHA256 (256-bit) checksums", "sha256sum [OPTION]... [FILE]..."),
|
||||
"sha384sum" => ("Print or check SHA384 (384-bit) checksums", "sha384sum [OPTION]... [FILE]..."),
|
||||
"sha512sum" => ("Print or check SHA512 (512-bit) checksums", "sha512sum [OPTION]... [FILE]..."),
|
||||
"b2sum" => ("Print or check BLAKE2b (512-bit) checksums", "b2sum [OPTION]... [FILE]..."),
|
||||
_ => ("Print or check checksums", "checksum [OPTION]... [FILE]..."),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn run_standalone(
|
||||
bin: &'static str,
|
||||
algo: AlgoKind,
|
||||
cmd: Command,
|
||||
argv: Vec<OsString>,
|
||||
) -> i32 {
|
||||
run_with_optional_length(bin, algo, cmd, argv, None)
|
||||
}
|
||||
|
||||
/// Context-safe entrypoint for b2sum and other standalone hashes supporting
|
||||
/// `--length`. The validator is applied only when that option is present.
|
||||
pub fn run_standalone_with_length(
|
||||
bin: &'static str,
|
||||
algo: AlgoKind,
|
||||
cmd: Command,
|
||||
argv: Vec<OsString>,
|
||||
validate_len: fn(&str) -> UResult<usize>,
|
||||
) -> i32 {
|
||||
run_with_optional_length(bin, algo, cmd, argv, Some(validate_len))
|
||||
}
|
||||
|
||||
fn run_with_optional_length(
|
||||
bin: &'static str,
|
||||
algo: AlgoKind,
|
||||
cmd: Command,
|
||||
argv: Vec<OsString>,
|
||||
validate_len: Option<fn(&str) -> UResult<usize>>,
|
||||
) -> i32 {
|
||||
COMMAND_NAME.with(|name| *name.borrow_mut() = bin);
|
||||
let matches = match cmd.try_get_matches_from(argv) {
|
||||
Ok(matches) => matches,
|
||||
Err(err) => {
|
||||
let rendered = err.to_string();
|
||||
if err.use_stderr() {
|
||||
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
|
||||
return 2;
|
||||
}
|
||||
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
|
||||
return 0;
|
||||
}
|
||||
};
|
||||
let length = match validate_len {
|
||||
Some(validate_len) => match matches
|
||||
.get_one::<String>(options::LENGTH)
|
||||
.map(String::as_str)
|
||||
.map(validate_len)
|
||||
.transpose()
|
||||
{
|
||||
Ok(length) => length,
|
||||
Err(err) => return finish_error(bin, err),
|
||||
},
|
||||
None => None,
|
||||
};
|
||||
let text = !matches.get_flag(options::BINARY);
|
||||
let tag = matches.get_flag(options::TAG);
|
||||
let format = OutputFormat::from_standalone(text, tag);
|
||||
match checksum_main(Some(algo), length, matches, format) {
|
||||
Ok(()) => pi_uutils_ctx::exit_code(),
|
||||
Err(err) => finish_error(bin, err),
|
||||
}
|
||||
}
|
||||
|
||||
fn finish_error(bin: &str, err: Box<dyn UError>) -> i32 {
|
||||
let code = err.code();
|
||||
let message = err.to_string();
|
||||
if !message.is_empty() {
|
||||
let _ = writeln!(pi_uutils_ctx::stderr(), "{bin}: {message}");
|
||||
}
|
||||
if code == 0 { 1 } else { code }
|
||||
}
|
||||
|
||||
pub fn default_checksum_app(about: impl Into<String>, usage: impl Into<String>) -> Command {
|
||||
Command::new("")
|
||||
.version("0.8.0")
|
||||
.about(about.into())
|
||||
.override_usage(usage.into())
|
||||
.infer_long_args(true)
|
||||
.args_override_self(true)
|
||||
.after_help("With no FILE or when FILE is -, read standard input")
|
||||
.arg(
|
||||
Arg::new(options::FILE)
|
||||
.hide(true)
|
||||
.action(ArgAction::Append)
|
||||
.value_parser(ValueParser::os_string())
|
||||
.default_value("-")
|
||||
.hide_default_value(true)
|
||||
.value_hint(ValueHint::FilePath),
|
||||
)
|
||||
}
|
||||
|
||||
pub fn standalone_checksum_app_with_length(
|
||||
about: impl Into<String>,
|
||||
usage: impl Into<String>,
|
||||
) -> Command {
|
||||
default_checksum_app(about, usage)
|
||||
.with_binary().with_check_and_opts().with_length().with_tag(false).with_text(true).with_zero()
|
||||
}
|
||||
|
||||
pub fn standalone_checksum_app(
|
||||
about: impl Into<String>,
|
||||
usage: impl Into<String>,
|
||||
) -> Command {
|
||||
default_checksum_app(about, usage)
|
||||
.with_binary().with_check_and_opts().with_tag(false).with_text(true).with_zero()
|
||||
}
|
||||
|
||||
pub fn checksum_main(
|
||||
algo: Option<AlgoKind>,
|
||||
length: Option<usize>,
|
||||
matches: ArgMatches,
|
||||
output_format: OutputFormat,
|
||||
) -> UResult<()> {
|
||||
let check = matches.get_flag(options::CHECK);
|
||||
let check_flag = |flag| match (check, matches.get_flag(flag)) {
|
||||
(_, false) => Ok(false),
|
||||
(true, true) => Ok(true),
|
||||
(false, true) => Err(ChecksumError::CheckOnlyFlag(flag.into())),
|
||||
};
|
||||
let ignore_missing = check_flag(options::IGNORE_MISSING)?;
|
||||
let warn = check_flag(options::WARN)?;
|
||||
let quiet = check_flag(options::QUIET)?;
|
||||
let strict = check_flag(options::STRICT)?;
|
||||
let status = check_flag(options::STATUS)?;
|
||||
let text_flag = matches.get_flag(options::TEXT);
|
||||
let binary_flag = matches.get_flag(options::BINARY);
|
||||
let tag = matches.get_flag(options::TAG);
|
||||
let files = matches.get_many::<OsString>(options::FILE).unwrap().map(Borrow::borrow);
|
||||
|
||||
if text_flag && tag { return Err(ChecksumError::TextAfterTag.into()); }
|
||||
if check {
|
||||
if algo.is_some_and(AlgoKind::is_legacy) { return Err(ChecksumError::AlgorithmNotSupportedWithCheck.into()); }
|
||||
if tag { return Err(ChecksumError::TagCheck.into()); }
|
||||
if binary_flag || text_flag { return Err(ChecksumError::BinaryTextConflict.into()); }
|
||||
let opts = ChecksumValidateOptions {
|
||||
ignore_missing,
|
||||
strict,
|
||||
verbose: ChecksumVerbose::new(status, quiet, warn),
|
||||
};
|
||||
return validate::perform_checksum_validation(files, algo, length, opts);
|
||||
}
|
||||
|
||||
let algo = SizedAlgoKind::from_unsized(algo.unwrap_or(AlgoKind::Crc), length)?;
|
||||
let opts = ChecksumComputeOptions {
|
||||
algo_kind: algo,
|
||||
output_format,
|
||||
line_ending: LineEnding::from_zero_flag(matches.get_flag(options::ZERO)),
|
||||
};
|
||||
compute::perform_checksum_computation(opts, files)
|
||||
}
|
||||
+995
@@ -0,0 +1,995 @@
|
||||
// This file is part of the uutils coreutils package.
|
||||
//
|
||||
// For the full copyright and license information, please view the LICENSE
|
||||
// file that was distributed with this source code.
|
||||
|
||||
// spell-checker:ignore rsplit hexdigit bitlen invalidchecksum inva idchecksum xffname
|
||||
|
||||
use std::ffi::OsStr;
|
||||
use std::fmt::Display;
|
||||
use std::fs::File;
|
||||
use std::io::{self, BufReader, Read, Write};
|
||||
|
||||
use os_display::Quotable;
|
||||
|
||||
use uucore::checksum::{
|
||||
AlgoKind, BlakeLength, ChecksumError, ReadingMode, ShaLength, SizedAlgoKind, digest_reader,
|
||||
parse_blake_length, unescape_filename,
|
||||
};
|
||||
use uucore::error::{FromIo, UError, UIoError, UResult, USimpleError};
|
||||
use uucore::quoting_style::{QuotingStyle, locale_aware_escape_name};
|
||||
use uucore::sum::{self, Blake2b, Blake3, DigestOutput};
|
||||
use uucore::{os_str_as_bytes, os_str_from_bytes, read_os_string_lines};
|
||||
use crate::{command_name, report_error, report_warning};
|
||||
|
||||
/// To what level should checksum validation print logging info.
|
||||
#[derive(Debug, PartialEq, Eq, PartialOrd, Clone, Copy, Default)]
|
||||
pub enum ChecksumVerbose {
|
||||
Status,
|
||||
Quiet,
|
||||
#[default]
|
||||
Normal,
|
||||
Warning,
|
||||
}
|
||||
|
||||
impl ChecksumVerbose {
|
||||
pub fn new(status: bool, quiet: bool, warn: bool) -> Self {
|
||||
use ChecksumVerbose::*;
|
||||
|
||||
// Assume only one of the three booleans will be enabled at once.
|
||||
// This is ensured by clap's overriding arguments.
|
||||
match (status, quiet, warn) {
|
||||
(true, _, _) => Status,
|
||||
(_, true, _) => Quiet,
|
||||
(_, _, true) => Warning,
|
||||
_ => Normal,
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn over_status(self) -> bool {
|
||||
self > Self::Status
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn over_quiet(self) -> bool {
|
||||
self > Self::Quiet
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn at_least_warning(self) -> bool {
|
||||
self >= Self::Warning
|
||||
}
|
||||
}
|
||||
|
||||
/// This struct regroups CLI flags.
|
||||
#[derive(Debug, Default, Clone, Copy)]
|
||||
pub struct ChecksumValidateOptions {
|
||||
pub ignore_missing: bool,
|
||||
pub strict: bool,
|
||||
pub verbose: ChecksumVerbose,
|
||||
}
|
||||
|
||||
/// This structure holds the count of checksum test lines' outcomes.
|
||||
#[derive(Default)]
|
||||
struct ChecksumResult {
|
||||
/// Number of lines in the file where the computed checksum MATCHES
|
||||
/// the expectation.
|
||||
pub correct: u32,
|
||||
/// Number of lines in the file where the computed checksum DIFFERS
|
||||
/// from the expectation.
|
||||
pub failed_cksum: u32,
|
||||
pub failed_open_file: u32,
|
||||
/// Number of improperly formatted lines.
|
||||
pub bad_format: u32,
|
||||
/// Total number of non-empty, non-comment lines.
|
||||
pub total: u32,
|
||||
}
|
||||
|
||||
impl ChecksumResult {
|
||||
#[inline]
|
||||
fn total_properly_formatted(&self) -> u32 {
|
||||
self.total - self.bad_format
|
||||
}
|
||||
}
|
||||
|
||||
/// Represents a reason for which the processing of a checksum line
|
||||
/// could not proceed to digest comparison.
|
||||
enum LineCheckError {
|
||||
/// a generic UError was encountered in sub-functions
|
||||
UError(Box<dyn UError>),
|
||||
/// the computed checksum digest differs from the expected one
|
||||
DigestMismatch,
|
||||
/// the line is empty or is a comment
|
||||
Skipped,
|
||||
/// the line has a formatting error
|
||||
ImproperlyFormatted,
|
||||
/// file exists but is impossible to read
|
||||
CantOpenFile,
|
||||
/// there is nothing at the given path
|
||||
FileNotFound,
|
||||
/// the given path leads to a directory
|
||||
FileIsDirectory,
|
||||
}
|
||||
|
||||
impl From<Box<dyn UError>> for LineCheckError {
|
||||
fn from(value: Box<dyn UError>) -> Self {
|
||||
Self::UError(value)
|
||||
}
|
||||
}
|
||||
|
||||
impl From<ChecksumError> for LineCheckError {
|
||||
fn from(value: ChecksumError) -> Self {
|
||||
Self::UError(Box::new(value))
|
||||
}
|
||||
}
|
||||
|
||||
/// Represents an error that was encountered when processing a checksum file.
|
||||
enum FileCheckError {
|
||||
/// a generic UError was encountered in sub-functions
|
||||
UError(Box<dyn UError>),
|
||||
/// reading of the checksum file failed
|
||||
CantOpenChecksumFile,
|
||||
/// processing of the file is considered as a failure regarding the
|
||||
/// provided flags. This however does not stop the processing of
|
||||
/// further files.
|
||||
Failed,
|
||||
}
|
||||
|
||||
impl From<Box<dyn UError>> for FileCheckError {
|
||||
fn from(value: Box<dyn UError>) -> Self {
|
||||
Self::UError(value)
|
||||
}
|
||||
}
|
||||
|
||||
impl From<ChecksumError> for FileCheckError {
|
||||
fn from(value: ChecksumError) -> Self {
|
||||
Self::UError(Box::new(value))
|
||||
}
|
||||
}
|
||||
|
||||
fn print_cksum_report(res: &ChecksumResult) {
|
||||
if res.bad_format > 0 {
|
||||
report_warning(&format!("WARNING: {} line(s) are improperly formatted", res.bad_format));
|
||||
}
|
||||
|
||||
if res.failed_cksum > 0 {
|
||||
report_warning(&format!("WARNING: {} computed checksum(s) did NOT match", res.failed_cksum));
|
||||
}
|
||||
|
||||
if res.failed_open_file > 0 {
|
||||
report_warning(&format!("WARNING: {} listed file(s) could not be read", res.failed_open_file));
|
||||
}
|
||||
}
|
||||
|
||||
/// Print a "no properly formatted lines" message in stderr
|
||||
#[inline]
|
||||
fn log_no_properly_formatted(filename: impl Display) {
|
||||
let _ = writeln!(
|
||||
pi_uutils_ctx::stderr(),
|
||||
"{}: {}",
|
||||
command_name(),
|
||||
format!("{}: no properly formatted checksum lines found", filename)
|
||||
);
|
||||
}
|
||||
|
||||
/// Print a "no file was verified" message in stderr
|
||||
#[inline]
|
||||
fn log_no_file_verified(filename: impl Display) {
|
||||
let _ = writeln!(
|
||||
pi_uutils_ctx::stderr(),
|
||||
"{}: {}",
|
||||
command_name(),
|
||||
format!("{}: no file was verified", filename)
|
||||
);
|
||||
}
|
||||
|
||||
/// Represents the different outcomes that can happen to a file
|
||||
/// that is being checked.
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
enum FileChecksumResult {
|
||||
Ok,
|
||||
Failed,
|
||||
CantOpen,
|
||||
}
|
||||
|
||||
impl FileChecksumResult {
|
||||
/// Creates a `FileChecksumResult` from a digest comparison that
|
||||
/// either succeeded or failed.
|
||||
fn from_bool(checksum_correct: bool) -> Self {
|
||||
if checksum_correct {
|
||||
Self::Ok
|
||||
} else {
|
||||
Self::Failed
|
||||
}
|
||||
}
|
||||
|
||||
/// The cli options might prevent to display on the outcome of the
|
||||
/// comparison on STDOUT.
|
||||
fn can_display(self, verbose: ChecksumVerbose) -> bool {
|
||||
match self {
|
||||
Self::Ok => verbose.over_quiet(),
|
||||
Self::Failed => verbose.over_status(),
|
||||
Self::CantOpen => true,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl Display for FileChecksumResult {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
match self {
|
||||
Self::Ok => write!(f, "OK"),
|
||||
Self::Failed => write!(f, "FAILED"),
|
||||
Self::CantOpen => write!(f, "FAILED open or read"),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Write to the given buffer the checksum validation status of a file which
|
||||
/// name might contain non-utf-8 characters.
|
||||
fn write_file_report<W: Write>(
|
||||
mut w: W,
|
||||
filename: &[u8],
|
||||
result: FileChecksumResult,
|
||||
prefix: &str,
|
||||
verbose: ChecksumVerbose,
|
||||
) {
|
||||
if result.can_display(verbose) {
|
||||
let _ = write!(w, "{prefix}");
|
||||
let _ = w.write_all(filename);
|
||||
let _ = writeln!(w, ": {result}");
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, PartialEq, Eq, Clone, Copy)]
|
||||
enum LineFormat {
|
||||
AlgoBased,
|
||||
SingleSpace,
|
||||
Untagged,
|
||||
}
|
||||
|
||||
impl LineFormat {
|
||||
/// parse [tagged output format]
|
||||
/// Normally the format is simply space separated but openssl does not
|
||||
/// respect the gnu definition.
|
||||
///
|
||||
/// [tagged output format]: https://www.gnu.org/software/coreutils/manual/html_node/cksum-output-modes.html#cksum-output-modes-1
|
||||
fn parse_algo_based(line: &[u8]) -> Option<LineInfo> {
|
||||
// r"\MD5 (a\\ b) = abc123",
|
||||
// BLAKE2b(44)= a45a4c4883cce4b50d844fab460414cc2080ca83690e74d850a9253e757384366382625b218c8585daee80f34dc9eb2f2fde5fb959db81cd48837f9216e7b0fa
|
||||
let trimmed = line.trim_ascii_start();
|
||||
let algo_start = usize::from(trimmed.starts_with(b"\\"));
|
||||
let rest = &trimmed[algo_start..];
|
||||
|
||||
enum SubCase {
|
||||
Posix,
|
||||
OpenSSL,
|
||||
}
|
||||
// find the next parenthesis using byte search (not next whitespace) because openssl's
|
||||
// tagged format does not put a space before (filename)
|
||||
|
||||
let par_idx = rest.iter().position(|&b| b == b'(')?;
|
||||
let sub_case = if rest[par_idx - 1] == b' ' {
|
||||
SubCase::Posix
|
||||
} else {
|
||||
SubCase::OpenSSL
|
||||
};
|
||||
|
||||
let algo_substring = match sub_case {
|
||||
SubCase::Posix => &rest[..par_idx - 1],
|
||||
SubCase::OpenSSL => &rest[..par_idx],
|
||||
};
|
||||
let mut algo_parts = algo_substring.splitn(2, |&b| b == b'-');
|
||||
let algo = algo_parts.next()?;
|
||||
|
||||
// Parse algo_bits if present
|
||||
let algo_bits = algo_parts
|
||||
.next()
|
||||
.and_then(|s| std::str::from_utf8(s).ok()?.parse::<usize>().ok());
|
||||
|
||||
// Check algo format: uppercase ASCII or digits or "BLAKE2b"
|
||||
let is_valid_algo = algo == b"BLAKE2b"
|
||||
|| algo
|
||||
.iter()
|
||||
.all(|&b| b.is_ascii_uppercase() || b.is_ascii_digit());
|
||||
if !is_valid_algo {
|
||||
return None;
|
||||
}
|
||||
// SAFETY: we just validated the contents of algo, we can unsafely make a
|
||||
// String from it
|
||||
let algo_utf8 = unsafe { String::from_utf8_unchecked(algo.to_vec()) };
|
||||
// stripping '(' not ' (' since we matched on ( not whitespace because of openssl.
|
||||
let after_paren = rest.get(par_idx + 1..)?;
|
||||
let (filename, checksum) = match sub_case {
|
||||
SubCase::Posix => ByteSliceExt::rsplit_once(after_paren, b") = ")?,
|
||||
SubCase::OpenSSL => ByteSliceExt::rsplit_once(after_paren, b")= ")?,
|
||||
};
|
||||
|
||||
let checksum_utf8 = Self::validate_checksum_format(checksum)?;
|
||||
|
||||
Some(LineInfo {
|
||||
algo_name: Some(algo_utf8),
|
||||
algo_bit_len: algo_bits,
|
||||
checksum: checksum_utf8,
|
||||
filename: filename.to_vec(),
|
||||
format: Self::AlgoBased,
|
||||
})
|
||||
}
|
||||
|
||||
#[allow(rustdoc::invalid_html_tags)]
|
||||
/// parse [untagged output format]
|
||||
/// The format is simple, either "<checksum> <filename>" or
|
||||
/// "<checksum> *<filename>"
|
||||
///
|
||||
/// [untagged output format]: https://www.gnu.org/software/coreutils/manual/html_node/cksum-output-modes.html#cksum-output-modes-1
|
||||
fn parse_untagged(line: &[u8]) -> Option<LineInfo> {
|
||||
let space_idx = line.iter().position(|&b| b == b' ')?;
|
||||
let checksum = &line[..space_idx];
|
||||
|
||||
let checksum_utf8 = Self::validate_checksum_format(checksum)?;
|
||||
|
||||
let rest = &line[space_idx..];
|
||||
let filename = rest
|
||||
.strip_prefix(b" ")
|
||||
.or_else(|| rest.strip_prefix(b" *"))?;
|
||||
|
||||
Some(LineInfo {
|
||||
algo_name: None,
|
||||
algo_bit_len: None,
|
||||
checksum: checksum_utf8,
|
||||
filename: filename.to_vec(),
|
||||
format: Self::Untagged,
|
||||
})
|
||||
}
|
||||
|
||||
#[allow(rustdoc::invalid_html_tags)]
|
||||
/// parse [untagged output format]
|
||||
/// Normally the format is simple, either "<checksum> <filename>" or
|
||||
/// "<checksum> *<filename>"
|
||||
/// But the bsd tests expect special single space behavior where
|
||||
/// checksum and filename are separated only by a space, meaning the second
|
||||
/// space or asterisk is part of the file name.
|
||||
/// This parser accounts for this variation
|
||||
///
|
||||
/// [untagged output format]: https://www.gnu.org/software/coreutils/manual/html_node/cksum-output-modes.html#cksum-output-modes-1
|
||||
fn parse_single_space(line: &[u8]) -> Option<LineInfo> {
|
||||
// Find first space
|
||||
let space_idx = line.iter().position(|&b| b == b' ')?;
|
||||
let checksum = &line[..space_idx];
|
||||
if !checksum.iter().all(|&b| b.is_ascii_hexdigit()) || checksum.is_empty() {
|
||||
return None;
|
||||
}
|
||||
// SAFETY: we just validated the contents of checksum, we can unsafely make a
|
||||
// String from it
|
||||
let checksum_utf8 = unsafe { String::from_utf8_unchecked(checksum.to_vec()) };
|
||||
|
||||
let filename = line.get(space_idx + 1..)?; // Skip single space
|
||||
|
||||
Some(LineInfo {
|
||||
algo_name: None,
|
||||
algo_bit_len: None,
|
||||
checksum: checksum_utf8,
|
||||
filename: filename.to_vec(),
|
||||
format: Self::SingleSpace,
|
||||
})
|
||||
}
|
||||
|
||||
/// Ensure that the given checksum is syntactically valid (that it is either
|
||||
/// hexadecimal or base64 encoded).
|
||||
fn validate_checksum_format(checksum: &[u8]) -> Option<String> {
|
||||
if checksum.is_empty() {
|
||||
return None;
|
||||
}
|
||||
|
||||
let mut is_base64 = false;
|
||||
|
||||
for index in 0..checksum.len() {
|
||||
match checksum[index..] {
|
||||
// ASCII alphanumeric
|
||||
[b'A'..=b'Z' | b'a'..=b'z' | b'0'..=b'9', ..] => (),
|
||||
// Base64 special character
|
||||
[b'+' | b'/', ..] => is_base64 = true,
|
||||
// Base64 end of string padding
|
||||
[b'='] | [b'=', b'='] | [b'=', b'=', b'='] => {
|
||||
is_base64 = true;
|
||||
break;
|
||||
}
|
||||
// Any other character means the checksum is wrong
|
||||
_ => return None,
|
||||
}
|
||||
}
|
||||
|
||||
// If base64 characters were encountered, make sure the checksum has a
|
||||
// length multiple of 4.
|
||||
//
|
||||
// This check is not enough because it may allow base64-encoded
|
||||
// checksums that are fully alphanumeric. Another check happens later
|
||||
// when we are provided with a length hint to detect ambiguous
|
||||
// base64-encoded checksums.
|
||||
if is_base64 && !checksum.len().is_multiple_of(4) {
|
||||
return None;
|
||||
}
|
||||
|
||||
// SAFETY: we just validated the contents of checksum, we can unsafely make a
|
||||
// String from it
|
||||
Some(unsafe { String::from_utf8_unchecked(checksum.to_vec()) })
|
||||
}
|
||||
}
|
||||
|
||||
// Helper trait for byte slice operations
|
||||
trait ByteSliceExt {
|
||||
/// Look for a pattern from right to left, return surrounding parts if found.
|
||||
fn rsplit_once(&self, pattern: &[u8]) -> Option<(&Self, &Self)>;
|
||||
}
|
||||
|
||||
impl ByteSliceExt for [u8] {
|
||||
fn rsplit_once(&self, pattern: &[u8]) -> Option<(&Self, &Self)> {
|
||||
let pos = self
|
||||
.windows(pattern.len())
|
||||
.rev()
|
||||
.position(|w| w == pattern)?;
|
||||
Some((
|
||||
&self[..self.len() - pattern.len() - pos],
|
||||
&self[self.len() - pos..],
|
||||
))
|
||||
}
|
||||
}
|
||||
|
||||
/// Hold the data extracted from a checksum line.
|
||||
struct LineInfo {
|
||||
algo_name: Option<String>,
|
||||
algo_bit_len: Option<usize>,
|
||||
checksum: String,
|
||||
filename: Vec<u8>,
|
||||
format: LineFormat,
|
||||
}
|
||||
|
||||
impl LineInfo {
|
||||
/// Returns a `LineInfo` parsed from a checksum line.
|
||||
/// The function will run 3 parsers against the line and select the first one that matches
|
||||
/// to populate the fields of the struct.
|
||||
/// However, there is a catch to handle regarding the handling of `cached_line_format`.
|
||||
/// In case of non-algo-based format, if `cached_line_format` is Some, it must take the priority
|
||||
/// over the detected format. Otherwise, we must set it the the detected format.
|
||||
/// This specific behavior is emphasized by the test
|
||||
/// `test_md5sum::test_check_md5sum_only_one_space`.
|
||||
fn parse(s: impl AsRef<OsStr>, cached_line_format: &mut Option<LineFormat>) -> Option<Self> {
|
||||
let line_bytes = os_str_as_bytes(s.as_ref()).ok()?;
|
||||
|
||||
if let Some(info) = LineFormat::parse_algo_based(line_bytes) {
|
||||
return Some(info);
|
||||
}
|
||||
if let Some(cached_format) = cached_line_format {
|
||||
match cached_format {
|
||||
LineFormat::Untagged => LineFormat::parse_untagged(line_bytes),
|
||||
LineFormat::SingleSpace => LineFormat::parse_single_space(line_bytes),
|
||||
LineFormat::AlgoBased => unreachable!("we never catch the algo based format"),
|
||||
}
|
||||
} else if let Some(info) = LineFormat::parse_untagged(line_bytes) {
|
||||
*cached_line_format = Some(LineFormat::Untagged);
|
||||
Some(info)
|
||||
} else if let Some(info) = LineFormat::parse_single_space(line_bytes) {
|
||||
*cached_line_format = Some(LineFormat::SingleSpace);
|
||||
Some(info)
|
||||
} else {
|
||||
None
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Extract the expected digest from the checksum string and decode it
|
||||
fn get_raw_expected_digest(checksum: &str, bit_len_hint: Option<usize>) -> Option<Vec<u8>> {
|
||||
// If the length of the digest is not a multiple of 2, then it must be
|
||||
// improperly formatted (1 byte is 2 hex digits, and base64 strings should
|
||||
// always be a multiple of 4).
|
||||
if !checksum.len().is_multiple_of(2) {
|
||||
return None;
|
||||
}
|
||||
|
||||
let byte_len_hint = bit_len_hint.map(|n| n.div_ceil(8));
|
||||
|
||||
let checks_hint = |len| byte_len_hint.is_none_or(|hint| hint == len);
|
||||
|
||||
// If the length of the string matches the one to be expected (in case it's
|
||||
// given) AND the digest can be decoded as hexadecimal, just go with it.
|
||||
if checks_hint(checksum.len() / 2) {
|
||||
if let Ok(raw_ck) = hex::decode(checksum) {
|
||||
return Some(raw_ck);
|
||||
}
|
||||
}
|
||||
|
||||
// If the checksum cannot be decoded as hexadecimal, interpret it as Base64
|
||||
// instead.
|
||||
|
||||
// But first, verify the encoded checksum length, which should be a
|
||||
// multiple of 4.
|
||||
//
|
||||
// It is important to check it before trying to decode, because the
|
||||
// forgiving mode of decoding will ignore if padding characters '=' are
|
||||
// MISSING, but to match GNU's behavior, we must reject it.
|
||||
if !checksum.len().is_multiple_of(4) {
|
||||
return None;
|
||||
}
|
||||
|
||||
// Perform the decoding and be FORGIVING about it, to allow for checksums
|
||||
// with INVALID padding to still be decoded. This is enforced by
|
||||
// `test_untagged_base64_matching_tag` in `test_cksum.rs`
|
||||
|
||||
base64_simd::forgiving_decode_to_vec(checksum.as_bytes())
|
||||
.ok()
|
||||
.filter(|raw| checks_hint(raw.len()))
|
||||
}
|
||||
|
||||
/// Returns a reader that reads from the specified file, or from stdin if `filename_to_check` is "-".
|
||||
fn get_file_to_check(
|
||||
filename: &OsStr,
|
||||
opts: ChecksumValidateOptions,
|
||||
) -> Result<Box<dyn Read>, LineCheckError> {
|
||||
let filename_bytes = os_str_as_bytes(filename).map_err(|e| LineCheckError::UError(e.into()))?;
|
||||
|
||||
if filename == "-" {
|
||||
Ok(Box::new(pi_uutils_ctx::stdin())) // Use stdin if "-" is specified in the checksum file
|
||||
} else {
|
||||
let failed_open = || {
|
||||
write_file_report(
|
||||
pi_uutils_ctx::stdout(),
|
||||
filename_bytes,
|
||||
FileChecksumResult::CantOpen,
|
||||
"",
|
||||
opts.verbose,
|
||||
);
|
||||
};
|
||||
let print_error = |err: io::Error| {
|
||||
report_error(&err.map_err_context(|| locale_aware_escape_name(filename, QuotingStyle::SHELL_ESCAPE).to_string_lossy().to_string()));
|
||||
};
|
||||
match File::open(pi_uutils_ctx::resolve(filename)) {
|
||||
Ok(f) => {
|
||||
if f.metadata()
|
||||
.map_err(|_| LineCheckError::CantOpenFile)?
|
||||
.is_dir()
|
||||
{
|
||||
print_error(io::Error::new(
|
||||
io::ErrorKind::IsADirectory,
|
||||
"Is a directory",
|
||||
));
|
||||
// also regarded as a failed open
|
||||
failed_open();
|
||||
Err(LineCheckError::FileIsDirectory)
|
||||
} else {
|
||||
Ok(Box::new(f))
|
||||
}
|
||||
}
|
||||
Err(err) => {
|
||||
if !opts.ignore_missing {
|
||||
// yes, we have both stderr and stdout here
|
||||
print_error(err);
|
||||
failed_open();
|
||||
}
|
||||
// we could not open the file but we want to continue
|
||||
Err(LineCheckError::FileNotFound)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Returns a reader to the list of checksums
|
||||
fn get_input_file(filename: &OsStr) -> UResult<Box<dyn Read>> {
|
||||
match File::open(pi_uutils_ctx::resolve(filename)) {
|
||||
Ok(f) => {
|
||||
if f.metadata()?.is_dir() {
|
||||
Err(io::Error::other(
|
||||
format!("{}: Is a directory", filename.maybe_quote()),
|
||||
)
|
||||
.into())
|
||||
} else {
|
||||
Ok(Box::new(f))
|
||||
}
|
||||
}
|
||||
Err(_) => Err(io::Error::other(format!(
|
||||
"{}: {}",
|
||||
filename.maybe_quote(),
|
||||
"No such file or directory"
|
||||
))
|
||||
.into()),
|
||||
}
|
||||
}
|
||||
|
||||
/// Gets the algorithm name and length from the `LineInfo` if the algo-based format is matched.
|
||||
fn identify_algo_name_and_length(
|
||||
line_info: &LineInfo,
|
||||
algo_name_input: Option<AlgoKind>,
|
||||
last_algo: &mut Option<String>,
|
||||
) -> Result<(AlgoKind, Option<usize>), LineCheckError> {
|
||||
use AlgoKind as ak;
|
||||
let algo_from_line = line_info.algo_name.clone().unwrap_or_default();
|
||||
let Ok(line_algo) = AlgoKind::from_cksum(algo_from_line.to_lowercase()) else {
|
||||
// Unknown algorithm
|
||||
return Err(LineCheckError::ImproperlyFormatted);
|
||||
};
|
||||
*last_algo = Some(algo_from_line);
|
||||
|
||||
// check if we are called with XXXsum (example: md5sum) but we detected a
|
||||
// different algo parsing the file (for example SHA1 (f) = d...)
|
||||
//
|
||||
// Also handle the case cksum -s sm3 but the file contains other formats
|
||||
if let Some(algo_name_input) = algo_name_input {
|
||||
match (algo_name_input, line_algo) {
|
||||
(l, r) if l == r => (),
|
||||
// Edge case for SHA2, which matches SHA(224|256|384|512)
|
||||
(ak::Sha2, ak::Sha224 | ak::Sha256 | ak::Sha384 | ak::Sha512) => (),
|
||||
_ => return Err(LineCheckError::ImproperlyFormatted),
|
||||
}
|
||||
}
|
||||
|
||||
let bytes = if let Some(bitlen) = line_info.algo_bit_len {
|
||||
match line_algo {
|
||||
algo @ (ak::Blake2b | ak::Blake3) => {
|
||||
match parse_blake_length(algo, BlakeLength::Int(bitlen)) {
|
||||
Ok(len) => Some(len),
|
||||
Err(_) => return Err(LineCheckError::ImproperlyFormatted),
|
||||
}
|
||||
}
|
||||
ak::Sha2 | ak::Sha3 if [224, 256, 384, 512].contains(&bitlen) => Some(bitlen),
|
||||
ak::Shake128 | ak::Shake256 => Some(bitlen),
|
||||
// Either
|
||||
// the algo based line is provided with a bit length with an
|
||||
// algorithm that does not support it (only Blake2b, Blake3, sha2,
|
||||
// and sha3 do).
|
||||
//
|
||||
// eg: MD5-128 (foo.txt) = fffffffff
|
||||
// ^ This is illegal
|
||||
// OR
|
||||
// the given length is wrong because it's not a multiple of 8.
|
||||
_ => return Err(LineCheckError::ImproperlyFormatted),
|
||||
}
|
||||
} else if line_algo == ak::Blake2b {
|
||||
// Default length with BLAKE2b,
|
||||
Some(Blake2b::DEFAULT_BYTE_SIZE)
|
||||
} else if line_algo == ak::Blake3 {
|
||||
// Default length with BLAKE3,
|
||||
Some(Blake3::DEFAULT_BYTE_SIZE)
|
||||
} else {
|
||||
None
|
||||
};
|
||||
|
||||
Ok((line_algo, bytes))
|
||||
}
|
||||
|
||||
/// Given a filename and an algorithm, compute the digest and compare it with
|
||||
/// the expected one.
|
||||
fn compute_and_check_digest_from_file(
|
||||
filename: &[u8],
|
||||
expected_checksum: &[u8],
|
||||
algo: SizedAlgoKind,
|
||||
opts: ChecksumValidateOptions,
|
||||
) -> Result<(), LineCheckError> {
|
||||
let (filename_to_check_unescaped, prefix) = unescape_filename(filename);
|
||||
let real_filename_to_check = os_str_from_bytes(&filename_to_check_unescaped)?;
|
||||
|
||||
// Open the input file
|
||||
let file_to_check = get_file_to_check(&real_filename_to_check, opts)?;
|
||||
let mut file_reader = BufReader::new(file_to_check);
|
||||
|
||||
// Read the file and calculate the checksum
|
||||
let mut digest = algo.create_digest();
|
||||
|
||||
// Set binary to false because --binary is not supported with --check
|
||||
|
||||
let (calculated_checksum, _) =
|
||||
match digest_reader(&mut digest, &mut file_reader, ReadingMode::Text) {
|
||||
Ok(result) => result,
|
||||
Err(err) => {
|
||||
report_error(&err.map_err_context(|| locale_aware_escape_name(&real_filename_to_check, QuotingStyle::SHELL_ESCAPE).to_string_lossy().to_string()));
|
||||
|
||||
write_file_report(
|
||||
pi_uutils_ctx::stdout(),
|
||||
filename,
|
||||
FileChecksumResult::CantOpen,
|
||||
prefix,
|
||||
opts.verbose,
|
||||
);
|
||||
return Err(LineCheckError::CantOpenFile);
|
||||
}
|
||||
};
|
||||
|
||||
// Do the checksum validation
|
||||
let checksum_correct = match calculated_checksum {
|
||||
DigestOutput::Vec(data) => data == expected_checksum,
|
||||
DigestOutput::Crc(n) => n.to_be_bytes() == expected_checksum,
|
||||
DigestOutput::U16(n) => n.to_be_bytes() == expected_checksum,
|
||||
};
|
||||
write_file_report(
|
||||
pi_uutils_ctx::stdout(),
|
||||
filename,
|
||||
FileChecksumResult::from_bool(checksum_correct),
|
||||
prefix,
|
||||
opts.verbose,
|
||||
);
|
||||
|
||||
if checksum_correct {
|
||||
Ok(())
|
||||
} else {
|
||||
Err(LineCheckError::DigestMismatch)
|
||||
}
|
||||
}
|
||||
|
||||
/// Check a digest checksum with non-algo based pre-treatment.
|
||||
fn process_algo_based_line(
|
||||
line_info: &LineInfo,
|
||||
cli_algo_kind: Option<AlgoKind>,
|
||||
opts: ChecksumValidateOptions,
|
||||
last_algo: &mut Option<String>,
|
||||
) -> Result<(), LineCheckError> {
|
||||
let filename_to_check = line_info.filename.as_slice();
|
||||
|
||||
let (algo_kind, algo_len) = identify_algo_name_and_length(line_info, cli_algo_kind, last_algo)?;
|
||||
|
||||
// If the digest bitlen is known, we can check the format of the expected
|
||||
// checksum with it.
|
||||
let digest_bit_length_hint = match (algo_kind, algo_len) {
|
||||
(AlgoKind::Blake2b | AlgoKind::Blake3, Some(byte_len)) => Some(byte_len * 8),
|
||||
(AlgoKind::Shake128 | AlgoKind::Shake256, Some(bit_len)) => Some(bit_len),
|
||||
(AlgoKind::Shake128, None) => Some(sum::Shake128::DEFAULT_BIT_SIZE),
|
||||
(AlgoKind::Shake256, None) => Some(sum::Shake256::DEFAULT_BIT_SIZE),
|
||||
_ => None,
|
||||
};
|
||||
|
||||
let expected_checksum = get_raw_expected_digest(&line_info.checksum, digest_bit_length_hint)
|
||||
.ok_or(LineCheckError::ImproperlyFormatted)?;
|
||||
|
||||
let algo = SizedAlgoKind::from_unsized(algo_kind, algo_len)
|
||||
.map_err(|_| LineCheckError::ImproperlyFormatted)?;
|
||||
|
||||
compute_and_check_digest_from_file(filename_to_check, &expected_checksum, algo, opts)
|
||||
}
|
||||
|
||||
/// Check a digest checksum with non-algo based pre-treatment.
|
||||
fn process_non_algo_based_line(
|
||||
line_number: usize,
|
||||
line_info: &LineInfo,
|
||||
cli_algo_kind: AlgoKind,
|
||||
cli_algo_length: Option<usize>,
|
||||
opts: ChecksumValidateOptions,
|
||||
) -> Result<(), LineCheckError> {
|
||||
use AlgoKind as ak;
|
||||
let mut filename_to_check = line_info.filename.as_slice();
|
||||
if filename_to_check.starts_with(b"*")
|
||||
&& line_number == 0
|
||||
&& line_info.format == LineFormat::SingleSpace
|
||||
{
|
||||
// Remove the leading asterisk if present - only for the first line
|
||||
filename_to_check = &filename_to_check[1..];
|
||||
}
|
||||
|
||||
let expected_digest_sum = cli_algo_kind.expected_digest_bit_len();
|
||||
let expected_checksum = get_raw_expected_digest(&line_info.checksum, expected_digest_sum)
|
||||
.ok_or(LineCheckError::ImproperlyFormatted)?;
|
||||
|
||||
// When a specific algorithm name is input, use it and use the provided
|
||||
// bits except when dealing with blake2b, sha2 and sha3, where we will
|
||||
// detect the length.
|
||||
let algo_byte_len = match cli_algo_kind {
|
||||
ak::Blake2b | ak::Blake3 => Some(expected_checksum.len()),
|
||||
ak::Sha2 | ak::Sha3 => {
|
||||
// multiplication by 8 to get the number of bits
|
||||
Some(
|
||||
ShaLength::try_from(expected_checksum.len() * 8)
|
||||
.map_err(|_| LineCheckError::ImproperlyFormatted)?
|
||||
.as_usize(),
|
||||
)
|
||||
}
|
||||
_ => cli_algo_length,
|
||||
};
|
||||
|
||||
let algo = SizedAlgoKind::from_unsized(cli_algo_kind, algo_byte_len)?;
|
||||
|
||||
compute_and_check_digest_from_file(filename_to_check, &expected_checksum, algo, opts)
|
||||
}
|
||||
|
||||
/// Parses a checksum line, detect the algorithm to use, read the file and produce
|
||||
/// its digest, and compare it to the expected value.
|
||||
///
|
||||
/// Returns `Ok(bool)` if the comparison happened, bool indicates if the digest
|
||||
/// matched the expected.
|
||||
/// If the comparison didn't happen, return a `LineChecksumError`.
|
||||
fn process_checksum_line(
|
||||
line: &OsStr,
|
||||
i: usize,
|
||||
cli_algo_name: Option<AlgoKind>,
|
||||
cli_algo_length: Option<usize>,
|
||||
opts: ChecksumValidateOptions,
|
||||
cached_line_format: &mut Option<LineFormat>,
|
||||
last_algo: &mut Option<String>,
|
||||
) -> Result<(), LineCheckError> {
|
||||
let line_bytes = os_str_as_bytes(line).map_err(|e| LineCheckError::UError(Box::new(e)))?;
|
||||
|
||||
// Early return on empty or commented lines.
|
||||
if line.is_empty() || line_bytes.starts_with(b"#") {
|
||||
return Err(LineCheckError::Skipped);
|
||||
}
|
||||
|
||||
// Use `LineInfo` to extract the data of a line.
|
||||
// Then, depending on its format, apply a different pre-treatment.
|
||||
let Some(line_info) = LineInfo::parse(line, cached_line_format) else {
|
||||
return Err(LineCheckError::ImproperlyFormatted);
|
||||
};
|
||||
|
||||
if line_info.format == LineFormat::AlgoBased {
|
||||
process_algo_based_line(&line_info, cli_algo_name, opts, last_algo)
|
||||
} else if let Some(cli_algo) = cli_algo_name {
|
||||
// If we match a non-algo based parser, we expect a cli argument
|
||||
// to give us the algorithm to use
|
||||
process_non_algo_based_line(i, &line_info, cli_algo, cli_algo_length, opts)
|
||||
} else {
|
||||
// We have no clue of what algorithm to use
|
||||
Err(LineCheckError::ImproperlyFormatted)
|
||||
}
|
||||
}
|
||||
|
||||
fn process_checksum_file(
|
||||
filename_input: &OsStr,
|
||||
cli_algo_kind: Option<AlgoKind>,
|
||||
cli_algo_length: Option<usize>,
|
||||
opts: ChecksumValidateOptions,
|
||||
) -> Result<(), FileCheckError> {
|
||||
let mut res = ChecksumResult::default();
|
||||
|
||||
let input_is_stdin = filename_input == OsStr::new("-");
|
||||
|
||||
let file: Box<dyn Read> = if input_is_stdin {
|
||||
// Use stdin if "-" is specified
|
||||
Box::new(pi_uutils_ctx::stdin())
|
||||
} else {
|
||||
match get_input_file(filename_input) {
|
||||
Ok(f) => f,
|
||||
Err(e) => {
|
||||
// Could not read the file, show the error and continue to the next file
|
||||
let _ = writeln!(pi_uutils_ctx::stderr(), "{}: {e}", command_name());
|
||||
return Err(FileCheckError::CantOpenChecksumFile);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
let reader = BufReader::new(file);
|
||||
|
||||
// cached_line_format is used to ensure that several non algo-based checksum line
|
||||
// will use the same parser.
|
||||
let mut cached_line_format = None;
|
||||
// last_algo caches the algorithm used in the last line to print a warning
|
||||
// message for the current line if improperly formatted.
|
||||
// Behavior tested in gnu_cksum_c::test_warn
|
||||
let mut last_algo = None;
|
||||
|
||||
for (i, line_res) in read_os_string_lines(reader).enumerate() {
|
||||
let line = line_res.map_err(|e| {
|
||||
USimpleError::new(
|
||||
UIoError::from(e).code(),
|
||||
format!("{}: read error", filename_input.maybe_quote()),
|
||||
)
|
||||
})?;
|
||||
|
||||
let line_result = process_checksum_line(
|
||||
&line,
|
||||
i,
|
||||
cli_algo_kind,
|
||||
cli_algo_length,
|
||||
opts,
|
||||
&mut cached_line_format,
|
||||
&mut last_algo,
|
||||
);
|
||||
|
||||
// Match a first time to elude critical UErrors, and increment the total
|
||||
// in all cases except on skipped.
|
||||
use LineCheckError::*;
|
||||
match line_result {
|
||||
Err(UError(e)) => return Err(e.into()),
|
||||
Err(Skipped) => (),
|
||||
_ => res.total += 1,
|
||||
}
|
||||
|
||||
// Match a second time to update the right field of `res`.
|
||||
match line_result {
|
||||
Ok(()) => res.correct += 1,
|
||||
Err(DigestMismatch) => res.failed_cksum += 1,
|
||||
Err(ImproperlyFormatted) => {
|
||||
res.bad_format += 1;
|
||||
|
||||
if opts.verbose.at_least_warning() {
|
||||
let algo = if let Some(algo_name_input) = cli_algo_kind {
|
||||
algo_name_input.to_uppercase()
|
||||
} else if let Some(algo) = &last_algo {
|
||||
algo.as_str()
|
||||
} else {
|
||||
"Unknown algorithm"
|
||||
};
|
||||
let _ = writeln!(
|
||||
pi_uutils_ctx::stderr(),
|
||||
"{}: {}",
|
||||
command_name(),
|
||||
format!("{}: line {}: improperly formatted {} checksum line", filename_input.maybe_quote(), i + 1, algo)
|
||||
);
|
||||
}
|
||||
}
|
||||
Err(CantOpenFile | FileIsDirectory) => res.failed_open_file += 1,
|
||||
Err(FileNotFound) if !opts.ignore_missing => res.failed_open_file += 1,
|
||||
_ => (),
|
||||
}
|
||||
}
|
||||
|
||||
let filename_display = || {
|
||||
if input_is_stdin {
|
||||
"standard input".maybe_quote()
|
||||
} else {
|
||||
filename_input.maybe_quote()
|
||||
}
|
||||
};
|
||||
|
||||
// not a single line correctly formatted found
|
||||
// return an error
|
||||
if res.total_properly_formatted() == 0 {
|
||||
if opts.verbose.over_status() {
|
||||
log_no_properly_formatted(filename_display());
|
||||
}
|
||||
return Err(FileCheckError::Failed);
|
||||
}
|
||||
|
||||
// if any incorrectly formatted line, show it
|
||||
if opts.verbose.over_status() {
|
||||
print_cksum_report(&res);
|
||||
}
|
||||
|
||||
if opts.ignore_missing && res.correct == 0 {
|
||||
// we have only bad format
|
||||
// and we had ignore-missing
|
||||
if opts.verbose.over_status() {
|
||||
log_no_file_verified(filename_display());
|
||||
}
|
||||
return Err(FileCheckError::Failed);
|
||||
}
|
||||
|
||||
// strict means that we should have an exit code.
|
||||
if opts.strict && res.bad_format > 0 {
|
||||
return Err(FileCheckError::Failed);
|
||||
}
|
||||
|
||||
// If a file was missing, return an error unless we explicitly ignore it.
|
||||
if res.failed_open_file > 0 && !opts.ignore_missing {
|
||||
return Err(FileCheckError::Failed);
|
||||
}
|
||||
|
||||
// Obviously, if a checksum failed at some point, report the error.
|
||||
if res.failed_cksum > 0 {
|
||||
return Err(FileCheckError::Failed);
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Do the checksum validation (can be strict or not)
|
||||
pub fn perform_checksum_validation<'a, I>(
|
||||
files: I,
|
||||
algo_kind: Option<AlgoKind>,
|
||||
length_input: Option<usize>,
|
||||
opts: ChecksumValidateOptions,
|
||||
) -> UResult<()>
|
||||
where
|
||||
I: Iterator<Item = &'a OsStr>,
|
||||
{
|
||||
let mut failed = false;
|
||||
|
||||
// if cksum has several input files, it will print the result for each file
|
||||
for filename_input in files {
|
||||
use FileCheckError::*;
|
||||
match process_checksum_file(filename_input, algo_kind, length_input, opts) {
|
||||
Err(UError(e)) => return Err(e),
|
||||
Err(Failed | CantOpenChecksumFile) => failed = true,
|
||||
Ok(_) => (),
|
||||
}
|
||||
}
|
||||
|
||||
if failed {
|
||||
Err(USimpleError::new(1, ""))
|
||||
} else {
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
Vendored
+16
@@ -0,0 +1,16 @@
|
||||
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/comm), patched for
|
||||
# in-process embedding through pi-uutils-ctx.
|
||||
[package]
|
||||
name = "uu_comm"
|
||||
version = "0.8.0"
|
||||
edition = "2024"
|
||||
license = "MIT"
|
||||
description = "comm ~ (uutils) compare two sorted files line by line (vendored + patched for in-process embedding)"
|
||||
|
||||
[lib]
|
||||
path = "src/comm.rs"
|
||||
|
||||
[dependencies]
|
||||
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
|
||||
uucore = "0.8.0"
|
||||
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
|
||||
Vendored
+19
@@ -0,0 +1,19 @@
|
||||
Copyright (c) uutils developers
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
this software and associated documentation files (the "Software"), to deal in
|
||||
the Software without restriction, including without limitation the rights to
|
||||
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
|
||||
the Software, and to permit persons to whom the Software is furnished to do so,
|
||||
subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
Vendored
+204
@@ -0,0 +1,204 @@
|
||||
// This file is part of the uutils coreutils package.
|
||||
//
|
||||
// For the full copyright and license information, please view the LICENSE
|
||||
// file that was distributed with this source code.
|
||||
// Vendored from uutils/coreutils 0.8.0 and patched for pi-uutils context I/O.
|
||||
|
||||
use std::cmp::Ordering;
|
||||
use std::ffi::{OsStr, OsString};
|
||||
use std::fs::{self, File};
|
||||
use std::io::{self, BufRead, BufReader, BufWriter, Read, Write};
|
||||
use std::path::Path;
|
||||
|
||||
use clap::{Arg, ArgAction, ArgMatches, Command};
|
||||
use pi_uutils_ctx::format_usage;
|
||||
use uucore::display::Quotable;
|
||||
use uucore::error::{FromIo, UResult, USimpleError};
|
||||
use uucore::line_ending::LineEnding;
|
||||
|
||||
mod options {
|
||||
pub const COLUMN_1: &str = "1";
|
||||
pub const COLUMN_2: &str = "2";
|
||||
pub const COLUMN_3: &str = "3";
|
||||
pub const DELIMITER: &str = "output-delimiter";
|
||||
pub const FILE_1: &str = "FILE1";
|
||||
pub const FILE_2: &str = "FILE2";
|
||||
pub const TOTAL: &str = "total";
|
||||
pub const ZERO_TERMINATED: &str = "zero-terminated";
|
||||
pub const CHECK_ORDER: &str = "check-order";
|
||||
pub const NO_CHECK_ORDER: &str = "nocheck-order";
|
||||
}
|
||||
|
||||
#[derive(Clone, Copy)]
|
||||
enum FileNumber { One, Two }
|
||||
impl FileNumber {
|
||||
fn as_str(self) -> &'static str { match self { Self::One => "1", Self::Two => "2" } }
|
||||
}
|
||||
|
||||
struct OrderChecker {
|
||||
last_line: Vec<u8>,
|
||||
file_num: FileNumber,
|
||||
check_order: bool,
|
||||
has_error: bool,
|
||||
}
|
||||
impl OrderChecker {
|
||||
fn new(file_num: FileNumber, check_order: bool) -> Self {
|
||||
Self { last_line: Vec::new(), file_num, check_order, has_error: false }
|
||||
}
|
||||
fn verify_order(&mut self, line: &[u8]) -> bool {
|
||||
if self.last_line.is_empty() {
|
||||
self.last_line = line.to_vec();
|
||||
return true;
|
||||
}
|
||||
let ordered = line >= self.last_line.as_slice();
|
||||
if !ordered && !self.has_error {
|
||||
let _ = writeln!(pi_uutils_ctx::stderr(), "comm: file {} is not in sorted order", self.file_num.as_str());
|
||||
self.has_error = true;
|
||||
}
|
||||
self.last_line.clear();
|
||||
self.last_line.extend_from_slice(line);
|
||||
ordered || !self.check_order
|
||||
}
|
||||
}
|
||||
|
||||
struct LineReader {
|
||||
line_ending: u8,
|
||||
input: Box<dyn BufRead>,
|
||||
}
|
||||
impl LineReader {
|
||||
fn new(input: Box<dyn BufRead>, line_ending: LineEnding) -> Self {
|
||||
Self { input, line_ending: line_ending.into() }
|
||||
}
|
||||
fn read_line(&mut self, buf: &mut Vec<u8>) -> io::Result<usize> {
|
||||
let result = self.input.read_until(self.line_ending, buf)?;
|
||||
if result != 0 && !buf.ends_with(&[self.line_ending]) { buf.push(self.line_ending); }
|
||||
Ok(result)
|
||||
}
|
||||
}
|
||||
|
||||
fn files_identical(path1: &Path, path2: &Path) -> io::Result<bool> {
|
||||
let m1 = fs::metadata(path1)?;
|
||||
let m2 = fs::metadata(path2)?;
|
||||
if !m1.is_file() || !m2.is_file() || m1.len() != m2.len() { return Ok(false); }
|
||||
let mut a = BufReader::new(File::open(path1)?);
|
||||
let mut b = BufReader::new(File::open(path2)?);
|
||||
let mut ba = [0; 8192];
|
||||
let mut bb = [0; 8192];
|
||||
loop {
|
||||
let na = loop { match a.read(&mut ba) { Err(e) if e.kind() == io::ErrorKind::Interrupted => {}, r => break r? } };
|
||||
let nb = loop { match b.read(&mut bb) { Err(e) if e.kind() == io::ErrorKind::Interrupted => {}, r => break r? } };
|
||||
if na != nb || ba[..na] != bb[..nb] { return Ok(false); }
|
||||
if na == 0 { return Ok(true); }
|
||||
}
|
||||
}
|
||||
|
||||
fn write_delimited(writer: &mut impl Write, delim: &[u8], line: &[u8]) -> UResult<()> {
|
||||
writer.write_all(delim).map_err_context(|| "write error".to_string())?;
|
||||
writer.write_all(line).map_err_context(|| "write error".to_string())
|
||||
}
|
||||
|
||||
fn compare(a: &mut LineReader, b: &mut LineReader, name1: &OsStr, name2: &OsStr, delim: &str, opts: &ArgMatches, identical: bool) -> UResult<bool> {
|
||||
let col2 = delim.repeat(usize::from(!opts.get_flag(options::COLUMN_1)));
|
||||
let col3 = delim.repeat(usize::from(!opts.get_flag(options::COLUMN_1)) + usize::from(!opts.get_flag(options::COLUMN_2)));
|
||||
let mut writer = BufWriter::new(pi_uutils_ctx::stdout());
|
||||
let (mut ra, mut rb) = (Vec::new(), Vec::new());
|
||||
let mut na = a.read_line(&mut ra).map_err_context(|| name1.maybe_quote().to_string())?;
|
||||
let mut nb = b.read_line(&mut rb).map_err_context(|| name2.maybe_quote().to_string())?;
|
||||
let (mut n1, mut n2, mut n3) = (0usize, 0usize, 0usize);
|
||||
let explicit = opts.get_flag(options::CHECK_ORDER);
|
||||
let should_check = !opts.get_flag(options::NO_CHECK_ORDER) && (explicit || !identical);
|
||||
let (mut c1, mut c2) = (OrderChecker::new(FileNumber::One, explicit), OrderChecker::new(FileNumber::Two, explicit));
|
||||
let mut delayed_error = false;
|
||||
while na != 0 || nb != 0 {
|
||||
let ord = match (na, nb) { (0, _) => Ordering::Greater, (_, 0) => Ordering::Less, _ => ra.cmp(&rb) };
|
||||
match ord {
|
||||
Ordering::Less => {
|
||||
if should_check && !c1.verify_order(&ra) { break; }
|
||||
if !opts.get_flag(options::COLUMN_1) { writer.write_all(&ra).map_err_context(|| "write error".to_string())?; }
|
||||
ra.clear(); na = a.read_line(&mut ra).map_err_context(|| name1.maybe_quote().to_string())?; n1 += 1;
|
||||
},
|
||||
Ordering::Greater => {
|
||||
if should_check && !c2.verify_order(&rb) { break; }
|
||||
if !opts.get_flag(options::COLUMN_2) { write_delimited(&mut writer, col2.as_bytes(), &rb)?; }
|
||||
rb.clear(); nb = b.read_line(&mut rb).map_err_context(|| name2.maybe_quote().to_string())?; n2 += 1;
|
||||
},
|
||||
Ordering::Equal => {
|
||||
if should_check && (!c1.verify_order(&ra) || !c2.verify_order(&rb)) { break; }
|
||||
if !opts.get_flag(options::COLUMN_3) { write_delimited(&mut writer, col3.as_bytes(), &ra)?; }
|
||||
ra.clear(); rb.clear();
|
||||
na = a.read_line(&mut ra).map_err_context(|| name1.maybe_quote().to_string())?;
|
||||
nb = b.read_line(&mut rb).map_err_context(|| name2.maybe_quote().to_string())?; n3 += 1;
|
||||
},
|
||||
}
|
||||
if (c1.has_error || c2.has_error) && !explicit { delayed_error = true; }
|
||||
}
|
||||
if opts.get_flag(options::TOTAL) {
|
||||
let ending = LineEnding::from_zero_flag(opts.get_flag(options::ZERO_TERMINATED));
|
||||
write!(writer, "{n1}{delim}{n2}{delim}{n3}{delim}total{ending}").map_err_context(|| "write error".to_string())?;
|
||||
}
|
||||
writer.flush().map_err_context(|| "write error".to_string())?;
|
||||
if should_check && (c1.has_error || c2.has_error) {
|
||||
if delayed_error { let _ = writeln!(pi_uutils_ctx::stderr(), "comm: input is not in sorted order"); }
|
||||
Ok(false)
|
||||
} else { Ok(true) }
|
||||
}
|
||||
|
||||
fn open_file(name: &OsStr, ending: LineEnding) -> io::Result<LineReader> {
|
||||
if name == "-" { return Ok(LineReader::new(Box::new(BufReader::new(pi_uutils_ctx::stdin())), ending)); }
|
||||
let resolved = pi_uutils_ctx::resolve(name);
|
||||
if fs::metadata(&resolved)?.is_dir() { return Err(io::Error::other("is a directory")); }
|
||||
Ok(LineReader::new(Box::new(BufReader::new(File::open(resolved)?)), ending))
|
||||
}
|
||||
|
||||
fn comm_main(matches: &ArgMatches) -> UResult<bool> {
|
||||
let name1 = matches.get_one::<OsString>(options::FILE_1).unwrap();
|
||||
let name2 = matches.get_one::<OsString>(options::FILE_2).unwrap();
|
||||
if name1 == "-" && name2 == "-" { return Err(USimpleError::new(1, "standard input is specified twice")); }
|
||||
let ending = LineEnding::from_zero_flag(matches.get_flag(options::ZERO_TERMINATED));
|
||||
let mut f1 = open_file(name1, ending).map_err_context(|| name1.maybe_quote().to_string())?;
|
||||
let mut f2 = open_file(name2, ending).map_err_context(|| name2.maybe_quote().to_string())?;
|
||||
let delimiters: Vec<_> = matches.get_many::<String>(options::DELIMITER).unwrap().collect();
|
||||
if delimiters[1..].iter().any(|d| *d != delimiters[0]) {
|
||||
return Err(USimpleError::new(1, "multiple conflicting output delimiters specified"));
|
||||
}
|
||||
let delim = if delimiters[0].is_empty() { "\0" } else { delimiters[0] };
|
||||
let identical = if name1 == "-" || name2 == "-" { false } else {
|
||||
files_identical(&pi_uutils_ctx::resolve(name1), &pi_uutils_ctx::resolve(name2)).unwrap_or(false)
|
||||
};
|
||||
compare(&mut f1, &mut f2, name1, name2, delim, matches, identical)
|
||||
}
|
||||
|
||||
/// Context-safe in-process entrypoint.
|
||||
pub fn run(argv: Vec<OsString>) -> i32 {
|
||||
let matches = match uu_app().try_get_matches_from(argv) {
|
||||
Ok(m) => m,
|
||||
Err(e) => {
|
||||
let rendered = e.to_string();
|
||||
if e.use_stderr() { let _ = write!(pi_uutils_ctx::stderr(), "{rendered}"); return 1; }
|
||||
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}"); return 0;
|
||||
},
|
||||
};
|
||||
match comm_main(&matches) {
|
||||
Ok(true) => pi_uutils_ctx::exit_code(),
|
||||
Ok(false) => 1,
|
||||
Err(e) => { let code = e.code(); let _ = writeln!(pi_uutils_ctx::stderr(), "comm: {e}"); if code == 0 { 1 } else { code } },
|
||||
}
|
||||
}
|
||||
|
||||
pub fn uu_app() -> Command {
|
||||
Command::new("comm")
|
||||
.version(uucore::crate_version!())
|
||||
.about("Compare sorted files FILE1 and FILE2 line by line.")
|
||||
.override_usage(format_usage("comm [OPTION]... FILE1 FILE2"))
|
||||
.infer_long_args(true).args_override_self(true)
|
||||
.arg(Arg::new(options::COLUMN_1).short('1').help("suppress column 1 (lines unique to FILE1)").action(ArgAction::SetTrue))
|
||||
.arg(Arg::new(options::COLUMN_2).short('2').help("suppress column 2 (lines unique to FILE2)").action(ArgAction::SetTrue))
|
||||
.arg(Arg::new(options::COLUMN_3).short('3').help("suppress column 3 (lines that appear in both files)").action(ArgAction::SetTrue))
|
||||
.arg(Arg::new(options::DELIMITER).long(options::DELIMITER).help("separate columns with STR").value_name("STR").default_value("\t").allow_hyphen_values(true).action(ArgAction::Append).hide_default_value(true))
|
||||
.arg(Arg::new(options::ZERO_TERMINATED).long(options::ZERO_TERMINATED).short('z').overrides_with(options::ZERO_TERMINATED).help("line delimiter is NUL, not newline").action(ArgAction::SetTrue))
|
||||
.arg(Arg::new(options::FILE_1).required(true).value_hint(clap::ValueHint::FilePath).value_parser(clap::value_parser!(OsString)))
|
||||
.arg(Arg::new(options::FILE_2).required(true).value_hint(clap::ValueHint::FilePath).value_parser(clap::value_parser!(OsString)))
|
||||
.arg(Arg::new(options::TOTAL).long(options::TOTAL).help("output a summary").action(ArgAction::SetTrue))
|
||||
.arg(Arg::new(options::CHECK_ORDER).long(options::CHECK_ORDER).help("check that input is correctly sorted, even if all input lines are pairable").action(ArgAction::SetTrue))
|
||||
.arg(Arg::new(options::NO_CHECK_ORDER).long(options::NO_CHECK_ORDER).help("do not check that input is correctly sorted").action(ArgAction::SetTrue).conflicts_with(options::CHECK_ORDER))
|
||||
}
|
||||
Vendored
+19
@@ -0,0 +1,19 @@
|
||||
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/cut), patched to route I/O
|
||||
# and path resolution through pi-uutils-ctx so it can run in-process as a shell
|
||||
# builtin. See src/cut.rs for the patch markers (`pi-uutils:` comments).
|
||||
[package]
|
||||
name = "uu_cut"
|
||||
version = "0.8.0"
|
||||
edition = "2024"
|
||||
license = "MIT"
|
||||
description = "cut ~ (uutils) display byte/field columns of input lines (vendored + patched for in-process embedding)"
|
||||
|
||||
[lib]
|
||||
path = "src/cut.rs"
|
||||
|
||||
[dependencies]
|
||||
bstr = "1.12.0"
|
||||
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
|
||||
memchr = "2.7.4"
|
||||
uucore = { version = "0.8.0", features = ["ranges"] }
|
||||
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
|
||||
Vendored
+18
@@ -0,0 +1,18 @@
|
||||
Copyright (c) uutils developers
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
this software and associated documentation files (the "Software"), to deal in
|
||||
the Software without restriction, including without limitation the rights to
|
||||
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
|
||||
the Software, and to permit persons to whom the Software is furnished to do so,
|
||||
subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
|
||||
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
||||
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
Vendored
+788
@@ -0,0 +1,788 @@
|
||||
// This file is part of the uutils coreutils package.
|
||||
//
|
||||
// For the full copyright and license information, please view the LICENSE
|
||||
// file that was distributed with this source code.
|
||||
|
||||
// spell-checker:ignore (ToDO) delim sourcefiles undelimited
|
||||
|
||||
use bstr::io::BufReadExt;
|
||||
use clap::{Arg, ArgAction, ArgMatches, Command, builder::ValueParser};
|
||||
use std::ffi::OsString;
|
||||
use std::fs::File;
|
||||
use std::io::{BufRead, BufReader, BufWriter, Read, Write};
|
||||
use std::path::Path;
|
||||
use uucore::display::Quotable;
|
||||
use uucore::error::{FromIo, UResult, USimpleError};
|
||||
use uucore::line_ending::LineEnding;
|
||||
use uucore::os_str_as_bytes;
|
||||
|
||||
use self::searcher::Searcher;
|
||||
use matcher::{ExactMatcher, Matcher, WhitespaceMatcher};
|
||||
use uucore::ranges::Range;
|
||||
use pi_uutils_ctx::format_usage;
|
||||
|
||||
mod matcher;
|
||||
mod searcher;
|
||||
|
||||
struct Options<'a> {
|
||||
out_delimiter: Option<&'a [u8]>,
|
||||
line_ending: LineEnding,
|
||||
field_opts: Option<FieldOptions<'a>>,
|
||||
}
|
||||
|
||||
enum Delimiter<'a> {
|
||||
Whitespace,
|
||||
Slice(&'a [u8]),
|
||||
}
|
||||
|
||||
struct FieldOptions<'a> {
|
||||
delimiter: Delimiter<'a>,
|
||||
only_delimited: bool,
|
||||
}
|
||||
|
||||
enum Mode<'a> {
|
||||
Bytes(Vec<Range>, Options<'a>),
|
||||
Characters(Vec<Range>, Options<'a>),
|
||||
Fields(Vec<Range>, Options<'a>),
|
||||
}
|
||||
|
||||
impl Default for Delimiter<'_> {
|
||||
fn default() -> Self {
|
||||
Self::Slice(b"\t")
|
||||
}
|
||||
}
|
||||
|
||||
impl<'a> From<&'a OsString> for Delimiter<'a> {
|
||||
fn from(s: &'a OsString) -> Self {
|
||||
Self::Slice(os_str_as_bytes(s).unwrap())
|
||||
}
|
||||
}
|
||||
|
||||
fn list_to_ranges(list: &str, complement: bool) -> Result<Vec<Range>, String> {
|
||||
if complement {
|
||||
Range::from_list(list).map(|r| uucore::ranges::complement(&r))
|
||||
} else {
|
||||
Range::from_list(list)
|
||||
}
|
||||
}
|
||||
|
||||
fn cut_bytes<R: Read, W: Write>(
|
||||
reader: R,
|
||||
out: &mut W,
|
||||
ranges: &[Range],
|
||||
opts: &Options,
|
||||
) -> UResult<()> {
|
||||
let newline_char = opts.line_ending.into();
|
||||
let mut buf_in = BufReader::new(reader);
|
||||
let out_delim = opts.out_delimiter.unwrap_or(b"\t");
|
||||
|
||||
let result = buf_in.for_byte_record(newline_char, |line| {
|
||||
let mut print_delim = false;
|
||||
for &Range { low, high } in ranges {
|
||||
if low > line.len() {
|
||||
break;
|
||||
}
|
||||
if print_delim {
|
||||
out.write_all(out_delim)?;
|
||||
} else if opts.out_delimiter.is_some() {
|
||||
print_delim = true;
|
||||
}
|
||||
// change `low` from 1-indexed value to 0-index value
|
||||
let low = low - 1;
|
||||
let high = high.min(line.len());
|
||||
out.write_all(&line[low..high])?;
|
||||
}
|
||||
out.write_all(&[newline_char])?;
|
||||
Ok(true)
|
||||
});
|
||||
|
||||
if let Err(e) = result {
|
||||
return Err(USimpleError::new(1, e.to_string()));
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Output delimiter is explicitly specified
|
||||
fn cut_fields_explicit_out_delim<R: Read, W: Write, M: Matcher>(
|
||||
reader: R,
|
||||
out: &mut W,
|
||||
matcher: &M,
|
||||
ranges: &[Range],
|
||||
only_delimited: bool,
|
||||
newline_char: u8,
|
||||
out_delim: &[u8],
|
||||
) -> UResult<()> {
|
||||
let mut buf_in = BufReader::new(reader);
|
||||
|
||||
let result = buf_in.for_byte_record_with_terminator(newline_char, |line| {
|
||||
let mut fields_pos = 1;
|
||||
let mut low_idx = 0;
|
||||
let mut delim_search = Searcher::new(matcher, line).peekable();
|
||||
let mut print_delim = false;
|
||||
|
||||
if delim_search.peek().is_none() {
|
||||
if !only_delimited {
|
||||
// Always write the entire line, even if it doesn't end with `newline_char`
|
||||
out.write_all(line)?;
|
||||
if line.is_empty() || line[line.len() - 1] != newline_char {
|
||||
out.write_all(&[newline_char])?;
|
||||
}
|
||||
}
|
||||
|
||||
return Ok(true);
|
||||
}
|
||||
|
||||
for &Range { low, high } in ranges {
|
||||
if low - fields_pos > 0 {
|
||||
// current field is not in the range, so jump to the field corresponding to the
|
||||
// beginning of the range if any
|
||||
low_idx = match delim_search.nth(low - fields_pos - 1) {
|
||||
Some((_, last)) => last,
|
||||
None => break,
|
||||
};
|
||||
}
|
||||
|
||||
// at this point, current field is the first in the range
|
||||
for _ in 0..=high - low {
|
||||
// skip printing delimiter if this is the first matching field for this line
|
||||
if print_delim {
|
||||
out.write_all(out_delim)?;
|
||||
} else {
|
||||
print_delim = true;
|
||||
}
|
||||
|
||||
if let Some((first, last)) = delim_search.next() {
|
||||
// print the current field up to the next field delim
|
||||
let segment = &line[low_idx..first];
|
||||
|
||||
out.write_all(segment)?;
|
||||
|
||||
low_idx = last;
|
||||
fields_pos = high + 1;
|
||||
} else {
|
||||
// this is the last field in the line, so print the rest
|
||||
let segment = &line[low_idx..];
|
||||
|
||||
out.write_all(segment)?;
|
||||
|
||||
if line[line.len() - 1] == newline_char {
|
||||
return Ok(true);
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
out.write_all(&[newline_char])?;
|
||||
Ok(true)
|
||||
});
|
||||
|
||||
if let Err(e) = result {
|
||||
return Err(USimpleError::new(1, e.to_string()));
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Output delimiter is the same as input delimiter
|
||||
fn cut_fields_implicit_out_delim<R: Read, W: Write, M: Matcher>(
|
||||
reader: R,
|
||||
out: &mut W,
|
||||
matcher: &M,
|
||||
ranges: &[Range],
|
||||
only_delimited: bool,
|
||||
newline_char: u8,
|
||||
) -> UResult<()> {
|
||||
let mut buf_in = BufReader::new(reader);
|
||||
|
||||
let result = buf_in.for_byte_record_with_terminator(newline_char, |line| {
|
||||
let mut fields_pos = 1;
|
||||
let mut low_idx = 0;
|
||||
let mut delim_search = Searcher::new(matcher, line).peekable();
|
||||
let mut print_delim = false;
|
||||
|
||||
if delim_search.peek().is_none() {
|
||||
if !only_delimited {
|
||||
// Always write the entire line, even if it doesn't end with `newline_char`
|
||||
out.write_all(line)?;
|
||||
if line.is_empty() || line[line.len() - 1] != newline_char {
|
||||
out.write_all(&[newline_char])?;
|
||||
}
|
||||
}
|
||||
|
||||
return Ok(true);
|
||||
}
|
||||
|
||||
for &Range { low, high } in ranges {
|
||||
if low - fields_pos > 0 {
|
||||
if let Some((first, last)) = delim_search.nth(low - fields_pos - 1) {
|
||||
low_idx = if print_delim { first } else { last }
|
||||
} else {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if let Some((first, _)) = delim_search.nth(high - low) {
|
||||
let segment = &line[low_idx..first];
|
||||
|
||||
out.write_all(segment)?;
|
||||
|
||||
print_delim = true;
|
||||
low_idx = first;
|
||||
fields_pos = high + 1;
|
||||
} else {
|
||||
let segment = &line[low_idx..line.len()];
|
||||
|
||||
out.write_all(segment)?;
|
||||
|
||||
if line[line.len() - 1] == newline_char {
|
||||
return Ok(true);
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
out.write_all(&[newline_char])?;
|
||||
Ok(true)
|
||||
});
|
||||
|
||||
if let Err(e) = result {
|
||||
return Err(USimpleError::new(1, e.to_string()));
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Streams and filters fields where the record terminator and
|
||||
/// field delimiter are the same character (specified by `newline_char`)
|
||||
fn cut_fields_newline_char_delim<R: Read, W: Write>(
|
||||
reader: R,
|
||||
out: &mut W,
|
||||
ranges: &[Range],
|
||||
newline_char: u8,
|
||||
out_delim: &[u8],
|
||||
only_delimited: bool,
|
||||
) -> UResult<()> {
|
||||
let mut reader = BufReader::new(reader);
|
||||
let mut line = Vec::new();
|
||||
|
||||
// We start at 1 because 'cut' field indexing is 1-based
|
||||
let mut current_field_idx = 1;
|
||||
let mut first_field_printed = false;
|
||||
let mut has_data = false;
|
||||
let mut suppressed = false;
|
||||
|
||||
let mut range_idx = 0;
|
||||
|
||||
loop {
|
||||
line.clear();
|
||||
|
||||
let is_selected = range_idx < ranges.len() && current_field_idx >= ranges[range_idx].low;
|
||||
let needs_data = is_selected || current_field_idx == 1;
|
||||
|
||||
let mut has_processed_data = false;
|
||||
|
||||
if needs_data {
|
||||
// Standard read: copies bytes into `line`
|
||||
loop {
|
||||
let buf = reader.fill_buf()?;
|
||||
if buf.is_empty() {
|
||||
break;
|
||||
}
|
||||
|
||||
has_processed_data = true;
|
||||
|
||||
if let Some(pos) = memchr::memchr(newline_char, buf) {
|
||||
let amt = pos + 1;
|
||||
line.extend_from_slice(&buf[..amt]);
|
||||
reader.consume(amt);
|
||||
|
||||
break;
|
||||
}
|
||||
let len = buf.len();
|
||||
line.extend_from_slice(buf);
|
||||
reader.consume(len);
|
||||
}
|
||||
} else {
|
||||
// Zero-allocation skip: scans the buffer and advances the cursor without copying
|
||||
loop {
|
||||
let buf = reader.fill_buf()?;
|
||||
if buf.is_empty() {
|
||||
break; // EOF
|
||||
}
|
||||
|
||||
has_processed_data = true;
|
||||
|
||||
if let Some(pos) = memchr::memchr(newline_char, buf) {
|
||||
let bytes_to_consume = pos + 1;
|
||||
reader.consume(bytes_to_consume);
|
||||
break;
|
||||
}
|
||||
|
||||
let len = buf.len();
|
||||
reader.consume(len);
|
||||
}
|
||||
}
|
||||
|
||||
if !has_processed_data {
|
||||
break;
|
||||
}
|
||||
has_data = true;
|
||||
|
||||
// To comply with -s when the stream consists of only a single field.
|
||||
if current_field_idx == 1 {
|
||||
let is_eof_next = reader.fill_buf()?.is_empty();
|
||||
|
||||
if is_eof_next && line.last() != Some(&newline_char) {
|
||||
if only_delimited {
|
||||
suppressed = true;
|
||||
} else {
|
||||
// GNU cut prints the whole line if no delimiter is found.
|
||||
out.write_all(&line)?;
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if range_idx < ranges.len() && current_field_idx > ranges[range_idx].high {
|
||||
range_idx += 1;
|
||||
|
||||
// EARLY EXIT: If we've exhausted all ranges, stop reading the stream entirely.
|
||||
if range_idx == ranges.len() {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// Check if the current field falls inside the current active range
|
||||
let is_selected = range_idx < ranges.len() && current_field_idx >= ranges[range_idx].low;
|
||||
|
||||
if is_selected {
|
||||
if first_field_printed {
|
||||
out.write_all(out_delim)?;
|
||||
}
|
||||
|
||||
let has_newline = line.last() == Some(&newline_char);
|
||||
let content = if has_newline {
|
||||
&line[..line.len() - 1]
|
||||
} else {
|
||||
&line[..]
|
||||
};
|
||||
|
||||
out.write_all(content)?;
|
||||
first_field_printed = true;
|
||||
}
|
||||
|
||||
current_field_idx += 1;
|
||||
}
|
||||
|
||||
if has_data && !suppressed {
|
||||
out.write_all(&[newline_char])?;
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn cut_fields<R: Read, W: Write>(
|
||||
reader: R,
|
||||
out: &mut W,
|
||||
ranges: &[Range],
|
||||
opts: &Options,
|
||||
) -> UResult<()> {
|
||||
let newline_char = opts.line_ending.into();
|
||||
let field_opts = opts.field_opts.as_ref().unwrap(); // it is safe to unwrap() here - field_opts will always be Some() for cut_fields() call
|
||||
match field_opts.delimiter {
|
||||
Delimiter::Slice(delim) if delim == [newline_char] => {
|
||||
let out_delim = opts.out_delimiter.unwrap_or(delim);
|
||||
cut_fields_newline_char_delim(
|
||||
reader,
|
||||
out,
|
||||
ranges,
|
||||
newline_char,
|
||||
out_delim,
|
||||
field_opts.only_delimited,
|
||||
)
|
||||
}
|
||||
Delimiter::Slice(delim) => {
|
||||
let matcher = ExactMatcher::new(delim);
|
||||
match opts.out_delimiter {
|
||||
Some(out_delim) => cut_fields_explicit_out_delim(
|
||||
reader,
|
||||
out,
|
||||
&matcher,
|
||||
ranges,
|
||||
field_opts.only_delimited,
|
||||
newline_char,
|
||||
out_delim,
|
||||
),
|
||||
None => cut_fields_implicit_out_delim(
|
||||
reader,
|
||||
out,
|
||||
&matcher,
|
||||
ranges,
|
||||
field_opts.only_delimited,
|
||||
newline_char,
|
||||
),
|
||||
}
|
||||
}
|
||||
Delimiter::Whitespace => {
|
||||
let matcher = WhitespaceMatcher {};
|
||||
cut_fields_explicit_out_delim(
|
||||
reader,
|
||||
out,
|
||||
&matcher,
|
||||
ranges,
|
||||
field_opts.only_delimited,
|
||||
newline_char,
|
||||
opts.out_delimiter.unwrap_or(b"\t"),
|
||||
)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// pi-uutils: route standard streams through the invocation context, and
|
||||
// resolve each relative operand only when opening it while retaining the
|
||||
// original operand for diagnostics.
|
||||
fn cut_files<'a, I>(filenames: I, mode: &Mode)
|
||||
where
|
||||
I: IntoIterator<Item = &'a OsString>,
|
||||
{
|
||||
let mut stdin_read = false;
|
||||
let mut out = BufWriter::new(pi_uutils_ctx::stdout());
|
||||
|
||||
for filename in filenames {
|
||||
if filename == "-" {
|
||||
if stdin_read {
|
||||
continue;
|
||||
}
|
||||
let result = match mode {
|
||||
Mode::Bytes(ranges, opts) | Mode::Characters(ranges, opts) =>
|
||||
cut_bytes(pi_uutils_ctx::stdin(), &mut out, ranges, opts),
|
||||
Mode::Fields(ranges, opts) =>
|
||||
cut_fields(pi_uutils_ctx::stdin(), &mut out, ranges, opts),
|
||||
};
|
||||
if let Err(err) = result {
|
||||
let _ = writeln!(pi_uutils_ctx::stderr(), "cut: {err}");
|
||||
pi_uutils_ctx::set_exit_code(1);
|
||||
}
|
||||
stdin_read = true;
|
||||
} else {
|
||||
let result = File::open(pi_uutils_ctx::resolve(Path::new(filename)))
|
||||
.map_err_context(|| filename.maybe_quote().to_string())
|
||||
.and_then(|file| match mode {
|
||||
Mode::Bytes(ranges, opts) | Mode::Characters(ranges, opts) =>
|
||||
cut_bytes(file, &mut out, ranges, opts),
|
||||
Mode::Fields(ranges, opts) => cut_fields(file, &mut out, ranges, opts),
|
||||
});
|
||||
if let Err(err) = result {
|
||||
let _ = writeln!(pi_uutils_ctx::stderr(), "cut: {err}");
|
||||
pi_uutils_ctx::set_exit_code(1);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if let Err(err) = out.flush().map_err_context(|| "write error".to_string()) {
|
||||
let _ = writeln!(pi_uutils_ctx::stderr(), "cut: {err}");
|
||||
pi_uutils_ctx::set_exit_code(1);
|
||||
}
|
||||
}
|
||||
|
||||
/// Get delimiter and output delimiter from `-d`/`--delimiter` and `--output-delimiter` options respectively
|
||||
/// Allow either delimiter to have a value that is neither UTF-8 nor ASCII to align with GNU behavior
|
||||
fn get_delimiters(matches: &ArgMatches) -> UResult<(Delimiter<'_>, Option<&[u8]>)> {
|
||||
let whitespace_delimited = matches.get_flag(options::WHITESPACE_DELIMITED);
|
||||
let delim_opt = matches.get_one::<OsString>(options::DELIMITER);
|
||||
let delim = match delim_opt {
|
||||
Some(_) if whitespace_delimited => {
|
||||
return Err(USimpleError::new(
|
||||
1,
|
||||
"invalid input: Only one of --delimiter (-d) or -w option can be specified",
|
||||
));
|
||||
}
|
||||
Some(os_string) => {
|
||||
if os_string.is_empty() {
|
||||
Delimiter::Slice(b"\0")
|
||||
} else {
|
||||
// For delimiter `-d` option value - allow both UTF-8 (possibly multi-byte) characters
|
||||
// and Non UTF-8 (and not ASCII) single byte "characters", like `b"\xAD"` to align with GNU behavior
|
||||
let bytes = os_str_as_bytes(os_string)?;
|
||||
if os_string.to_str().is_some_and(|s| s.chars().count() > 1)
|
||||
|| os_string.to_str().is_none() && bytes.len() > 1
|
||||
{
|
||||
return Err(USimpleError::new(
|
||||
1,
|
||||
"the delimiter must be a single character",
|
||||
));
|
||||
}
|
||||
Delimiter::from(os_string)
|
||||
}
|
||||
}
|
||||
None => {
|
||||
if whitespace_delimited {
|
||||
Delimiter::Whitespace
|
||||
} else {
|
||||
Delimiter::default()
|
||||
}
|
||||
}
|
||||
};
|
||||
let out_delim = matches
|
||||
.get_one::<OsString>(options::OUTPUT_DELIMITER)
|
||||
.map(|os_string| {
|
||||
if os_string.is_empty() {
|
||||
b"\0"
|
||||
} else {
|
||||
os_str_as_bytes(os_string).unwrap()
|
||||
}
|
||||
});
|
||||
Ok((delim, out_delim))
|
||||
}
|
||||
|
||||
mod options {
|
||||
pub const BYTES: &str = "bytes";
|
||||
pub const CHARACTERS: &str = "characters";
|
||||
pub const DELIMITER: &str = "delimiter";
|
||||
pub const FIELDS: &str = "fields";
|
||||
pub const ZERO_TERMINATED: &str = "zero-terminated";
|
||||
pub const ONLY_DELIMITED: &str = "only-delimited";
|
||||
pub const OUTPUT_DELIMITER: &str = "output-delimiter";
|
||||
pub const WHITESPACE_DELIMITED: &str = "whitespace-delimited";
|
||||
pub const COMPLEMENT: &str = "complement";
|
||||
pub const FILE: &str = "file";
|
||||
// ignored option
|
||||
pub const NOTHING: &str = "nothing";
|
||||
}
|
||||
|
||||
// pi-uutils: replace the terminating uucore entry macro and localization-aware
|
||||
// clap handler with an in-process entry point and literal English messages.
|
||||
/// Run `cut` against the streams and working directory installed by
|
||||
/// `pi-uutils-ctx`.
|
||||
pub fn run(argv: Vec<OsString>) -> i32 {
|
||||
// GNU cut accepts `-d=` as a delimiter spelling. Clap otherwise parses it
|
||||
// as an empty value assigned to `-d`.
|
||||
let argv = argv
|
||||
.into_iter()
|
||||
.map(|arg| if arg == "-d=" { OsString::from("--delimiter==") } else { arg })
|
||||
.collect::<Vec<_>>();
|
||||
let matches = match uu_app().try_get_matches_from(argv) {
|
||||
Ok(matches) => matches,
|
||||
Err(err) => {
|
||||
let rendered = err.to_string();
|
||||
if err.use_stderr() {
|
||||
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
|
||||
return 1;
|
||||
}
|
||||
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
|
||||
return 0;
|
||||
}
|
||||
};
|
||||
|
||||
match cut_main(&matches) {
|
||||
Ok(()) => pi_uutils_ctx::exit_code(),
|
||||
Err(err) => {
|
||||
let code = err.code();
|
||||
let _ = writeln!(pi_uutils_ctx::stderr(), "cut: {err}");
|
||||
if code == 0 { 1 } else { code }
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn cut_main(matches: &ArgMatches) -> UResult<()> {
|
||||
let complement = matches.get_flag(options::COMPLEMENT);
|
||||
let only_delimited = matches.get_flag(options::ONLY_DELIMITED);
|
||||
|
||||
let (delimiter, out_delimiter) = get_delimiters(&matches)?;
|
||||
let line_ending = LineEnding::from_zero_flag(matches.get_flag(options::ZERO_TERMINATED));
|
||||
|
||||
// Only one, and only one of cutting mode arguments, i.e. `-b`, `-c`, `-f`,
|
||||
// is expected. The number of those arguments is used for parsing a cutting
|
||||
// mode and handling the error cases.
|
||||
let mode_args_count = [
|
||||
matches.indices_of(options::BYTES),
|
||||
matches.indices_of(options::CHARACTERS),
|
||||
matches.indices_of(options::FIELDS),
|
||||
]
|
||||
.into_iter()
|
||||
.map(|indices| indices.unwrap_or_default().count())
|
||||
.sum();
|
||||
|
||||
let mode_parse = match (
|
||||
mode_args_count,
|
||||
matches.get_one::<String>(options::BYTES),
|
||||
matches.get_one::<String>(options::CHARACTERS),
|
||||
matches.get_one::<String>(options::FIELDS),
|
||||
) {
|
||||
(1, Some(byte_ranges), None, None) => {
|
||||
list_to_ranges(byte_ranges, complement).map(|ranges| {
|
||||
Mode::Bytes(
|
||||
ranges,
|
||||
Options {
|
||||
out_delimiter,
|
||||
line_ending,
|
||||
field_opts: None,
|
||||
},
|
||||
)
|
||||
})
|
||||
}
|
||||
|
||||
(1, None, Some(char_ranges), None) => {
|
||||
list_to_ranges(char_ranges, complement).map(|ranges| {
|
||||
Mode::Characters(
|
||||
ranges,
|
||||
Options {
|
||||
out_delimiter,
|
||||
line_ending,
|
||||
field_opts: None,
|
||||
},
|
||||
)
|
||||
})
|
||||
}
|
||||
|
||||
(1, None, None, Some(field_ranges)) => {
|
||||
list_to_ranges(field_ranges, complement).map(|ranges| {
|
||||
Mode::Fields(
|
||||
ranges,
|
||||
Options {
|
||||
out_delimiter,
|
||||
line_ending,
|
||||
field_opts: Some(FieldOptions {
|
||||
delimiter,
|
||||
only_delimited,
|
||||
}),
|
||||
},
|
||||
)
|
||||
})
|
||||
}
|
||||
|
||||
(2.., _, _, _) => Err("invalid usage: expects no more than one of --fields (-f), --chars (-c) or --bytes (-b)".to_owned()),
|
||||
_ => Err("invalid usage: expects one of --fields (-f), --chars (-c) or --bytes (-b)".to_owned()),
|
||||
};
|
||||
|
||||
let mode_parse = match mode_parse {
|
||||
Err(_) => mode_parse,
|
||||
Ok(mode) => match mode {
|
||||
Mode::Bytes(_, _) | Mode::Characters(_, _)
|
||||
if matches.contains_id(options::DELIMITER) =>
|
||||
{
|
||||
Err("invalid input: The '--delimiter' ('-d') option can only be used when printing a sequence of fields".to_owned())
|
||||
}
|
||||
Mode::Bytes(_, _) | Mode::Characters(_, _)
|
||||
if matches.get_flag(options::WHITESPACE_DELIMITED) =>
|
||||
{
|
||||
Err("invalid input: The '-w' option can only be used when printing a sequence of fields".to_owned())
|
||||
}
|
||||
Mode::Bytes(_, _) | Mode::Characters(_, _)
|
||||
if matches.get_flag(options::ONLY_DELIMITED) =>
|
||||
{
|
||||
Err("invalid input: The '--only-delimited' ('-s') option can only be used when printing a sequence of fields".to_owned())
|
||||
}
|
||||
_ => Ok(mode),
|
||||
},
|
||||
};
|
||||
|
||||
let mode = mode_parse.map_err(|e| USimpleError::new(1, e))?;
|
||||
#[allow(clippy::unwrap_used, reason = "clap provides '-' by default")]
|
||||
let files = matches.get_many::<OsString>(options::FILE).unwrap();
|
||||
|
||||
cut_files(files, &mode);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub fn uu_app() -> Command {
|
||||
Command::new("cut")
|
||||
.version(env!("CARGO_PKG_VERSION"))
|
||||
.override_usage(format_usage("cut OPTION... [FILE]..."))
|
||||
.about("Print specified byte or field columns from each line of stdin or input files")
|
||||
.after_help("Each invocation must specify exactly one of --bytes, --characters, or --fields. Use - as a file operand to read standard input.")
|
||||
.infer_long_args(true)
|
||||
// While `args_override_self(true)` for some arguments, such as `-d`
|
||||
// and `--output-delimiter`, is consistent to the behavior of GNU cut,
|
||||
// arguments related to cutting mode, i.e. `-b`, `-c`, `-f`, should
|
||||
// cause an error when there is more than one of them, as described in
|
||||
// the manual of GNU cut: "Use one, and only one of -b, -c or -f".
|
||||
// `ArgAction::Append` is used on `-b`, `-c`, `-f` arguments, so that
|
||||
// the occurrences of those could be counted and be handled accordingly.
|
||||
.args_override_self(true)
|
||||
.arg(
|
||||
Arg::new(options::BYTES)
|
||||
.short('b')
|
||||
.long(options::BYTES)
|
||||
.help("filter byte columns from the input source")
|
||||
.allow_hyphen_values(true)
|
||||
.value_name("LIST")
|
||||
.action(ArgAction::Append),
|
||||
)
|
||||
.arg(
|
||||
Arg::new(options::CHARACTERS)
|
||||
.short('c')
|
||||
.long(options::CHARACTERS)
|
||||
.help("alias for character mode")
|
||||
.allow_hyphen_values(true)
|
||||
.value_name("LIST")
|
||||
.action(ArgAction::Append),
|
||||
)
|
||||
.arg(
|
||||
Arg::new(options::DELIMITER)
|
||||
.short('d')
|
||||
.long(options::DELIMITER)
|
||||
.value_parser(ValueParser::os_string())
|
||||
.help("specify the delimiter character that separates fields in the input source (default: Tab)")
|
||||
.value_name("DELIM"),
|
||||
)
|
||||
.arg(
|
||||
Arg::new(options::WHITESPACE_DELIMITED)
|
||||
.short('w')
|
||||
.help("use any amount of whitespace (Space, Tab) to separate fields (FreeBSD extension)")
|
||||
.value_name("WHITESPACE")
|
||||
.action(ArgAction::SetTrue),
|
||||
)
|
||||
.arg(
|
||||
Arg::new(options::FIELDS)
|
||||
.short('f')
|
||||
.long(options::FIELDS)
|
||||
.help("filter field columns from the input source")
|
||||
.allow_hyphen_values(true)
|
||||
.value_name("LIST")
|
||||
.action(ArgAction::Append),
|
||||
)
|
||||
.arg(
|
||||
Arg::new(options::COMPLEMENT)
|
||||
.long(options::COMPLEMENT)
|
||||
.help("invert the filter, displaying all but the selected columns")
|
||||
.action(ArgAction::SetTrue),
|
||||
)
|
||||
.arg(
|
||||
Arg::new(options::ONLY_DELIMITED)
|
||||
.short('s')
|
||||
.long(options::ONLY_DELIMITED)
|
||||
.help("in field mode, only print lines which contain the delimiter")
|
||||
.action(ArgAction::SetTrue),
|
||||
)
|
||||
.arg(
|
||||
Arg::new(options::ZERO_TERMINATED)
|
||||
.short('z')
|
||||
.long(options::ZERO_TERMINATED)
|
||||
.help("filter records separated by NUL instead of newline")
|
||||
.action(ArgAction::SetTrue),
|
||||
)
|
||||
.arg(
|
||||
Arg::new(options::OUTPUT_DELIMITER)
|
||||
.long(options::OUTPUT_DELIMITER)
|
||||
.value_parser(ValueParser::os_string())
|
||||
.help("in field mode, replace the delimiter in output lines with this argument")
|
||||
.value_name("NEW_DELIM"),
|
||||
)
|
||||
.arg(
|
||||
Arg::new(options::FILE)
|
||||
.hide(true)
|
||||
.action(ArgAction::Append)
|
||||
.value_hint(clap::ValueHint::FilePath)
|
||||
.default_value("-")
|
||||
.value_parser(clap::value_parser!(OsString)),
|
||||
)
|
||||
.arg(
|
||||
Arg::new(options::NOTHING)
|
||||
.short('n')
|
||||
.help("(ignored)")
|
||||
.action(ArgAction::SetTrue),
|
||||
)
|
||||
}
|
||||
Vendored
+117
@@ -0,0 +1,117 @@
|
||||
// This file is part of the uutils coreutils package.
|
||||
//
|
||||
// For the full copyright and license information, please view the LICENSE
|
||||
// file that was distributed with this source code.
|
||||
|
||||
use memchr::{memchr, memchr2};
|
||||
|
||||
// Find the next matching byte sequence positions
|
||||
// Return (first, last) where haystack[first..last] corresponds to the matched pattern
|
||||
pub trait Matcher {
|
||||
fn next_match(&self, haystack: &[u8]) -> Option<(usize, usize)>;
|
||||
}
|
||||
|
||||
// Matches for the exact byte sequence pattern
|
||||
pub struct ExactMatcher<'a> {
|
||||
needle: &'a [u8],
|
||||
}
|
||||
|
||||
impl<'a> ExactMatcher<'a> {
|
||||
pub fn new(needle: &'a [u8]) -> Self {
|
||||
assert!(!needle.is_empty());
|
||||
Self { needle }
|
||||
}
|
||||
}
|
||||
|
||||
impl Matcher for ExactMatcher<'_> {
|
||||
fn next_match(&self, haystack: &[u8]) -> Option<(usize, usize)> {
|
||||
let mut pos = 0usize;
|
||||
loop {
|
||||
let match_idx = memchr(self.needle[0], &haystack[pos..])?;
|
||||
let match_idx = match_idx + pos; // account for starting from pos
|
||||
|
||||
if self.needle.len() == 1 || haystack[match_idx + 1..].starts_with(&self.needle[1..]) {
|
||||
return Some((match_idx, match_idx + self.needle.len()));
|
||||
}
|
||||
|
||||
pos = match_idx + 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Matches for any number of SPACE or TAB
|
||||
pub struct WhitespaceMatcher {}
|
||||
|
||||
impl Matcher for WhitespaceMatcher {
|
||||
fn next_match(&self, haystack: &[u8]) -> Option<(usize, usize)> {
|
||||
let match_idx = memchr2(b' ', b'\t', haystack)?;
|
||||
let mut skip = match_idx + 1;
|
||||
|
||||
while skip < haystack.len() {
|
||||
match haystack[skip] {
|
||||
b' ' | b'\t' => skip += 1,
|
||||
_ => break,
|
||||
}
|
||||
}
|
||||
|
||||
Some((match_idx, skip))
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod matcher_tests {
|
||||
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn test_exact_matcher_single_byte() {
|
||||
let matcher = ExactMatcher::new(":".as_bytes());
|
||||
// spell-checker:disable
|
||||
assert_eq!(matcher.next_match("".as_bytes()), None);
|
||||
assert_eq!(matcher.next_match(":".as_bytes()), Some((0, 1)));
|
||||
assert_eq!(matcher.next_match(":abcxyz".as_bytes()), Some((0, 1)));
|
||||
assert_eq!(matcher.next_match("abc:xyz".as_bytes()), Some((3, 4)));
|
||||
assert_eq!(matcher.next_match("abcxyz:".as_bytes()), Some((6, 7)));
|
||||
assert_eq!(matcher.next_match("abcxyz".as_bytes()), None);
|
||||
// spell-checker:enable
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_exact_matcher_multi_bytes() {
|
||||
let matcher = ExactMatcher::new("<>".as_bytes());
|
||||
// spell-checker:disable
|
||||
assert_eq!(matcher.next_match("".as_bytes()), None);
|
||||
assert_eq!(matcher.next_match("<>".as_bytes()), Some((0, 2)));
|
||||
assert_eq!(matcher.next_match("<>abcxyz".as_bytes()), Some((0, 2)));
|
||||
assert_eq!(matcher.next_match("abc<>xyz".as_bytes()), Some((3, 5)));
|
||||
assert_eq!(matcher.next_match("abcxyz<>".as_bytes()), Some((6, 8)));
|
||||
assert_eq!(matcher.next_match("abcxyz".as_bytes()), None);
|
||||
// spell-checker:enable
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_whitespace_matcher_single_space() {
|
||||
let matcher = WhitespaceMatcher {};
|
||||
// spell-checker:disable
|
||||
assert_eq!(matcher.next_match("".as_bytes()), None);
|
||||
assert_eq!(matcher.next_match(" ".as_bytes()), Some((0, 1)));
|
||||
assert_eq!(matcher.next_match("\tabcxyz".as_bytes()), Some((0, 1)));
|
||||
assert_eq!(matcher.next_match("abc\txyz".as_bytes()), Some((3, 4)));
|
||||
assert_eq!(matcher.next_match("abcxyz ".as_bytes()), Some((6, 7)));
|
||||
assert_eq!(matcher.next_match("abcxyz".as_bytes()), None);
|
||||
// spell-checker:enable
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_whitespace_matcher_multi_spaces() {
|
||||
let matcher = WhitespaceMatcher {};
|
||||
// spell-checker:disable
|
||||
assert_eq!(matcher.next_match("".as_bytes()), None);
|
||||
assert_eq!(matcher.next_match(" \t ".as_bytes()), Some((0, 3)));
|
||||
assert_eq!(matcher.next_match("\t\tabcxyz".as_bytes()), Some((0, 2)));
|
||||
assert_eq!(matcher.next_match("abc \txyz".as_bytes()), Some((3, 5)));
|
||||
assert_eq!(matcher.next_match("abcxyz ".as_bytes()), Some((6, 8)));
|
||||
assert_eq!(matcher.next_match("abcxyz".as_bytes()), None);
|
||||
// spell-checker:enable
|
||||
}
|
||||
}
|
||||
Vendored
+180
@@ -0,0 +1,180 @@
|
||||
// This file is part of the uutils coreutils package.
|
||||
//
|
||||
// For the full copyright and license information, please view the LICENSE
|
||||
// file that was distributed with this source code.
|
||||
|
||||
// spell-checker:ignore multispace
|
||||
|
||||
use super::matcher::Matcher;
|
||||
|
||||
// Generic searcher that relies on a specific matcher
|
||||
pub struct Searcher<'a, 'b, M: Matcher> {
|
||||
matcher: &'a M,
|
||||
haystack: &'b [u8],
|
||||
position: usize,
|
||||
}
|
||||
|
||||
impl<'a, 'b, M: Matcher> Searcher<'a, 'b, M> {
|
||||
pub fn new(matcher: &'a M, haystack: &'b [u8]) -> Self {
|
||||
Self {
|
||||
matcher,
|
||||
haystack,
|
||||
position: 0,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Iterate over field delimiters
|
||||
// Returns (first, last) positions of each sequence, where `haystack[first..last]`
|
||||
// corresponds to the delimiter.
|
||||
impl<M: Matcher> Iterator for Searcher<'_, '_, M> {
|
||||
type Item = (usize, usize);
|
||||
|
||||
fn next(&mut self) -> Option<Self::Item> {
|
||||
let (first, last) = self.matcher.next_match(&self.haystack[self.position..])?;
|
||||
let result = (first + self.position, last + self.position);
|
||||
self.position += last;
|
||||
|
||||
Some(result)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod exact_searcher_tests {
|
||||
|
||||
use super::super::matcher::ExactMatcher;
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn test_normal() {
|
||||
let matcher = ExactMatcher::new("a".as_bytes());
|
||||
let iter = Searcher::new(&matcher, "a.a.a".as_bytes());
|
||||
let items: Vec<(usize, usize)> = iter.collect();
|
||||
assert_eq!(vec![(0, 1), (2, 3), (4, 5)], items);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_empty() {
|
||||
let matcher = ExactMatcher::new("a".as_bytes());
|
||||
let iter = Searcher::new(&matcher, "".as_bytes());
|
||||
let items: Vec<(usize, usize)> = iter.collect();
|
||||
assert!(items.is_empty());
|
||||
}
|
||||
|
||||
fn test_multibyte(line: &[u8], expected: &[(usize, usize)]) {
|
||||
let matcher = ExactMatcher::new("ab".as_bytes());
|
||||
let iter = Searcher::new(&matcher, line);
|
||||
let items: Vec<(usize, usize)> = iter.collect();
|
||||
assert_eq!(expected, items);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_multibyte_normal() {
|
||||
test_multibyte("...ab...ab...".as_bytes(), &[(3, 5), (8, 10)]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_multibyte_needle_head_at_end() {
|
||||
test_multibyte("a".as_bytes(), &[]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_multibyte_starting_needle() {
|
||||
test_multibyte("ab...ab...".as_bytes(), &[(0, 2), (5, 7)]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_multibyte_trailing_needle() {
|
||||
test_multibyte("...ab...ab".as_bytes(), &[(3, 5), (8, 10)]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_multibyte_first_byte_false_match() {
|
||||
test_multibyte("aA..aCaC..ab..aD".as_bytes(), &[(10, 12)]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_searcher_with_exact_matcher() {
|
||||
let matcher = ExactMatcher::new("<>".as_bytes());
|
||||
let haystack = "<><>a<>b<><>cd<><>".as_bytes();
|
||||
let mut searcher = Searcher::new(&matcher, haystack);
|
||||
assert_eq!(searcher.next(), Some((0, 2)));
|
||||
assert_eq!(searcher.next(), Some((2, 4)));
|
||||
assert_eq!(searcher.next(), Some((5, 7)));
|
||||
assert_eq!(searcher.next(), Some((8, 10)));
|
||||
assert_eq!(searcher.next(), Some((10, 12)));
|
||||
assert_eq!(searcher.next(), Some((14, 16)));
|
||||
assert_eq!(searcher.next(), Some((16, 18)));
|
||||
assert_eq!(searcher.next(), None);
|
||||
assert_eq!(searcher.next(), None);
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod whitespace_searcher_tests {
|
||||
|
||||
use super::super::matcher::WhitespaceMatcher;
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn test_space() {
|
||||
let matcher = WhitespaceMatcher {};
|
||||
let iter = Searcher::new(&matcher, " . . ".as_bytes());
|
||||
let items: Vec<(usize, usize)> = iter.collect();
|
||||
assert_eq!(vec![(0, 1), (2, 3), (4, 5)], items);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_tab() {
|
||||
let matcher = WhitespaceMatcher {};
|
||||
let iter = Searcher::new(&matcher, "\t.\t.\t".as_bytes());
|
||||
let items: Vec<(usize, usize)> = iter.collect();
|
||||
assert_eq!(vec![(0, 1), (2, 3), (4, 5)], items);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_empty() {
|
||||
let matcher = WhitespaceMatcher {};
|
||||
let iter = Searcher::new(&matcher, "".as_bytes());
|
||||
let items: Vec<(usize, usize)> = iter.collect();
|
||||
assert!(items.is_empty());
|
||||
}
|
||||
|
||||
fn test_multispace(line: &[u8], expected: &[(usize, usize)]) {
|
||||
let matcher = WhitespaceMatcher {};
|
||||
let iter = Searcher::new(&matcher, line);
|
||||
let items: Vec<(usize, usize)> = iter.collect();
|
||||
assert_eq!(expected, items);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_multispace_normal() {
|
||||
test_multispace(
|
||||
"... ... \t...\t ... \t ...".as_bytes(),
|
||||
&[(3, 5), (8, 10), (13, 15), (18, 21)],
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_multispace_begin() {
|
||||
test_multispace(" \t\t...".as_bytes(), &[(0, 3)]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_multispace_end() {
|
||||
test_multispace("...\t ".as_bytes(), &[(3, 6)]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_searcher_with_whitespace_matcher() {
|
||||
let matcher = WhitespaceMatcher {};
|
||||
let haystack = "\t a b \t cd\t\t".as_bytes();
|
||||
let mut searcher = Searcher::new(&matcher, haystack);
|
||||
assert_eq!(searcher.next(), Some((0, 2)));
|
||||
assert_eq!(searcher.next(), Some((3, 4)));
|
||||
assert_eq!(searcher.next(), Some((5, 8)));
|
||||
assert_eq!(searcher.next(), Some((10, 12)));
|
||||
assert_eq!(searcher.next(), None);
|
||||
assert_eq!(searcher.next(), None);
|
||||
}
|
||||
}
|
||||
Vendored
+19
@@ -0,0 +1,19 @@
|
||||
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/dirname), patched to expose
|
||||
# an in-process entrypoint using pi-uutils-ctx streams.
|
||||
[package]
|
||||
name = "uu_dirname"
|
||||
version = "0.8.0"
|
||||
edition = "2024"
|
||||
license = "MIT"
|
||||
description = "dirname ~ (uutils) display parent directory of PATHNAME (vendored + patched for in-process embedding)"
|
||||
|
||||
[lib]
|
||||
path = "src/dirname.rs"
|
||||
|
||||
[dependencies]
|
||||
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
|
||||
uucore = { version = "0.8.0" }
|
||||
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
|
||||
|
||||
[dev-dependencies]
|
||||
parking_lot = "0.12"
|
||||
Vendored
+18
@@ -0,0 +1,18 @@
|
||||
Copyright (c) uutils developers
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
this software and associated documentation files (the "Software"), to deal in
|
||||
the Software without restriction, including without limitation the rights to
|
||||
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
|
||||
the Software, and to permit persons to whom the Software is furnished to do so,
|
||||
subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
|
||||
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
||||
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
+297
@@ -0,0 +1,297 @@
|
||||
// This file is part of the uutils coreutils package.
|
||||
//
|
||||
// For the full copyright and license information, please view the LICENSE
|
||||
// file that was distributed with this source code.
|
||||
|
||||
// pi-uutils: modified for in-process embedding using pi-uutils-ctx streams.
|
||||
|
||||
use clap::{Arg, ArgAction, Command, ArgMatches};
|
||||
use std::borrow::Cow;
|
||||
use std::ffi::OsString;
|
||||
use std::io::Write;
|
||||
use uucore::error::{UResult, UUsageError};
|
||||
use pi_uutils_ctx::format_usage;
|
||||
|
||||
mod options {
|
||||
pub const ZERO: &str = "zero";
|
||||
pub const DIR: &str = "dir";
|
||||
}
|
||||
|
||||
/// Perform dirname as pure string manipulation per POSIX/GNU behavior.
|
||||
///
|
||||
/// dirname should NOT normalize paths. It does simple string manipulation:
|
||||
/// 1. Strip trailing slashes (unless path is all slashes)
|
||||
/// 2. If ends with `/.` (possibly `//.` or `///.`), strip the `/+.` pattern
|
||||
/// 3. Otherwise, remove everything after the last `/`
|
||||
/// 4. If no `/` found, return `.`
|
||||
/// 5. Strip trailing slashes from result (unless result would be empty)
|
||||
///
|
||||
/// Examples:
|
||||
/// - `foo/.` → `foo`
|
||||
/// - `foo/./bar` → `foo/.`
|
||||
/// - `foo/bar` → `foo`
|
||||
/// - `a/b/c` → `a/b`
|
||||
///
|
||||
/// Per POSIX.1-2017 dirname specification and GNU coreutils manual:
|
||||
/// - POSIX: <https://pubs.opengroup.org/onlinepubs/9699919799/utilities/dirname.html>
|
||||
/// - GNU: <https://www.gnu.org/software/coreutils/manual/html_node/dirname-invocation.html>
|
||||
///
|
||||
/// See issue #8910 and similar fix in basename (#8373, commit c5268a897).
|
||||
fn dirname_string_manipulation(path_bytes: &[u8]) -> Cow<'_, [u8]> {
|
||||
if path_bytes.is_empty() {
|
||||
return Cow::Borrowed(b".");
|
||||
}
|
||||
|
||||
let mut bytes = path_bytes;
|
||||
|
||||
// Step 1: Strip trailing slashes (but not if the entire path is slashes)
|
||||
let all_slashes = bytes.iter().all(|&b| b == b'/');
|
||||
if all_slashes {
|
||||
return Cow::Borrowed(b"/");
|
||||
}
|
||||
|
||||
while bytes.len() > 1 && bytes.ends_with(b"/") {
|
||||
bytes = &bytes[..bytes.len() - 1];
|
||||
}
|
||||
|
||||
// Step 2: Check if it ends with `/.` and strip the `/+.` pattern
|
||||
if bytes.ends_with(b".") && bytes.len() >= 2 {
|
||||
let dot_pos = bytes.len() - 1;
|
||||
if bytes[dot_pos - 1] == b'/' {
|
||||
// Find where the slashes before the dot start
|
||||
let mut slash_start = dot_pos - 1;
|
||||
while slash_start > 0 && bytes[slash_start - 1] == b'/' {
|
||||
slash_start -= 1;
|
||||
}
|
||||
// Return the stripped result
|
||||
if slash_start == 0 {
|
||||
// Result would be empty
|
||||
return if path_bytes.starts_with(b"/") {
|
||||
Cow::Borrowed(b"/")
|
||||
} else {
|
||||
Cow::Borrowed(b".")
|
||||
};
|
||||
}
|
||||
return Cow::Borrowed(&bytes[..slash_start]);
|
||||
}
|
||||
}
|
||||
|
||||
// Step 3: Normal dirname - find last / and remove everything after it
|
||||
if let Some(last_slash_pos) = bytes.iter().rposition(|&b| b == b'/') {
|
||||
// Found a slash, remove everything after it
|
||||
let mut result = &bytes[..last_slash_pos];
|
||||
|
||||
// Strip trailing slashes from result (but keep at least one if at the start)
|
||||
while result.len() > 1 && result.ends_with(b"/") {
|
||||
result = &result[..result.len() - 1];
|
||||
}
|
||||
|
||||
if result.is_empty() {
|
||||
return Cow::Borrowed(b"/");
|
||||
}
|
||||
|
||||
return Cow::Borrowed(result);
|
||||
}
|
||||
|
||||
// No slash found, return "."
|
||||
Cow::Borrowed(b".")
|
||||
}
|
||||
|
||||
/// In-process builtin entry point. Unlike upstream's `uumain`, this parses the
|
||||
/// arguments directly, renders clap help/usage/version to the context
|
||||
/// streams, and maps the `UResult` to an exit code, so it is safe to run inside
|
||||
/// the host shell process.
|
||||
pub fn run(argv: Vec<OsString>) -> i32 {
|
||||
let matches = match uu_app().try_get_matches_from(argv) {
|
||||
Ok(matches) => matches,
|
||||
Err(err) => {
|
||||
let rendered = err.to_string();
|
||||
if err.use_stderr() {
|
||||
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
|
||||
return 1;
|
||||
}
|
||||
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
|
||||
return 0;
|
||||
}
|
||||
};
|
||||
match dirname_main(&matches) {
|
||||
Ok(()) => pi_uutils_ctx::exit_code(),
|
||||
Err(err) => {
|
||||
let code = err.code();
|
||||
let _ = writeln!(pi_uutils_ctx::stderr(), "dirname: {err}");
|
||||
if code == 0 { 1 } else { code }
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn dirname_main(matches: &ArgMatches) -> UResult<()> {
|
||||
let dirnames: Vec<OsString> = matches
|
||||
.get_many::<OsString>(options::DIR)
|
||||
.unwrap_or_default()
|
||||
.cloned()
|
||||
.collect();
|
||||
|
||||
if dirnames.is_empty() {
|
||||
return Err(UUsageError::new(1, "missing operand".to_string()));
|
||||
}
|
||||
|
||||
let line_ending = if matches.get_flag(options::ZERO) {
|
||||
b"\0" as &[u8]
|
||||
} else {
|
||||
b"\n" as &[u8]
|
||||
};
|
||||
|
||||
let mut stdout = pi_uutils_ctx::stdout();
|
||||
|
||||
for path in &dirnames {
|
||||
let path_bytes = uucore::os_str_as_bytes(path.as_os_str())?;
|
||||
let result = dirname_string_manipulation(path_bytes);
|
||||
|
||||
stdout.write_all(&result)?;
|
||||
stdout.write_all(line_ending)?;
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub fn uu_app() -> Command {
|
||||
Command::new("dirname")
|
||||
.about("Strip last component from file name")
|
||||
.version(uucore::crate_version!())
|
||||
.override_usage(format_usage("dirname [OPTION] NAME..."))
|
||||
.args_override_self(true)
|
||||
.infer_long_args(true)
|
||||
.after_help("Output each NAME with its last non-slash component and trailing slashes\n removed; if NAME contains no /'s, output '.' (meaning the current directory).")
|
||||
.arg(
|
||||
Arg::new(options::ZERO)
|
||||
.long(options::ZERO)
|
||||
.short('z')
|
||||
.help("separate output with NUL rather than newline")
|
||||
.action(ArgAction::SetTrue),
|
||||
)
|
||||
.arg(
|
||||
Arg::new(options::DIR)
|
||||
.hide(true)
|
||||
.action(ArgAction::Append)
|
||||
.value_hint(clap::ValueHint::AnyPath)
|
||||
.value_parser(clap::value_parser!(OsString)),
|
||||
)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use std::sync::Arc;
|
||||
use std::collections::HashMap;
|
||||
use std::path::PathBuf;
|
||||
use pi_uutils_ctx::ScopeIo;
|
||||
use parking_lot::Mutex;
|
||||
|
||||
fn run_test(args: Vec<&str>) -> (i32, String, String) {
|
||||
let stdout_buf = Arc::new(Mutex::new(Vec::new()));
|
||||
let stderr_buf = Arc::new(Mutex::new(Vec::new()));
|
||||
|
||||
#[derive(Clone)]
|
||||
struct SharedWriter {
|
||||
buf: Arc<Mutex<Vec<u8>>>,
|
||||
}
|
||||
impl Write for SharedWriter {
|
||||
fn write(&mut self, buf: &[u8]) -> std::io::Result<usize> {
|
||||
self.buf.lock().write(buf)
|
||||
}
|
||||
fn flush(&mut self) -> std::io::Result<()> {
|
||||
self.buf.lock().flush()
|
||||
}
|
||||
}
|
||||
|
||||
let io = ScopeIo {
|
||||
stdin: Box::new(std::io::empty()),
|
||||
stdin_fd: None,
|
||||
stdin_is_search_input: false,
|
||||
stdout: Box::new(SharedWriter { buf: stdout_buf.clone() }),
|
||||
stderr: Box::new(SharedWriter { buf: stderr_buf.clone() }),
|
||||
cwd: PathBuf::from("."),
|
||||
env: HashMap::new(),
|
||||
cancel: Arc::new(std::sync::atomic::AtomicBool::new(false)),
|
||||
};
|
||||
|
||||
let argv: Vec<OsString> = std::iter::once("dirname")
|
||||
.chain(args)
|
||||
.map(OsString::from)
|
||||
.collect();
|
||||
|
||||
let code = pi_uutils_ctx::scope(io, || {
|
||||
run(argv)
|
||||
});
|
||||
|
||||
let out_str = String::from_utf8(stdout_buf.lock().clone()).unwrap();
|
||||
let err_str = String::from_utf8(stderr_buf.lock().clone()).unwrap();
|
||||
|
||||
(code, out_str, err_str)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_normal() {
|
||||
let (code, stdout, stderr) = run_test(vec!["foo/bar"]);
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(stdout, "foo\n");
|
||||
assert_eq!(stderr, "");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_trailing_slash() {
|
||||
let (code, stdout, stderr) = run_test(vec!["foo/bar/"]);
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(stdout, "foo\n");
|
||||
assert_eq!(stderr, "");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_root() {
|
||||
let (code, stdout, stderr) = run_test(vec!["/"]);
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(stdout, "/\n");
|
||||
assert_eq!(stderr, "");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_multiple() {
|
||||
let (code, stdout, stderr) = run_test(vec!["a/b", "c/d/e"]);
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(stdout, "a\nc/d\n");
|
||||
assert_eq!(stderr, "");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_zero_delimited() {
|
||||
let (code, stdout, stderr) = run_test(vec!["-z", "a/b", "c/d/e"]);
|
||||
assert_eq!(code, 0);
|
||||
assert_eq!(stdout, "a\0c/d\0");
|
||||
assert_eq!(stderr, "");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_help() {
|
||||
let (code, stdout, stderr) = run_test(vec!["--help"]);
|
||||
assert_eq!(code, 0);
|
||||
assert!(stdout.contains("Usage:"));
|
||||
assert!(stdout.contains("Strip last component"));
|
||||
assert_eq!(stderr, "");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_invalid_arg() {
|
||||
let (code, stdout, stderr) = run_test(vec!["--invalid-flag"]);
|
||||
assert_eq!(code, 1);
|
||||
assert_eq!(stdout, "");
|
||||
assert!(stderr.contains("unexpected argument"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_missing_operand() {
|
||||
let (code, stdout, stderr) = run_test(vec![]);
|
||||
assert_eq!(code, 1);
|
||||
assert_eq!(stdout, "");
|
||||
assert!(stderr.contains("missing operand"));
|
||||
}
|
||||
}
|
||||
Vendored
+16
@@ -0,0 +1,16 @@
|
||||
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/md5sum), with its
|
||||
# standalone entrypoint supplied by the context-safe uu-checksum-common macro.
|
||||
[package]
|
||||
name = "uu_md5sum"
|
||||
version = "0.8.0"
|
||||
edition = "2024"
|
||||
license = "MIT"
|
||||
description = "md5sum ~ (uutils) Print or check the MD5 checksums (vendored + patched for in-process embedding)"
|
||||
|
||||
[lib]
|
||||
path = "src/md5sum.rs"
|
||||
|
||||
[dependencies]
|
||||
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
|
||||
uucore = { version = "0.8.0", features = ["checksum", "encoding", "sum", "hardware"] }
|
||||
uu_checksum_common = { path = "../uu-checksum-common" }
|
||||
Vendored
+18
@@ -0,0 +1,18 @@
|
||||
Copyright (c) uutils developers
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
this software and associated documentation files (the "Software"), to deal in
|
||||
the Software without restriction, including without limitation the rights to
|
||||
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
|
||||
the Software, and to permit persons to whom the Software is furnished to do so,
|
||||
subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
|
||||
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
||||
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
+6
@@ -0,0 +1,6 @@
|
||||
// This file is part of the uutils coreutils package.
|
||||
//
|
||||
// For the full copyright and license information, please view the LICENSE
|
||||
// file that was distributed with this source code.
|
||||
|
||||
uu_checksum_common::declare_standalone!("md5sum", uucore::checksum::AlgoKind::Md5);
|
||||
Vendored
+17
@@ -0,0 +1,17 @@
|
||||
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/paste), patched to route I/O
|
||||
# and path resolution through pi-uutils-ctx so it can run in-process as a shell
|
||||
# builtin. See src/paste.rs for the patch markers (`pi-uutils:` comments).
|
||||
[package]
|
||||
name = "uu_paste"
|
||||
version = "0.8.0"
|
||||
edition = "2024"
|
||||
license = "MIT"
|
||||
description = "paste ~ (uutils) merge lines of files (vendored + patched for in-process embedding)"
|
||||
|
||||
[lib]
|
||||
path = "src/paste.rs"
|
||||
|
||||
[dependencies]
|
||||
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
|
||||
uucore = { version = "0.8.0", features = ["i18n-charmap"] }
|
||||
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
|
||||
Vendored
+19
@@ -0,0 +1,19 @@
|
||||
Copyright (c) uutils developers
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
this software and associated documentation files (the "Software"), to deal in
|
||||
the Software without restriction, including without limitation the rights to
|
||||
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
|
||||
the Software, and to permit persons to whom the Software is furnished to do so,
|
||||
subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
Vendored
+226
@@ -0,0 +1,226 @@
|
||||
// This file is part of the uutils coreutils package.
|
||||
//
|
||||
// For the full copyright and license information, please view the LICENSE
|
||||
// file that was distributed with this source code.
|
||||
|
||||
use clap::{Arg, ArgAction, Command};
|
||||
use std::cell::RefCell;
|
||||
use std::ffi::OsString;
|
||||
use std::fs::File;
|
||||
use std::io::{BufRead, BufReader, Read, Write};
|
||||
use std::iter::Cycle;
|
||||
use std::rc::Rc;
|
||||
use std::slice::Iter;
|
||||
use uucore::error::{UResult, USimpleError, strip_errno};
|
||||
use uucore::i18n::charmap::mb_char_len;
|
||||
|
||||
mod options {
|
||||
pub const DELIMITER: &str = "delimiters";
|
||||
pub const SERIAL: &str = "serial";
|
||||
pub const FILE: &str = "file";
|
||||
pub const ZERO_TERMINATED: &str = "zero-terminated";
|
||||
}
|
||||
|
||||
/// In-process entry point. Clap and utility I/O are routed exclusively through
|
||||
/// the invocation context; no uucore entry macro may terminate the host.
|
||||
pub fn run(argv: Vec<OsString>) -> i32 {
|
||||
let matches = match uu_app().try_get_matches_from(argv) {
|
||||
Ok(matches) => matches,
|
||||
Err(err) => {
|
||||
let rendered = err.to_string();
|
||||
if err.use_stderr() {
|
||||
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
|
||||
return 1;
|
||||
}
|
||||
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
|
||||
return 0;
|
||||
},
|
||||
};
|
||||
|
||||
let serial = matches.get_flag(options::SERIAL);
|
||||
let delimiters = matches.get_one::<OsString>(options::DELIMITER).unwrap();
|
||||
let files = matches.get_many::<OsString>(options::FILE).unwrap().cloned().collect();
|
||||
let line_ending = if matches.get_flag(options::ZERO_TERMINATED) { b'\0' } else { b'\n' };
|
||||
|
||||
match paste(files, serial, delimiters, line_ending) {
|
||||
Ok(()) => pi_uutils_ctx::exit_code(),
|
||||
Err(err) => {
|
||||
let code = err.code();
|
||||
let _ = writeln!(pi_uutils_ctx::stderr(), "paste: {err}");
|
||||
if code == 0 { 1 } else { code }
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
pub fn uu_app() -> Command {
|
||||
Command::new("paste")
|
||||
.version(uucore::crate_version!())
|
||||
.about("Merge lines of files")
|
||||
.override_usage(pi_uutils_ctx::format_usage("paste [OPTION]... [FILE]..."))
|
||||
.infer_long_args(true)
|
||||
.arg(Arg::new(options::SERIAL).long(options::SERIAL).short('s').help("paste one file at a time instead of in parallel").action(ArgAction::SetTrue))
|
||||
.arg(Arg::new(options::DELIMITER).long(options::DELIMITER).short('d').help("reuse characters from LIST instead of TABs").value_name("LIST").default_value("\t").hide_default_value(true).value_parser(clap::value_parser!(OsString)))
|
||||
.arg(Arg::new(options::FILE).value_name("FILE").action(ArgAction::Append).default_value("-").value_hint(clap::ValueHint::FilePath).value_parser(clap::value_parser!(OsString)))
|
||||
.arg(Arg::new(options::ZERO_TERMINATED).long(options::ZERO_TERMINATED).short('z').help("line delimiter is NUL, not newline").action(ArgAction::SetTrue))
|
||||
}
|
||||
|
||||
fn paste(filenames: Vec<OsString>, serial: bool, delimiters: &OsString, line_ending: u8) -> UResult<()> {
|
||||
let delimiters = parse_delimiters(delimiters)?;
|
||||
// pi-uutils: all `-` operands share the scoped stdin and consume it in order.
|
||||
let stdin = Rc::new(RefCell::new(BufReader::new(pi_uutils_ctx::stdin())));
|
||||
let mut sources = Vec::with_capacity(filenames.len());
|
||||
for filename in filenames {
|
||||
if filename == "-" {
|
||||
sources.push(InputSource::StandardInput(stdin.clone()));
|
||||
} else {
|
||||
// pi-uutils: resolve filesystem access against shell cwd, while retaining
|
||||
// the user's spelling in diagnostics.
|
||||
let file = File::open(pi_uutils_ctx::resolve(&filename)).map_err(|err| {
|
||||
USimpleError::new(1, format!("{}: {}", filename.to_string_lossy(), strip_errno(&err)))
|
||||
})?;
|
||||
sources.push(InputSource::File(BufReader::new(file)));
|
||||
}
|
||||
}
|
||||
|
||||
let source_count = sources.len();
|
||||
let mut stdout = pi_uutils_ctx::stdout();
|
||||
if !serial && source_count == 1 {
|
||||
return write_single_input_source(&mut stdout, sources.pop().unwrap(), line_ending);
|
||||
}
|
||||
|
||||
let mut delimiter_state = DelimiterState::new(&delimiters);
|
||||
let mut output = Vec::new();
|
||||
if serial {
|
||||
for source in &mut sources {
|
||||
output.clear();
|
||||
loop {
|
||||
if source.read_until(line_ending, &mut output)? == 0 { break; }
|
||||
remove_trailing_line_ending(line_ending, &mut output);
|
||||
delimiter_state.write_delimiter(&mut output);
|
||||
}
|
||||
delimiter_state.remove_trailing_delimiter(&mut output);
|
||||
stdout.write_all(&output)?;
|
||||
stdout.write_all(&[line_ending])?;
|
||||
}
|
||||
} else {
|
||||
let mut eof = vec![false; source_count];
|
||||
loop {
|
||||
output.clear();
|
||||
let mut eof_count = 0;
|
||||
for (i, source) in sources.iter_mut().enumerate() {
|
||||
if eof[i] {
|
||||
eof_count += 1;
|
||||
} else if source.read_until(line_ending, &mut output)? == 0 {
|
||||
eof[i] = true;
|
||||
eof_count += 1;
|
||||
} else {
|
||||
remove_trailing_line_ending(line_ending, &mut output);
|
||||
}
|
||||
delimiter_state.write_delimiter(&mut output);
|
||||
}
|
||||
if eof_count == source_count { break; }
|
||||
delimiter_state.remove_trailing_delimiter(&mut output);
|
||||
stdout.write_all(&output)?;
|
||||
stdout.write_all(&[line_ending])?;
|
||||
delimiter_state.reset_to_first_delimiter();
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn write_single_input_source(writer: &mut impl Write, mut source: InputSource, line_ending: u8) -> UResult<()> {
|
||||
let mut buffer = [0_u8; 8192];
|
||||
let mut has_data = false;
|
||||
let mut last_byte = line_ending;
|
||||
loop {
|
||||
let count = source.read(&mut buffer)?;
|
||||
if count == 0 { break; }
|
||||
has_data = true;
|
||||
last_byte = buffer[count - 1];
|
||||
writer.write_all(&buffer[..count])?;
|
||||
}
|
||||
if has_data && last_byte != line_ending { writer.write_all(&[line_ending])?; }
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn parse_delimiters(delimiters: &OsString) -> UResult<Box<[Box<[u8]>]>> {
|
||||
let bytes = uucore::os_str_as_bytes(delimiters)?;
|
||||
let mut result = Vec::<Box<[u8]>>::with_capacity(bytes.len());
|
||||
let mut i = 0;
|
||||
while i < bytes.len() {
|
||||
if bytes[i] == b'\\' {
|
||||
i += 1;
|
||||
if i >= bytes.len() {
|
||||
return Err(USimpleError::new(1, format!("delimiter list ends with an unescaped backslash: {}", delimiters.to_string_lossy())));
|
||||
}
|
||||
match bytes[i] {
|
||||
b'0' => result.push(Box::new([])), b'\\' => result.push(Box::new([b'\\'])),
|
||||
b'n' => result.push(Box::new([b'\n'])), b't' => result.push(Box::new([b'\t'])),
|
||||
b'b' => result.push(Box::new([b'\x08'])), b'f' => result.push(Box::new([b'\x0c'])),
|
||||
b'r' => result.push(Box::new([b'\r'])), b'v' => result.push(Box::new([b'\x0b'])),
|
||||
_ => { let len = mb_char_len(&bytes[i..]).min(bytes.len() - i); result.push(Box::from(&bytes[i..i + len])); i += len; continue; }
|
||||
}
|
||||
i += 1;
|
||||
} else {
|
||||
let len = mb_char_len(&bytes[i..]).min(bytes.len() - i);
|
||||
result.push(Box::from(&bytes[i..i + len]));
|
||||
i += len;
|
||||
}
|
||||
}
|
||||
Ok(result.into_boxed_slice())
|
||||
}
|
||||
|
||||
fn remove_trailing_line_ending(line_ending: u8, output: &mut Vec<u8>) {
|
||||
if output.last() == Some(&line_ending) { output.pop(); }
|
||||
}
|
||||
|
||||
enum DelimiterState<'a> {
|
||||
NoDelimiters,
|
||||
OneDelimiter(&'a [u8]),
|
||||
MultipleDelimiters { current: &'a [u8], delimiters: &'a [Box<[u8]>], iterator: Cycle<Iter<'a, Box<[u8]>>> },
|
||||
}
|
||||
|
||||
impl<'a> DelimiterState<'a> {
|
||||
fn new(delimiters: &'a [Box<[u8]>]) -> Self {
|
||||
match delimiters {
|
||||
[] => Self::NoDelimiters,
|
||||
[only] if only.is_empty() => Self::NoDelimiters,
|
||||
[only] => Self::OneDelimiter(only),
|
||||
[first, ..] => Self::MultipleDelimiters { current: first, delimiters, iterator: delimiters.iter().cycle() },
|
||||
}
|
||||
}
|
||||
fn reset_to_first_delimiter(&mut self) {
|
||||
if let Self::MultipleDelimiters { delimiters, iterator, .. } = self { *iterator = delimiters.iter().cycle(); }
|
||||
}
|
||||
fn remove_trailing_delimiter(&self, output: &mut Vec<u8>) {
|
||||
let len = match self { Self::NoDelimiters => return, Self::OneDelimiter(d) => d.len(), Self::MultipleDelimiters { current, .. } => current.len() };
|
||||
if len > 0 { output.truncate(output.len().saturating_sub(len)); }
|
||||
}
|
||||
fn write_delimiter(&mut self, output: &mut Vec<u8>) {
|
||||
match self {
|
||||
Self::NoDelimiters => {},
|
||||
Self::OneDelimiter(d) => output.extend_from_slice(d),
|
||||
Self::MultipleDelimiters { current, iterator, .. } => { let d = iterator.next().unwrap(); output.extend_from_slice(d); *current = d; },
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
enum InputSource {
|
||||
File(BufReader<File>),
|
||||
StandardInput(Rc<RefCell<BufReader<pi_uutils_ctx::CtxStdin>>>),
|
||||
}
|
||||
|
||||
impl InputSource {
|
||||
fn read(&mut self, buf: &mut [u8]) -> UResult<usize> {
|
||||
Ok(match self {
|
||||
Self::File(reader) => reader.read(buf)?,
|
||||
Self::StandardInput(stdin) => stdin.try_borrow_mut().map_err(|err| USimpleError::new(1, format!("standard input is already borrowed: {err}")))?.read(buf)?,
|
||||
})
|
||||
}
|
||||
fn read_until(&mut self, byte: u8, buf: &mut Vec<u8>) -> UResult<usize> {
|
||||
Ok(match self {
|
||||
Self::File(reader) => reader.read_until(byte, buf)?,
|
||||
Self::StandardInput(stdin) => stdin.try_borrow_mut().map_err(|err| USimpleError::new(1, format!("standard input is already borrowed: {err}")))?.read_until(byte, buf)?,
|
||||
})
|
||||
}
|
||||
}
|
||||
Vendored
+18
@@ -0,0 +1,18 @@
|
||||
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/sha1sum), patched to route I/O
|
||||
# and path resolution through pi-uutils-ctx so it can run in-process as a shell
|
||||
# builtin. See src/sha1sum.rs for the patch markers (`pi-uutils:` comments).
|
||||
[package]
|
||||
name = "uu_sha1sum"
|
||||
version = "0.8.0"
|
||||
edition = "2024"
|
||||
license = "MIT"
|
||||
description = "sha1sum ~ (uutils) print or check SHA1 checksums (vendored + patched for in-process embedding)"
|
||||
|
||||
[lib]
|
||||
path = "src/sha1sum.rs"
|
||||
|
||||
[dependencies]
|
||||
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
|
||||
uucore = { version = "0.8.0", features = ["checksum", "encoding", "sum", "hardware"] }
|
||||
uu_checksum_common = { path = "../uu-checksum-common" }
|
||||
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
|
||||
Vendored
+18
@@ -0,0 +1,18 @@
|
||||
Copyright (c) uutils developers
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
this software and associated documentation files (the "Software"), to deal in
|
||||
the Software without restriction, including without limitation the rights to
|
||||
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
|
||||
the Software, and to permit persons to whom the Software is furnished to do so,
|
||||
subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
|
||||
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
||||
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
+9
@@ -0,0 +1,9 @@
|
||||
// This file is part of the uutils coreutils package.
|
||||
//
|
||||
// For the full copyright and license information, please view the LICENSE
|
||||
// file that was distributed with this source code.
|
||||
|
||||
// pi-uutils: Patched for in-process embedding via the shared `uu-checksum-common` crate,
|
||||
// which redirects all standard stream I/O and file resolution through `pi-uutils-ctx`.
|
||||
|
||||
uu_checksum_common::declare_standalone!("sha1sum", uucore::checksum::AlgoKind::Sha1);
|
||||
+18
@@ -0,0 +1,18 @@
|
||||
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/sha224sum), patched to route I/O
|
||||
# and path resolution through pi-uutils-ctx so it can run in-process as a shell
|
||||
# builtin. See src/sha224sum.rs for the patch markers (`pi-uutils:` comments).
|
||||
[package]
|
||||
name = "uu_sha224sum"
|
||||
version = "0.8.0"
|
||||
edition = "2024"
|
||||
license = "MIT"
|
||||
description = "sha224sum ~ (uutils) print or check SHA224 checksums (vendored + patched for in-process embedding)"
|
||||
|
||||
[lib]
|
||||
path = "src/sha224sum.rs"
|
||||
|
||||
[dependencies]
|
||||
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
|
||||
uucore = { version = "0.8.0", features = ["checksum", "encoding", "sum", "hardware"] }
|
||||
uu_checksum_common = { path = "../uu-checksum-common" }
|
||||
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
|
||||
Vendored
+18
@@ -0,0 +1,18 @@
|
||||
Copyright (c) uutils developers
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
this software and associated documentation files (the "Software"), to deal in
|
||||
the Software without restriction, including without limitation the rights to
|
||||
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
|
||||
the Software, and to permit persons to whom the Software is furnished to do so,
|
||||
subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
|
||||
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
||||
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
+9
@@ -0,0 +1,9 @@
|
||||
// This file is part of the uutils coreutils package.
|
||||
//
|
||||
// For the full copyright and license information, please view the LICENSE
|
||||
// file that was distributed with this source code.
|
||||
|
||||
// pi-uutils: Patched for in-process embedding via the shared `uu-checksum-common` crate,
|
||||
// which redirects all standard stream I/O and file resolution through `pi-uutils-ctx`.
|
||||
|
||||
uu_checksum_common::declare_standalone!("sha224sum", uucore::checksum::AlgoKind::Sha224);
|
||||
+18
@@ -0,0 +1,18 @@
|
||||
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/sha256sum), patched to route I/O
|
||||
# and path resolution through pi-uutils-ctx so it can run in-process as a shell
|
||||
# builtin. See src/sha256sum.rs for the patch markers (`pi-uutils:` comments).
|
||||
[package]
|
||||
name = "uu_sha256sum"
|
||||
version = "0.8.0"
|
||||
edition = "2024"
|
||||
license = "MIT"
|
||||
description = "sha256sum ~ (uutils) print or check SHA256 checksums (vendored + patched for in-process embedding)"
|
||||
|
||||
[lib]
|
||||
path = "src/sha256sum.rs"
|
||||
|
||||
[dependencies]
|
||||
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
|
||||
uucore = { version = "0.8.0", features = ["checksum", "encoding", "sum", "hardware"] }
|
||||
uu_checksum_common = { path = "../uu-checksum-common" }
|
||||
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
|
||||
Vendored
+18
@@ -0,0 +1,18 @@
|
||||
Copyright (c) uutils developers
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
this software and associated documentation files (the "Software"), to deal in
|
||||
the Software without restriction, including without limitation the rights to
|
||||
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
|
||||
the Software, and to permit persons to whom the Software is furnished to do so,
|
||||
subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
|
||||
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
||||
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
+9
@@ -0,0 +1,9 @@
|
||||
// This file is part of the uutils coreutils package.
|
||||
//
|
||||
// For the full copyright and license information, please view the LICENSE
|
||||
// file that was distributed with this source code.
|
||||
|
||||
// pi-uutils: Patched for in-process embedding via the shared `uu-checksum-common` crate,
|
||||
// which redirects all standard stream I/O and file resolution through `pi-uutils-ctx`.
|
||||
|
||||
uu_checksum_common::declare_standalone!("sha256sum", uucore::checksum::AlgoKind::Sha256);
|
||||
+18
@@ -0,0 +1,18 @@
|
||||
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/sha384sum), patched to route I/O
|
||||
# and path resolution through pi-uutils-ctx so it can run in-process as a shell
|
||||
# builtin. See src/sha384sum.rs for the patch markers (`pi-uutils:` comments).
|
||||
[package]
|
||||
name = "uu_sha384sum"
|
||||
version = "0.8.0"
|
||||
edition = "2024"
|
||||
license = "MIT"
|
||||
description = "sha384sum ~ (uutils) print or check SHA384 checksums (vendored + patched for in-process embedding)"
|
||||
|
||||
[lib]
|
||||
path = "src/sha384sum.rs"
|
||||
|
||||
[dependencies]
|
||||
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
|
||||
uucore = { version = "0.8.0", features = ["checksum", "encoding", "sum", "hardware"] }
|
||||
uu_checksum_common = { path = "../uu-checksum-common" }
|
||||
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
|
||||
Vendored
+18
@@ -0,0 +1,18 @@
|
||||
Copyright (c) uutils developers
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
this software and associated documentation files (the "Software"), to deal in
|
||||
the Software without restriction, including without limitation the rights to
|
||||
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
|
||||
the Software, and to permit persons to whom the Software is furnished to do so,
|
||||
subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
|
||||
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
||||
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
+9
@@ -0,0 +1,9 @@
|
||||
// This file is part of the uutils coreutils package.
|
||||
//
|
||||
// For the full copyright and license information, please view the LICENSE
|
||||
// file that was distributed with this source code.
|
||||
|
||||
// pi-uutils: Patched for in-process embedding via the shared `uu-checksum-common` crate,
|
||||
// which redirects all standard stream I/O and file resolution through `pi-uutils-ctx`.
|
||||
|
||||
uu_checksum_common::declare_standalone!("sha384sum", uucore::checksum::AlgoKind::Sha384);
|
||||
+18
@@ -0,0 +1,18 @@
|
||||
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/sha512sum), patched to route I/O
|
||||
# and path resolution through pi-uutils-ctx so it can run in-process as a shell
|
||||
# builtin. See src/sha512sum.rs for the patch markers (`pi-uutils:` comments).
|
||||
[package]
|
||||
name = "uu_sha512sum"
|
||||
version = "0.8.0"
|
||||
edition = "2024"
|
||||
license = "MIT"
|
||||
description = "sha512sum ~ (uutils) print or check SHA512 checksums (vendored + patched for in-process embedding)"
|
||||
|
||||
[lib]
|
||||
path = "src/sha512sum.rs"
|
||||
|
||||
[dependencies]
|
||||
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
|
||||
uucore = { version = "0.8.0", features = ["checksum", "encoding", "sum", "hardware"] }
|
||||
uu_checksum_common = { path = "../uu-checksum-common" }
|
||||
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
|
||||
Vendored
+18
@@ -0,0 +1,18 @@
|
||||
Copyright (c) uutils developers
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
this software and associated documentation files (the "Software"), to deal in
|
||||
the Software without restriction, including without limitation the rights to
|
||||
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
|
||||
the Software, and to permit persons to whom the Software is furnished to do so,
|
||||
subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
|
||||
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
||||
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
+9
@@ -0,0 +1,9 @@
|
||||
// This file is part of the uutils coreutils package.
|
||||
//
|
||||
// For the full copyright and license information, please view the LICENSE
|
||||
// file that was distributed with this source code.
|
||||
|
||||
// pi-uutils: Patched for in-process embedding via the shared `uu-checksum-common` crate,
|
||||
// which redirects all standard stream I/O and file resolution through `pi-uutils-ctx`.
|
||||
|
||||
uu_checksum_common::declare_standalone!("sha512sum", uucore::checksum::AlgoKind::Sha512);
|
||||
Vendored
+17
@@ -0,0 +1,17 @@
|
||||
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/tee), patched to route I/O
|
||||
# and path resolution through pi-uutils-ctx so it can run in-process as a shell
|
||||
# builtin. See src/tee.rs for the patch markers (`pi-uutils:` comments).
|
||||
[package]
|
||||
name = "uu_tee"
|
||||
version = "0.8.0"
|
||||
edition = "2024"
|
||||
license = "MIT"
|
||||
description = "tee ~ (uutils) copy standard input to files and standard output (vendored + patched for in-process embedding)"
|
||||
|
||||
[lib]
|
||||
path = "src/tee.rs"
|
||||
|
||||
[dependencies]
|
||||
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
|
||||
uucore = "0.8.0"
|
||||
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
|
||||
Vendored
+18
@@ -0,0 +1,18 @@
|
||||
Copyright (c) uutils developers
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
this software and associated documentation files (the "Software"), to deal in
|
||||
the Software without restriction, including without limitation the rights to
|
||||
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
|
||||
the Software, and to permit persons to whom the Software is furnished to do so,
|
||||
subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
|
||||
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
||||
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
Vendored
+58
@@ -0,0 +1,58 @@
|
||||
// This file is part of the uutils coreutils package.
|
||||
//
|
||||
// For the full copyright and license information, please view the LICENSE
|
||||
// file that was distributed with this source code.
|
||||
|
||||
use clap::{Arg, ArgAction, Command, builder::PossibleValue};
|
||||
use std::ffi::OsString;
|
||||
|
||||
pub mod options {
|
||||
pub const APPEND: &str = "append";
|
||||
pub const IGNORE_INTERRUPTS: &str = "ignore-interrupts";
|
||||
pub const FILE: &str = "file";
|
||||
pub const IGNORE_PIPE_ERRORS: &str = "ignore-pipe-errors";
|
||||
pub const OUTPUT_ERROR: &str = "output-error";
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug)]
|
||||
pub enum OutputErrorMode {
|
||||
Warn,
|
||||
WarnNoPipe,
|
||||
Exit,
|
||||
ExitNoPipe,
|
||||
}
|
||||
|
||||
pub struct Options {
|
||||
pub append: bool,
|
||||
pub files: Vec<OsString>,
|
||||
pub output_error: Option<OutputErrorMode>,
|
||||
}
|
||||
|
||||
pub fn uu_app() -> Command {
|
||||
Command::new("tee")
|
||||
.version(env!("CARGO_PKG_VERSION"))
|
||||
.about("Copy standard input to each FILE, and also to standard output.")
|
||||
.override_usage("tee [OPTION]... [FILE]...")
|
||||
.after_help("If a FILE is -, copy again to standard output.")
|
||||
.infer_long_args(true)
|
||||
.disable_help_flag(true)
|
||||
.arg(Arg::new("--help").short('h').long("help").help("Print help").action(ArgAction::HelpLong))
|
||||
.arg(Arg::new(options::APPEND).long(options::APPEND).short('a').help("append to the given FILEs, do not overwrite").action(ArgAction::SetTrue))
|
||||
.arg(Arg::new(options::IGNORE_INTERRUPTS).long(options::IGNORE_INTERRUPTS).short('i').help("ignore interrupt signals (accepted without installing a process-global handler)").action(ArgAction::SetTrue))
|
||||
.arg(Arg::new(options::FILE).action(ArgAction::Append).value_hint(clap::ValueHint::FilePath).value_parser(clap::value_parser!(OsString)))
|
||||
.arg(Arg::new(options::IGNORE_PIPE_ERRORS).short('p').help("diagnose errors writing to non pipes").action(ArgAction::SetTrue))
|
||||
.arg(
|
||||
Arg::new(options::OUTPUT_ERROR)
|
||||
.long(options::OUTPUT_ERROR)
|
||||
.require_equals(true)
|
||||
.num_args(0..=1)
|
||||
.default_missing_value("warn-nopipe")
|
||||
.value_parser([
|
||||
PossibleValue::new("warn").help("diagnose errors writing to any output"),
|
||||
PossibleValue::new("warn-nopipe").help("diagnose errors writing to any output not a pipe"),
|
||||
PossibleValue::new("exit").help("exit on error writing to any output"),
|
||||
PossibleValue::new("exit-nopipe").help("exit on error writing to any output not a pipe"),
|
||||
])
|
||||
.help("set behavior on write error"),
|
||||
)
|
||||
}
|
||||
Vendored
+217
@@ -0,0 +1,217 @@
|
||||
// This file is part of the uutils coreutils package.
|
||||
//
|
||||
// For the full copyright and license information, please view the LICENSE
|
||||
// file that was distributed with this source code.
|
||||
|
||||
use std::ffi::OsString;
|
||||
use std::fs::{File, OpenOptions};
|
||||
use std::io::{Error, ErrorKind, Read, Result, Write};
|
||||
|
||||
use uucore::display::Quotable;
|
||||
|
||||
mod cli;
|
||||
pub use crate::cli::uu_app;
|
||||
use crate::cli::{Options, OutputErrorMode, options};
|
||||
|
||||
/// Context-safe in-process entry point. `argv` includes the command name.
|
||||
pub fn run(argv: Vec<OsString>) -> i32 {
|
||||
let matches = match uu_app().try_get_matches_from(argv) {
|
||||
Ok(matches) => matches,
|
||||
Err(err) => {
|
||||
let rendered = err.to_string();
|
||||
if err.use_stderr() {
|
||||
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
|
||||
return 1;
|
||||
}
|
||||
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
|
||||
return 0;
|
||||
},
|
||||
};
|
||||
|
||||
let output_error = matches
|
||||
.get_one::<String>(options::OUTPUT_ERROR)
|
||||
.map(|value| match value.as_str() {
|
||||
"warn" => OutputErrorMode::Warn,
|
||||
"warn-nopipe" => OutputErrorMode::WarnNoPipe,
|
||||
"exit" => OutputErrorMode::Exit,
|
||||
"exit-nopipe" => OutputErrorMode::ExitNoPipe,
|
||||
_ => unreachable!("clap validates output-error"),
|
||||
})
|
||||
.or_else(|| matches.get_flag(options::IGNORE_PIPE_ERRORS).then_some(OutputErrorMode::WarnNoPipe));
|
||||
let files = matches
|
||||
.get_many::<OsString>(options::FILE)
|
||||
.map(|values| values.cloned().collect())
|
||||
.unwrap_or_default();
|
||||
let opts = Options { append: matches.get_flag(options::APPEND), files, output_error };
|
||||
|
||||
match tee(&opts) {
|
||||
Ok(()) => 0,
|
||||
Err(err) => {
|
||||
let _ = writeln!(pi_uutils_ctx::stderr(), "tee: {err}");
|
||||
1
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
fn tee(options: &Options) -> Result<()> {
|
||||
// pi-uutils: deliberately do not honor -i by installing a process-global
|
||||
// signal handler. The host owns signal policy and cancellation.
|
||||
let mut writers = Vec::with_capacity(options.files.len() + 1);
|
||||
writers.push(NamedWriter { name: OsString::from("standard output"), inner: Writer::Stdout });
|
||||
let mut had_open_errors = false;
|
||||
for name in &options.files {
|
||||
if name == "-" {
|
||||
writers.push(NamedWriter { name: OsString::from("standard output"), inner: Writer::Stdout });
|
||||
continue;
|
||||
}
|
||||
match open(name, options.append) {
|
||||
Ok(writer) => writers.push(writer),
|
||||
Err(err) => {
|
||||
let _ = writeln!(pi_uutils_ctx::stderr(), "tee: {}: {err}", name.maybe_quote());
|
||||
had_open_errors = true;
|
||||
if matches!(
|
||||
options.output_error.as_ref(),
|
||||
Some(OutputErrorMode::Exit | OutputErrorMode::ExitNoPipe)
|
||||
) {
|
||||
return Err(err);
|
||||
}
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
let mut output = MultiWriter::new(writers, options.output_error.clone());
|
||||
let copy_result = copy(pi_uutils_ctx::stdin(), &mut output);
|
||||
let flush_result = output.flush();
|
||||
if had_open_errors || copy_result.is_err() || flush_result.is_err() || output.error_occurred() {
|
||||
Err(copy_result.err().or_else(|| flush_result.err()).unwrap_or_else(|| Error::other("output error")))
|
||||
} else {
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
fn copy(mut input: impl Read, mut output: impl Write) -> Result<usize> {
|
||||
const FIRST_BUF_SIZE: usize = 8 * 1024;
|
||||
let mut buffer = [0_u8; FIRST_BUF_SIZE];
|
||||
let mut len = 0;
|
||||
loop {
|
||||
match input.read(&mut buffer) {
|
||||
Ok(0) => return Ok(len),
|
||||
Ok(received) => {
|
||||
output.write_all(&buffer[..received])?;
|
||||
output.flush()?;
|
||||
len += received;
|
||||
},
|
||||
Err(err) if err.kind() == ErrorKind::Interrupted => {},
|
||||
Err(err) => {
|
||||
let _ = writeln!(pi_uutils_ctx::stderr(), "tee: error reading standard input: {err}");
|
||||
return Err(err);
|
||||
},
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn open(name: &OsString, append: bool) -> Result<NamedWriter> {
|
||||
let path = pi_uutils_ctx::resolve(name);
|
||||
let mut options = OpenOptions::new();
|
||||
if append {
|
||||
options.append(true);
|
||||
} else {
|
||||
options.truncate(true);
|
||||
}
|
||||
let file = options.write(true).create(true).open(path)?;
|
||||
Ok(NamedWriter { inner: Writer::File(file), name: name.clone() })
|
||||
}
|
||||
|
||||
struct MultiWriter {
|
||||
writers: Vec<NamedWriter>,
|
||||
output_error_mode: Option<OutputErrorMode>,
|
||||
ignored_errors: usize,
|
||||
}
|
||||
|
||||
impl MultiWriter {
|
||||
fn new(writers: Vec<NamedWriter>, output_error_mode: Option<OutputErrorMode>) -> Self {
|
||||
Self { writers, output_error_mode, ignored_errors: 0 }
|
||||
}
|
||||
fn error_occurred(&self) -> bool {
|
||||
self.ignored_errors != 0
|
||||
}
|
||||
fn process(&mut self, flush: bool, buf: &[u8]) -> Result<()> {
|
||||
let mode = self.output_error_mode.clone();
|
||||
let mut aborted = None;
|
||||
let mut errors = 0;
|
||||
self.writers.retain_mut(|writer| {
|
||||
let result = if flush { writer.flush() } else { writer.write_all(buf) };
|
||||
match result {
|
||||
Ok(()) => true,
|
||||
Err(err) => {
|
||||
let is_pipe = err.kind() == ErrorKind::BrokenPipe;
|
||||
let report =
|
||||
matches!(mode.as_ref(), Some(OutputErrorMode::Warn | OutputErrorMode::Exit))
|
||||
|| !is_pipe;
|
||||
if report {
|
||||
let _ = writeln!(pi_uutils_ctx::stderr(), "tee: {}: {err}", writer.name.maybe_quote());
|
||||
errors += 1;
|
||||
}
|
||||
let exit = matches!(mode.as_ref(), Some(OutputErrorMode::Exit))
|
||||
|| (matches!(mode.as_ref(), Some(OutputErrorMode::ExitNoPipe)) && !is_pipe);
|
||||
if exit && aborted.is_none() {
|
||||
aborted = Some(err);
|
||||
}
|
||||
false
|
||||
},
|
||||
}
|
||||
});
|
||||
self.ignored_errors += errors;
|
||||
if let Some(err) = aborted {
|
||||
Err(err)
|
||||
} else if self.writers.is_empty() {
|
||||
Err(Error::other("all outputs failed"))
|
||||
} else {
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl Write for MultiWriter {
|
||||
fn write(&mut self, buf: &[u8]) -> Result<usize> {
|
||||
self.process(false, buf)?;
|
||||
Ok(buf.len())
|
||||
}
|
||||
fn flush(&mut self) -> Result<()> {
|
||||
self.process(true, &[])
|
||||
}
|
||||
}
|
||||
|
||||
enum Writer {
|
||||
File(File),
|
||||
Stdout,
|
||||
}
|
||||
|
||||
impl Write for Writer {
|
||||
fn write(&mut self, buf: &[u8]) -> Result<usize> {
|
||||
match self {
|
||||
Self::File(file) => file.write(buf),
|
||||
Self::Stdout => pi_uutils_ctx::stdout().write(buf),
|
||||
}
|
||||
}
|
||||
fn flush(&mut self) -> Result<()> {
|
||||
match self {
|
||||
Self::File(file) => file.flush(),
|
||||
Self::Stdout => pi_uutils_ctx::stdout().flush(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
struct NamedWriter {
|
||||
inner: Writer,
|
||||
name: OsString,
|
||||
}
|
||||
|
||||
impl Write for NamedWriter {
|
||||
fn write(&mut self, buf: &[u8]) -> Result<usize> {
|
||||
self.inner.write(buf)
|
||||
}
|
||||
fn flush(&mut self) -> Result<()> {
|
||||
self.inner.flush()
|
||||
}
|
||||
}
|
||||
Vendored
+19
@@ -0,0 +1,19 @@
|
||||
# Vendored from uutils/coreutils tag 0.8.0 (src/uu/tr), patched to route I/O
|
||||
# through pi-uutils-ctx so it can run in-process as a shell builtin. See source
|
||||
# patch markers and the context-safe `run` entrypoint in src/tr.rs.
|
||||
[package]
|
||||
name = "uu_tr"
|
||||
version = "0.8.0"
|
||||
edition = "2024"
|
||||
license = "MIT"
|
||||
description = "tr ~ (uutils) translate characters within input and display (vendored + patched for in-process embedding)"
|
||||
|
||||
[lib]
|
||||
path = "src/tr.rs"
|
||||
|
||||
[dependencies]
|
||||
bytecount = { version = "0.6.8", features = ["runtime-dispatch-simd"] }
|
||||
clap = { version = "4.5", features = ["wrap_help", "cargo", "color"] }
|
||||
nom = "8.0.0"
|
||||
uucore = "0.8.0"
|
||||
pi-uutils-ctx = { path = "../../pi-uutils-ctx" }
|
||||
Vendored
+18
@@ -0,0 +1,18 @@
|
||||
Copyright (c) uutils developers
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
this software and associated documentation files (the "Software"), to deal in
|
||||
the Software without restriction, including without limitation the rights to
|
||||
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
|
||||
the Software, and to permit persons to whom the Software is furnished to do so,
|
||||
subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
|
||||
IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
||||
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
Vendored
+740
@@ -0,0 +1,740 @@
|
||||
// This file is part of the uutils coreutils package.
|
||||
//
|
||||
// For the full copyright and license information, please view the LICENSE
|
||||
// file that was distributed with this source code.
|
||||
|
||||
// spell-checker:ignore (strings) anychar combinator Alnum Punct Xdigit alnum punct xdigit cntrl
|
||||
|
||||
use crate::unicode_table;
|
||||
use nom::{
|
||||
IResult, Parser,
|
||||
branch::alt,
|
||||
bytes::complete::{tag, take, take_till, take_until},
|
||||
character::complete::one_of,
|
||||
combinator::{map, map_opt, peek, recognize, value},
|
||||
multi::{many_m_n, many0},
|
||||
sequence::{delimited, preceded, separated_pair, terminated},
|
||||
};
|
||||
use std::{
|
||||
char,
|
||||
error::Error,
|
||||
fmt::{Debug, Display},
|
||||
io::{BufRead, Write},
|
||||
};
|
||||
use uucore::error::{FromIo, UError, UResult};
|
||||
|
||||
/// Common trait for operations that can process chunks of data
|
||||
pub trait ChunkProcessor {
|
||||
fn process_chunk(&self, input: &[u8], output: &mut Vec<u8>);
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
pub enum BadSequence {
|
||||
MissingCharClassName,
|
||||
InvalidCharClass(String),
|
||||
MissingEquivalentClassChar,
|
||||
MultipleCharRepeatInSet2,
|
||||
CharRepeatInSet1,
|
||||
InvalidRepeatCount(String),
|
||||
EmptySet2WhenNotTruncatingSet1,
|
||||
ClassExceptLowerUpperInSet2,
|
||||
ClassInSet2NotMatchedBySet1,
|
||||
Set1LongerSet2EndsInClass,
|
||||
ComplementMoreThanOneUniqueInSet2,
|
||||
BackwardsRange { end: u32, start: u32 },
|
||||
MultipleCharInEquivalence(String),
|
||||
}
|
||||
|
||||
impl Display for BadSequence {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
match self {
|
||||
Self::MissingCharClassName => write!(f, "missing character class name '[::]'"),
|
||||
Self::InvalidCharClass(class) => write!(f, "invalid character class '{class}'"),
|
||||
Self::MissingEquivalentClassChar => {
|
||||
write!(f, "missing equivalence class character '[==]'")
|
||||
}
|
||||
Self::MultipleCharRepeatInSet2 => {
|
||||
write!(f, "only one [c*] repeat construct may appear in string2")
|
||||
}
|
||||
Self::CharRepeatInSet1 => {
|
||||
write!(f, "the [c*] repeat construct may not appear in string1")
|
||||
}
|
||||
Self::InvalidRepeatCount(count) => {
|
||||
write!(f, "invalid repeat count '{count}' in [c*n] construct")
|
||||
}
|
||||
Self::EmptySet2WhenNotTruncatingSet1 => {
|
||||
write!(f, "when not truncating set1, string2 must be non-empty")
|
||||
}
|
||||
Self::ClassExceptLowerUpperInSet2 => write!(
|
||||
f,
|
||||
"when translating, the only character classes that may appear in set2 are 'upper' and 'lower'"
|
||||
),
|
||||
Self::ClassInSet2NotMatchedBySet1 => write!(
|
||||
f,
|
||||
"when translating, every 'upper'/'lower' in set2 must be matched by a 'upper'/'lower' in the same position in set1"
|
||||
),
|
||||
Self::Set1LongerSet2EndsInClass => write!(
|
||||
f,
|
||||
"when translating with string1 longer than string2,\nthe latter string must not end with a character class"
|
||||
),
|
||||
Self::ComplementMoreThanOneUniqueInSet2 => write!(
|
||||
f,
|
||||
"when translating with complemented character classes,\nstring2 must map all characters in the domain to one"
|
||||
),
|
||||
Self::BackwardsRange { end, start } => {
|
||||
fn endpoint(value: u32) -> String {
|
||||
match char::from_u32(value) {
|
||||
Some(ch @ '\x20'..='\x7e') => ch.escape_default().to_string(),
|
||||
_ => format!("\\{value:03o}"),
|
||||
}
|
||||
}
|
||||
write!(
|
||||
f,
|
||||
"range-endpoints of '{}-{}' are in reverse collating sequence order",
|
||||
endpoint(*start),
|
||||
endpoint(*end)
|
||||
)
|
||||
}
|
||||
Self::MultipleCharInEquivalence(chars) => {
|
||||
write!(f, "{chars}: equivalence class operand must be a single character")
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl Error for BadSequence {}
|
||||
impl UError for BadSequence {}
|
||||
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
pub enum Class {
|
||||
Alnum,
|
||||
Alpha,
|
||||
Blank,
|
||||
Control,
|
||||
Digit,
|
||||
Graph,
|
||||
Lower,
|
||||
Print,
|
||||
Punct,
|
||||
Space,
|
||||
Upper,
|
||||
Xdigit,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
pub enum Sequence {
|
||||
Char(u8),
|
||||
CharRange(u8, u8),
|
||||
CharStar(u8),
|
||||
CharRepeat(u8, usize),
|
||||
Class(Class),
|
||||
}
|
||||
|
||||
impl Sequence {
|
||||
pub fn flatten(&self) -> Box<dyn Iterator<Item = u8>> {
|
||||
match self {
|
||||
Self::Char(c) => Box::new(std::iter::once(*c)),
|
||||
Self::CharRange(l, r) => Box::new(*l..=*r),
|
||||
Self::CharStar(c) => Box::new(std::iter::repeat(*c)),
|
||||
Self::CharRepeat(c, n) => Box::new(std::iter::repeat_n(*c, *n)),
|
||||
Self::Class(class) => match class {
|
||||
Class::Alnum => Box::new((b'0'..=b'9').chain(b'A'..=b'Z').chain(b'a'..=b'z')),
|
||||
Class::Alpha => Box::new((b'A'..=b'Z').chain(b'a'..=b'z')),
|
||||
Class::Blank => Box::new(unicode_table::BLANK.iter().copied()),
|
||||
Class::Control => Box::new((0..=31).chain(std::iter::once(127))),
|
||||
Class::Digit => Box::new(b'0'..=b'9'),
|
||||
Class::Graph => Box::new(
|
||||
(48..=57) // digit
|
||||
.chain(65..=90) // uppercase
|
||||
.chain(97..=122) // lowercase
|
||||
// punctuations
|
||||
.chain(33..=47)
|
||||
.chain(58..=64)
|
||||
.chain(91..=96)
|
||||
.chain(123..=126),
|
||||
),
|
||||
Class::Print => Box::new(
|
||||
(48..=57) // digit
|
||||
.chain(65..=90) // uppercase
|
||||
.chain(97..=122) // lowercase
|
||||
// punctuations
|
||||
.chain(33..=47)
|
||||
.chain(58..=64)
|
||||
.chain(91..=96)
|
||||
.chain(123..=126)
|
||||
.chain(std::iter::once(32)), // space
|
||||
),
|
||||
Class::Punct => Box::new((33..=47).chain(58..=64).chain(91..=96).chain(123..=126)),
|
||||
Class::Space => Box::new(unicode_table::SPACES.iter().copied()),
|
||||
Class::Xdigit => Box::new((b'0'..=b'9').chain(b'A'..=b'F').chain(b'a'..=b'f')),
|
||||
Class::Lower => Box::new(b'a'..=b'z'),
|
||||
Class::Upper => Box::new(b'A'..=b'Z'),
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// Hide all the nasty sh*t in here
|
||||
pub fn solve_set_characters(
|
||||
set1_str: &[u8],
|
||||
set2_str: &[u8],
|
||||
complement_flag: bool,
|
||||
truncate_set1_flag: bool,
|
||||
translating: bool,
|
||||
) -> Result<(Vec<u8>, Vec<u8>), BadSequence> {
|
||||
let is_char_star = |s: &&Self| -> bool { matches!(s, Self::CharStar(_)) };
|
||||
|
||||
let set1 = Self::from_str(set1_str)?;
|
||||
if set1.iter().filter(is_char_star).count() != 0 {
|
||||
return Err(BadSequence::CharRepeatInSet1);
|
||||
}
|
||||
|
||||
let mut set2 = Self::from_str(set2_str)?;
|
||||
if set2.iter().filter(is_char_star).count() > 1 {
|
||||
return Err(BadSequence::MultipleCharRepeatInSet2);
|
||||
}
|
||||
|
||||
if translating
|
||||
&& set2.iter().any(|&x| {
|
||||
matches!(x, Self::Class(_))
|
||||
&& !matches!(x, Self::Class(Class::Upper | Class::Lower))
|
||||
})
|
||||
{
|
||||
return Err(BadSequence::ClassExceptLowerUpperInSet2);
|
||||
}
|
||||
|
||||
let mut set1_solved: Vec<u8> = set1.iter().flat_map(Self::flatten).collect();
|
||||
if complement_flag {
|
||||
set1_solved = (0..=u8::MAX).filter(|x| !set1_solved.contains(x)).collect();
|
||||
}
|
||||
let set1_len = set1_solved.len();
|
||||
|
||||
let set2_len = set2
|
||||
.iter()
|
||||
.filter_map(|s| match s {
|
||||
Self::CharStar(_) => None,
|
||||
r => Some(r),
|
||||
})
|
||||
.flat_map(Self::flatten)
|
||||
.count();
|
||||
|
||||
let star_compensate_len = set1_len.saturating_sub(set2_len);
|
||||
//Replace CharStar with CharRepeat
|
||||
set2 = set2
|
||||
.iter()
|
||||
.filter_map(|s| match s {
|
||||
Self::CharStar(0) => None,
|
||||
Self::CharStar(c) => Some(Self::CharRepeat(*c, star_compensate_len)),
|
||||
r => Some(*r),
|
||||
})
|
||||
.collect();
|
||||
|
||||
// For every upper/lower in set2, there must be an upper/lower in set1 at the same position. The position is calculated by expanding everything before the upper/lower in both sets
|
||||
for (set2_pos, set2_item) in set2.iter().enumerate() {
|
||||
if matches!(set2_item, Self::Class(_)) {
|
||||
let mut set2_part_solved_len = 0;
|
||||
if set2_pos >= 1 {
|
||||
set2_part_solved_len =
|
||||
set2.iter().take(set2_pos).flat_map(Self::flatten).count();
|
||||
}
|
||||
|
||||
let mut class_matches = false;
|
||||
for (set1_pos, set1_item) in set1.iter().enumerate() {
|
||||
if matches!(set1_item, Self::Class(_)) {
|
||||
let mut set1_part_solved_len = 0;
|
||||
if set1_pos >= 1 {
|
||||
set1_part_solved_len =
|
||||
set1.iter().take(set1_pos).flat_map(Self::flatten).count();
|
||||
}
|
||||
|
||||
if set1_part_solved_len == set2_part_solved_len {
|
||||
class_matches = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if !class_matches {
|
||||
return Err(BadSequence::ClassInSet2NotMatchedBySet1);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
let set2_solved: Vec<_> = set2.iter().flat_map(Self::flatten).collect();
|
||||
|
||||
// Calculate the set of unique characters in set2
|
||||
let mut set2_uniques = set2_solved.clone();
|
||||
set2_uniques.sort_unstable();
|
||||
set2_uniques.dedup();
|
||||
|
||||
// If the complement flag is used in translate mode, only one unique
|
||||
// character may appear in set2. Validate this with the set of uniques
|
||||
// in set2 that we just generated.
|
||||
// Also, set2 must not overgrow set1, otherwise the mapping can't be 1:1.
|
||||
if set1.iter().any(|x| matches!(x, Self::Class(_)))
|
||||
&& translating
|
||||
&& complement_flag
|
||||
&& (set2_uniques.len() > 1 || set2_solved.len() > set1_len)
|
||||
{
|
||||
return Err(BadSequence::ComplementMoreThanOneUniqueInSet2);
|
||||
}
|
||||
|
||||
if set2_solved.len() < set1_solved.len() {
|
||||
if truncate_set1_flag {
|
||||
set1_solved.truncate(set2_solved.len());
|
||||
} else if matches!(
|
||||
set2.last().copied(),
|
||||
Some(Self::Class(Class::Upper | Class::Lower))
|
||||
) {
|
||||
return Err(BadSequence::Set1LongerSet2EndsInClass);
|
||||
}
|
||||
}
|
||||
|
||||
Ok((set1_solved, set2_solved))
|
||||
}
|
||||
}
|
||||
|
||||
impl Sequence {
|
||||
pub fn from_str(input: &[u8]) -> Result<Vec<Self>, BadSequence> {
|
||||
many0(alt((
|
||||
Self::parse_char_range,
|
||||
Self::parse_char_star,
|
||||
Self::parse_char_repeat,
|
||||
Self::parse_class,
|
||||
Self::parse_char_equal,
|
||||
// NOTE: This must be the last one
|
||||
map(Self::parse_backslash_or_char_with_warning, |s| {
|
||||
Ok(Self::Char(s))
|
||||
}),
|
||||
)))
|
||||
.parse(input)
|
||||
.map(|(_, r)| r)
|
||||
.unwrap()
|
||||
.into_iter()
|
||||
.collect::<Result<Vec<_>, _>>()
|
||||
}
|
||||
|
||||
fn parse_octal(input: &[u8]) -> IResult<&[u8], u8> {
|
||||
// For `parse_char_range`, `parse_char_star`, `parse_char_repeat`, `parse_char_equal`.
|
||||
// Because in these patterns, there's no ambiguous cases.
|
||||
preceded(tag("\\"), Self::parse_octal_up_to_three_digits).parse(input)
|
||||
}
|
||||
|
||||
fn parse_octal_with_warning(input: &[u8]) -> IResult<&[u8], u8> {
|
||||
preceded(
|
||||
tag("\\"),
|
||||
alt((
|
||||
Self::parse_octal_up_to_three_digits_with_warning,
|
||||
// Fallback for if the three digit octal escape is greater than \377 (0xFF), and therefore can't be
|
||||
// parsed as as a byte
|
||||
// See test `test_multibyte_octal_sequence`
|
||||
Self::parse_octal_two_digits,
|
||||
)),
|
||||
)
|
||||
.parse(input)
|
||||
}
|
||||
|
||||
fn parse_octal_up_to_three_digits(input: &[u8]) -> IResult<&[u8], u8> {
|
||||
map_opt(
|
||||
recognize(many_m_n(1, 3, one_of("01234567"))),
|
||||
|out: &[u8]| {
|
||||
let str_to_parse = std::str::from_utf8(out).unwrap();
|
||||
u8::from_str_radix(str_to_parse, 8).ok()
|
||||
},
|
||||
)
|
||||
.parse(input)
|
||||
}
|
||||
|
||||
fn parse_octal_up_to_three_digits_with_warning(input: &[u8]) -> IResult<&[u8], u8> {
|
||||
map_opt(
|
||||
recognize(many_m_n(1, 3, one_of("01234567"))),
|
||||
|out: &[u8]| {
|
||||
let str_to_parse = std::str::from_utf8(out).unwrap();
|
||||
let result = u8::from_str_radix(str_to_parse, 8).ok();
|
||||
if result.is_none() {
|
||||
if let Ok(origin_octal) = std::str::from_utf8(input) {
|
||||
let actual_octal_tail: &str = std::str::from_utf8(&input[0..2]).unwrap();
|
||||
let outstand_char: char = char::from_u32(input[2] as u32).unwrap();
|
||||
// pi-uutils: warning macros use process-global stderr; write to the scope.
|
||||
let _ = writeln!(
|
||||
pi_uutils_ctx::stderr(),
|
||||
"tr: warning: the ambiguous octal escape \\{origin_octal} is being interpreted as the 2-byte sequence \\0{actual_octal_tail}, {outstand_char}"
|
||||
);
|
||||
} else {
|
||||
let _ = writeln!(pi_uutils_ctx::stderr(), "tr: warning: invalid utf8 sequence");
|
||||
}
|
||||
}
|
||||
result
|
||||
},
|
||||
)
|
||||
.parse(input)
|
||||
}
|
||||
|
||||
fn parse_octal_two_digits(input: &[u8]) -> IResult<&[u8], u8> {
|
||||
map_opt(
|
||||
recognize(many_m_n(2, 2, one_of("01234567"))),
|
||||
|out: &[u8]| u8::from_str_radix(std::str::from_utf8(out).unwrap(), 8).ok(),
|
||||
)
|
||||
.parse(input)
|
||||
}
|
||||
|
||||
fn parse_backslash(input: &[u8]) -> IResult<&[u8], u8> {
|
||||
preceded(tag("\\"), Self::single_char)
|
||||
.parse(input)
|
||||
.map(|(l, a)| {
|
||||
let c = match a {
|
||||
b'a' => unicode_table::BEL,
|
||||
b'b' => unicode_table::BS,
|
||||
b'f' => unicode_table::FF,
|
||||
b'n' => unicode_table::LF,
|
||||
b'r' => unicode_table::CR,
|
||||
b't' => unicode_table::HT,
|
||||
b'v' => unicode_table::VT,
|
||||
x => x,
|
||||
};
|
||||
(l, c)
|
||||
})
|
||||
}
|
||||
|
||||
fn parse_backslash_or_char(input: &[u8]) -> IResult<&[u8], u8> {
|
||||
alt((Self::parse_octal, Self::parse_backslash, Self::single_char)).parse(input)
|
||||
}
|
||||
|
||||
fn parse_backslash_or_char_with_warning(input: &[u8]) -> IResult<&[u8], u8> {
|
||||
alt((
|
||||
Self::parse_octal_with_warning,
|
||||
Self::parse_backslash,
|
||||
Self::single_char,
|
||||
))
|
||||
.parse(input)
|
||||
}
|
||||
|
||||
fn single_char(input: &[u8]) -> IResult<&[u8], u8> {
|
||||
take(1usize)(input).map(|(l, a)| (l, a[0]))
|
||||
}
|
||||
|
||||
fn parse_char_range(input: &[u8]) -> IResult<&[u8], Result<Self, BadSequence>> {
|
||||
separated_pair(
|
||||
Self::parse_backslash_or_char,
|
||||
tag("-"),
|
||||
Self::parse_backslash_or_char,
|
||||
)
|
||||
.parse(input)
|
||||
.map(|(l, (a, b))| {
|
||||
(l, {
|
||||
let (start, end) = (u32::from(a), u32::from(b));
|
||||
|
||||
let range = start..=end;
|
||||
|
||||
if range.is_empty() {
|
||||
Err(BadSequence::BackwardsRange { end, start })
|
||||
} else {
|
||||
Ok(Self::CharRange(start as u8, end as u8))
|
||||
}
|
||||
})
|
||||
})
|
||||
}
|
||||
|
||||
fn parse_char_star(input: &[u8]) -> IResult<&[u8], Result<Self, BadSequence>> {
|
||||
delimited(tag("["), Self::parse_backslash_or_char, tag("*]"))
|
||||
.parse(input)
|
||||
.map(|(l, a)| (l, Ok(Self::CharStar(a))))
|
||||
}
|
||||
|
||||
fn parse_char_repeat(input: &[u8]) -> IResult<&[u8], Result<Self, BadSequence>> {
|
||||
delimited(
|
||||
tag("["),
|
||||
separated_pair(
|
||||
Self::parse_backslash_or_char,
|
||||
tag("*"),
|
||||
// TODO
|
||||
// Why are the opening and closing tags not sufficient?
|
||||
// Backslash check is a workaround for `check_against_gnu_tr_tests_repeat_bs_9`
|
||||
take_till(|ue| matches!(ue, b']' | b'\\')),
|
||||
),
|
||||
tag("]"),
|
||||
)
|
||||
.parse(input)
|
||||
.map(|(l, (c, cnt_str))| {
|
||||
let s = String::from_utf8_lossy(cnt_str);
|
||||
let result = if cnt_str.starts_with(b"0") {
|
||||
match usize::from_str_radix(&s, 8) {
|
||||
Ok(0) => Ok(Self::CharStar(c)),
|
||||
Ok(count) => Ok(Self::CharRepeat(c, count)),
|
||||
Err(_) => Err(BadSequence::InvalidRepeatCount(s.to_string())),
|
||||
}
|
||||
} else {
|
||||
match s.parse::<usize>() {
|
||||
Ok(0) => Ok(Self::CharStar(c)),
|
||||
Ok(count) => Ok(Self::CharRepeat(c, count)),
|
||||
Err(_) => Err(BadSequence::InvalidRepeatCount(s.to_string())),
|
||||
}
|
||||
};
|
||||
(l, result)
|
||||
})
|
||||
}
|
||||
|
||||
fn parse_class(input: &[u8]) -> IResult<&[u8], Result<Self, BadSequence>> {
|
||||
preceded(tag("[:"), terminated(take_until(":]"), tag(":]")))
|
||||
.parse(input)
|
||||
.map(|(l, class_name)| {
|
||||
(
|
||||
l,
|
||||
match class_name {
|
||||
b"" => Err(BadSequence::MissingCharClassName),
|
||||
b"alnum" => Ok(Self::Class(Class::Alnum)),
|
||||
b"alpha" => Ok(Self::Class(Class::Alpha)),
|
||||
b"blank" => Ok(Self::Class(Class::Blank)),
|
||||
b"cntrl" => Ok(Self::Class(Class::Control)),
|
||||
b"digit" => Ok(Self::Class(Class::Digit)),
|
||||
b"graph" => Ok(Self::Class(Class::Graph)),
|
||||
b"lower" => Ok(Self::Class(Class::Lower)),
|
||||
b"print" => Ok(Self::Class(Class::Print)),
|
||||
b"punct" => Ok(Self::Class(Class::Punct)),
|
||||
b"space" => Ok(Self::Class(Class::Space)),
|
||||
b"upper" => Ok(Self::Class(Class::Upper)),
|
||||
b"xdigit" => Ok(Self::Class(Class::Xdigit)),
|
||||
_ => Err(BadSequence::InvalidCharClass(format!(
|
||||
"[:{}:]",
|
||||
String::from_utf8_lossy(class_name)
|
||||
))),
|
||||
},
|
||||
)
|
||||
})
|
||||
}
|
||||
|
||||
fn parse_char_equal(input: &[u8]) -> IResult<&[u8], Result<Self, BadSequence>> {
|
||||
preceded(
|
||||
tag("[="),
|
||||
(
|
||||
alt((
|
||||
value(Err(()), peek(tag("=]"))),
|
||||
map(Self::parse_backslash_or_char, Ok),
|
||||
)),
|
||||
map(terminated(take_until("=]"), tag("=]")), |v: &[u8]| {
|
||||
if v.is_empty() { Ok(()) } else { Err(v) }
|
||||
}),
|
||||
),
|
||||
)
|
||||
.parse(input)
|
||||
.map(|(l, (a, b))| {
|
||||
(
|
||||
l,
|
||||
match (a, b) {
|
||||
(Err(()), _) => Err(BadSequence::MissingEquivalentClassChar),
|
||||
(Ok(c), Ok(())) => Ok(Self::Char(c)),
|
||||
(Ok(c), Err(v)) => Err(BadSequence::MultipleCharInEquivalence(format!(
|
||||
"{}{}",
|
||||
String::from_utf8_lossy(&[c]),
|
||||
String::from_utf8_lossy(v),
|
||||
))),
|
||||
},
|
||||
)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
pub trait SymbolTranslator {
|
||||
fn translate(&mut self, current: u8) -> Option<u8>;
|
||||
|
||||
/// Takes two [`SymbolTranslator`]s and creates a new [`SymbolTranslator`] over both in sequence.
|
||||
///
|
||||
/// This behaves pretty much identical to [`Iterator::chain`].
|
||||
fn chain<T>(self, other: T) -> ChainedSymbolTranslator<Self, T>
|
||||
where
|
||||
Self: Sized,
|
||||
{
|
||||
ChainedSymbolTranslator::<Self, T> {
|
||||
stage_a: self,
|
||||
stage_b: other,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub struct ChainedSymbolTranslator<A, B> {
|
||||
stage_a: A,
|
||||
stage_b: B,
|
||||
}
|
||||
|
||||
impl<A: SymbolTranslator, B: SymbolTranslator> SymbolTranslator for ChainedSymbolTranslator<A, B> {
|
||||
fn translate(&mut self, current: u8) -> Option<u8> {
|
||||
self.stage_a
|
||||
.translate(current)
|
||||
.and_then(|c| self.stage_b.translate(c))
|
||||
}
|
||||
}
|
||||
|
||||
/// Convert a set of bytes to a 256-element bitmap for O(1) lookup
|
||||
fn set_to_bitmap(set: &[u8]) -> [bool; 256] {
|
||||
let mut bitmap = [false; 256];
|
||||
for &byte in set {
|
||||
bitmap[byte as usize] = true;
|
||||
}
|
||||
bitmap
|
||||
}
|
||||
|
||||
#[derive(Debug)]
|
||||
pub struct DeleteOperation {
|
||||
pub(crate) delete_table: [bool; 256],
|
||||
}
|
||||
|
||||
impl DeleteOperation {
|
||||
pub fn new(set: Vec<u8>) -> Self {
|
||||
Self {
|
||||
delete_table: set_to_bitmap(&set),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl SymbolTranslator for DeleteOperation {
|
||||
fn translate(&mut self, current: u8) -> Option<u8> {
|
||||
// keep if not present in the delete set
|
||||
(!self.delete_table[current as usize]).then_some(current)
|
||||
}
|
||||
}
|
||||
|
||||
impl ChunkProcessor for DeleteOperation {
|
||||
fn process_chunk(&self, input: &[u8], output: &mut Vec<u8>) {
|
||||
use crate::simd::{find_single_change, process_single_delete};
|
||||
|
||||
// Check if this is single character deletion
|
||||
if let Some((delete_char, _)) =
|
||||
find_single_change(&self.delete_table, |_, &should_delete| should_delete)
|
||||
{
|
||||
process_single_delete(input, output, delete_char);
|
||||
} else {
|
||||
// Standard deletion
|
||||
output.extend(
|
||||
input
|
||||
.iter()
|
||||
.filter(|&&b| !self.delete_table[b as usize])
|
||||
.copied(),
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug)]
|
||||
pub struct TranslateOperation {
|
||||
pub(crate) translation_table: [u8; 256],
|
||||
}
|
||||
|
||||
impl TranslateOperation {
|
||||
pub fn new(set1: Vec<u8>, set2: Vec<u8>) -> Result<Self, BadSequence> {
|
||||
// Initialize translation table with identity mapping
|
||||
let mut translation_table = std::array::from_fn(|i| i as u8);
|
||||
|
||||
if let Some(fallback) = set2.last().copied() {
|
||||
// Apply translations from set1 to set2
|
||||
for (from, to) in set1
|
||||
.into_iter()
|
||||
.zip(set2.into_iter().chain(std::iter::repeat(fallback)))
|
||||
{
|
||||
translation_table[from as usize] = to;
|
||||
}
|
||||
|
||||
Ok(Self { translation_table })
|
||||
} else if set1.is_empty() && set2.is_empty() {
|
||||
// Identity mapping for empty sets
|
||||
Ok(Self { translation_table })
|
||||
} else {
|
||||
Err(BadSequence::EmptySet2WhenNotTruncatingSet1)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl SymbolTranslator for TranslateOperation {
|
||||
fn translate(&mut self, current: u8) -> Option<u8> {
|
||||
Some(self.translation_table[current as usize])
|
||||
}
|
||||
}
|
||||
|
||||
impl ChunkProcessor for TranslateOperation {
|
||||
fn process_chunk(&self, input: &[u8], output: &mut Vec<u8>) {
|
||||
use crate::simd::{find_single_change, process_single_char_replace};
|
||||
|
||||
// Check if this is a simple single-character translation
|
||||
if let Some((source, target)) =
|
||||
find_single_change(&self.translation_table, |i, &val| val != i as u8)
|
||||
{
|
||||
// Use SIMD-optimized single character replacement
|
||||
process_single_char_replace(input, output, source, target);
|
||||
} else {
|
||||
// Standard translation using table lookup
|
||||
output.extend(input.iter().map(|&b| self.translation_table[b as usize]));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct SqueezeOperation {
|
||||
squeeze_table: [bool; 256],
|
||||
previous: Option<u8>,
|
||||
}
|
||||
|
||||
impl SqueezeOperation {
|
||||
pub fn new(set1: Vec<u8>) -> Self {
|
||||
Self {
|
||||
squeeze_table: set_to_bitmap(&set1),
|
||||
previous: None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl SymbolTranslator for SqueezeOperation {
|
||||
fn translate(&mut self, current: u8) -> Option<u8> {
|
||||
let next = if self.squeeze_table[current as usize] {
|
||||
match self.previous {
|
||||
Some(v) if v == current => None,
|
||||
_ => Some(current),
|
||||
}
|
||||
} else {
|
||||
Some(current)
|
||||
};
|
||||
self.previous = Some(current);
|
||||
next
|
||||
}
|
||||
}
|
||||
|
||||
pub fn translate_input<T, R, W>(input: &mut R, output: &mut W, mut translator: T) -> UResult<()>
|
||||
where
|
||||
T: SymbolTranslator,
|
||||
R: BufRead,
|
||||
W: Write,
|
||||
{
|
||||
const BUFFER_SIZE: usize = 32768; // Large buffer for better throughput
|
||||
let mut buf = [0; BUFFER_SIZE];
|
||||
let mut output_buf = Vec::with_capacity(BUFFER_SIZE);
|
||||
|
||||
loop {
|
||||
let length = match input.read(&mut buf[..]) {
|
||||
Ok(0) => break, // EOF reached
|
||||
Ok(len) => len,
|
||||
Err(e) if e.kind() == std::io::ErrorKind::Interrupted => continue,
|
||||
Err(e) => return Err(e.map_err_context(|| "read error".to_string())),
|
||||
};
|
||||
|
||||
// Process the buffer and collect translated chars to output
|
||||
output_buf.clear();
|
||||
for &byte in &buf[..length] {
|
||||
if let Some(translated) = translator.translate(byte) {
|
||||
output_buf.push(translated);
|
||||
}
|
||||
}
|
||||
|
||||
if !output_buf.is_empty() {
|
||||
crate::simd::write_output(output, &output_buf)?;
|
||||
}
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Platform-specific flush operation
|
||||
#[inline]
|
||||
pub fn flush_output<W: Write>(output: &mut W) -> UResult<()> {
|
||||
#[cfg(not(target_os = "windows"))]
|
||||
return output.flush().map_err_context(|| "write error".to_string());
|
||||
|
||||
#[cfg(target_os = "windows")]
|
||||
return output.flush().map_err_context(|| "write error".to_string());
|
||||
}
|
||||
Vendored
+95
@@ -0,0 +1,95 @@
|
||||
// This file is part of the uutils coreutils package.
|
||||
//
|
||||
// For the full copyright and license information, please view the LICENSE
|
||||
// file that was distributed with this source code.
|
||||
|
||||
//! I/O processing infrastructure for tr operations with SIMD optimizations
|
||||
|
||||
use crate::operation::ChunkProcessor;
|
||||
use std::io::{BufRead, Write};
|
||||
use uucore::error::{FromIo, UResult};
|
||||
|
||||
/// Helper to detect single-character operations for optimization
|
||||
pub fn find_single_change<T, F>(table: &[T; 256], check: F) -> Option<(u8, T)>
|
||||
where
|
||||
F: Fn(usize, &T) -> bool,
|
||||
T: Copy,
|
||||
{
|
||||
let matches: Vec<_> = table
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter_map(|(i, val)| check(i, val).then_some((i as u8, *val)))
|
||||
.take(2)
|
||||
.collect();
|
||||
|
||||
(matches.len() == 1).then(|| matches[0])
|
||||
}
|
||||
|
||||
/// SIMD-optimized single character replacement
|
||||
#[inline]
|
||||
pub fn process_single_char_replace(
|
||||
input: &[u8],
|
||||
output: &mut Vec<u8>,
|
||||
source_char: u8,
|
||||
target_char: u8,
|
||||
) {
|
||||
let count = bytecount::count(input, source_char);
|
||||
if count == 0 {
|
||||
output.extend_from_slice(input);
|
||||
} else if count == input.len() {
|
||||
output.resize(output.len() + input.len(), target_char);
|
||||
} else {
|
||||
output.extend(
|
||||
input
|
||||
.iter()
|
||||
.map(|&b| if b == source_char { target_char } else { b }),
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// SIMD-optimized delete operation for single character
|
||||
pub fn process_single_delete(input: &[u8], output: &mut Vec<u8>, delete_char: u8) {
|
||||
let count = bytecount::count(input, delete_char);
|
||||
if count == 0 {
|
||||
output.extend_from_slice(input);
|
||||
} else if count < input.len() {
|
||||
output.extend(input.iter().filter(|&&b| b != delete_char).copied());
|
||||
}
|
||||
// If count == input.len(), all deleted, output nothing
|
||||
}
|
||||
|
||||
/// Unified I/O processing for all operations
|
||||
pub fn process_input<R, W, P>(input: &mut R, output: &mut W, processor: &P) -> UResult<()>
|
||||
where
|
||||
R: BufRead,
|
||||
W: Write,
|
||||
P: ChunkProcessor + ?Sized,
|
||||
{
|
||||
const BUFFER_SIZE: usize = 32768;
|
||||
let mut buf = [0; BUFFER_SIZE];
|
||||
let mut output_buf = Vec::with_capacity(BUFFER_SIZE);
|
||||
|
||||
loop {
|
||||
let length = match input.read(&mut buf[..]) {
|
||||
Ok(0) => break,
|
||||
Ok(len) => len,
|
||||
Err(e) if e.kind() == std::io::ErrorKind::Interrupted => continue,
|
||||
Err(e) => return Err(e.map_err_context(|| "read error".to_string())),
|
||||
};
|
||||
|
||||
output_buf.clear();
|
||||
processor.process_chunk(&buf[..length], &mut output_buf);
|
||||
|
||||
if !output_buf.is_empty() {
|
||||
write_output(output, &output_buf)?;
|
||||
}
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Helper function to handle platform-specific write operations
|
||||
#[inline]
|
||||
pub fn write_output<W: Write>(output: &mut W, buf: &[u8]) -> UResult<()> {
|
||||
output.write_all(buf).map_err_context(|| "write error".to_string())
|
||||
}
|
||||
Vendored
+218
@@ -0,0 +1,218 @@
|
||||
// This file is part of the uutils coreutils package.
|
||||
//
|
||||
// For the full copyright and license information, please view the LICENSE
|
||||
// file that was distributed with this source code.
|
||||
|
||||
mod operation;
|
||||
mod simd;
|
||||
mod unicode_table;
|
||||
|
||||
use std::{
|
||||
ffi::OsString,
|
||||
io::{BufReader, Write},
|
||||
};
|
||||
|
||||
use clap::{Arg, ArgAction, Command, value_parser};
|
||||
use operation::{
|
||||
DeleteOperation, Sequence, SqueezeOperation, SymbolTranslator, TranslateOperation,
|
||||
flush_output, translate_input,
|
||||
};
|
||||
use pi_uutils_ctx::format_usage;
|
||||
use simd::process_input;
|
||||
use uucore::display::Quotable;
|
||||
use uucore::error::{UResult, UUsageError};
|
||||
use uucore::os_str_as_bytes;
|
||||
|
||||
mod options {
|
||||
pub const COMPLEMENT: &str = "complement";
|
||||
pub const DELETE: &str = "delete";
|
||||
pub const SQUEEZE: &str = "squeeze-repeats";
|
||||
pub const TRUNCATE_SET1: &str = "truncate-set1";
|
||||
pub const SETS: &str = "sets";
|
||||
}
|
||||
|
||||
/// pi-uutils: context-safe in-process entry point. `argv` includes the command name;
|
||||
/// clap output and all utility diagnostics are written only to scoped streams.
|
||||
pub fn run(argv: Vec<OsString>) -> i32 {
|
||||
let matches = match uu_app().try_get_matches_from(argv) {
|
||||
Ok(matches) => matches,
|
||||
Err(err) => {
|
||||
let rendered = err.to_string();
|
||||
if err.use_stderr() {
|
||||
let _ = write!(pi_uutils_ctx::stderr(), "{rendered}");
|
||||
return 1;
|
||||
}
|
||||
let _ = write!(pi_uutils_ctx::stdout(), "{rendered}");
|
||||
return 0;
|
||||
}
|
||||
};
|
||||
|
||||
match tr_main(&matches) {
|
||||
Ok(()) => pi_uutils_ctx::exit_code(),
|
||||
Err(err) => {
|
||||
let code = err.code();
|
||||
let _ = writeln!(pi_uutils_ctx::stderr(), "tr: {err}");
|
||||
if code == 0 { 1 } else { code }
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn tr_main(matches: &clap::ArgMatches) -> UResult<()> {
|
||||
let delete_flag = matches.get_flag(options::DELETE);
|
||||
let complement_flag = matches.get_flag(options::COMPLEMENT);
|
||||
let squeeze_flag = matches.get_flag(options::SQUEEZE);
|
||||
let truncate_set1_flag = matches.get_flag(options::TRUNCATE_SET1);
|
||||
|
||||
let sets: Vec<_> = matches
|
||||
.get_many::<OsString>(options::SETS)
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.map(ToOwned::to_owned)
|
||||
.collect();
|
||||
|
||||
if sets.is_empty() {
|
||||
return Err(UUsageError::new(1, "missing operand"));
|
||||
}
|
||||
|
||||
let sets_len = sets.len();
|
||||
if !(delete_flag || squeeze_flag) && sets_len == 1 {
|
||||
return Err(UUsageError::new(
|
||||
1,
|
||||
format!(
|
||||
"missing operand after {}\nTwo strings must be given when translating.",
|
||||
sets[0].quote()
|
||||
),
|
||||
));
|
||||
}
|
||||
|
||||
if delete_flag && squeeze_flag && sets_len == 1 {
|
||||
return Err(UUsageError::new(
|
||||
1,
|
||||
format!(
|
||||
"missing operand after {}\nTwo strings must be given when deleting and squeezing.",
|
||||
sets[0].quote()
|
||||
),
|
||||
));
|
||||
}
|
||||
|
||||
if sets_len > 1 {
|
||||
if delete_flag && !squeeze_flag {
|
||||
let operand = sets[1].quote();
|
||||
let message = if sets_len == 2 {
|
||||
format!(
|
||||
"extra operand {operand}\nOnly one string may be given when deleting without squeezing repeats."
|
||||
)
|
||||
} else {
|
||||
format!("extra operand {operand}")
|
||||
};
|
||||
return Err(UUsageError::new(1, message));
|
||||
}
|
||||
if sets_len > 2 {
|
||||
return Err(UUsageError::new(
|
||||
1,
|
||||
format!("extra operand {}", sets[2].quote()),
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
if let Some(first) = sets.first() {
|
||||
let bytes = os_str_as_bytes(first)?;
|
||||
let trailing_backslashes = bytes.iter().rev().take_while(|&&byte| byte == b'\\').count();
|
||||
if trailing_backslashes % 2 == 1 {
|
||||
let _ = writeln!(
|
||||
pi_uutils_ctx::stderr(),
|
||||
"tr: warning: an unescaped backslash at end of string is not portable"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
let translating = !delete_flag && sets.len() > 1;
|
||||
let mut sets_iter = sets.iter().map(OsString::as_os_str);
|
||||
let (set1, set2) = Sequence::solve_set_characters(
|
||||
os_str_as_bytes(sets_iter.next().unwrap_or_default())?,
|
||||
os_str_as_bytes(sets_iter.next().unwrap_or_default())?,
|
||||
complement_flag,
|
||||
truncate_set1_flag && translating,
|
||||
translating,
|
||||
)?;
|
||||
|
||||
// pi-uutils: replace process-global stdin/stdout with the invocation context.
|
||||
let mut input = BufReader::new(pi_uutils_ctx::stdin());
|
||||
let mut output = pi_uutils_ctx::stdout();
|
||||
|
||||
if delete_flag {
|
||||
if squeeze_flag {
|
||||
let operation = DeleteOperation::new(set1).chain(SqueezeOperation::new(set2));
|
||||
translate_input(&mut input, &mut output, operation)?;
|
||||
} else {
|
||||
process_input(&mut input, &mut output, &DeleteOperation::new(set1))?;
|
||||
}
|
||||
} else if squeeze_flag {
|
||||
if sets_len == 1 {
|
||||
translate_input(&mut input, &mut output, SqueezeOperation::new(set1))?;
|
||||
} else {
|
||||
let operation = TranslateOperation::new(set1, set2.clone())?
|
||||
.chain(SqueezeOperation::new(set2));
|
||||
translate_input(&mut input, &mut output, operation)?;
|
||||
}
|
||||
} else {
|
||||
process_input(
|
||||
&mut input,
|
||||
&mut output,
|
||||
&TranslateOperation::new(set1, set2)?,
|
||||
)?;
|
||||
}
|
||||
|
||||
flush_output(&mut output)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub fn uu_app() -> Command {
|
||||
Command::new("tr")
|
||||
.version(env!("CARGO_PKG_VERSION"))
|
||||
.about("Translate or delete characters")
|
||||
.override_usage(format_usage("tr [OPTION]... SET1 [SET2]"))
|
||||
.after_help(
|
||||
"Translate, squeeze, and/or delete characters from standard input, writing to standard output.",
|
||||
)
|
||||
.infer_long_args(true)
|
||||
.trailing_var_arg(true)
|
||||
.arg(
|
||||
Arg::new(options::COMPLEMENT)
|
||||
.visible_short_alias('C')
|
||||
.short('c')
|
||||
.long(options::COMPLEMENT)
|
||||
.help("use the complement of SET1")
|
||||
.action(ArgAction::SetTrue)
|
||||
.overrides_with(options::COMPLEMENT),
|
||||
)
|
||||
.arg(
|
||||
Arg::new(options::DELETE)
|
||||
.short('d')
|
||||
.long(options::DELETE)
|
||||
.help("delete characters in SET1, do not translate")
|
||||
.action(ArgAction::SetTrue)
|
||||
.overrides_with(options::DELETE),
|
||||
)
|
||||
.arg(
|
||||
Arg::new(options::SQUEEZE)
|
||||
.long(options::SQUEEZE)
|
||||
.short('s')
|
||||
.help("replace each sequence of a repeated character listed in the last specified SET with a single occurrence")
|
||||
.action(ArgAction::SetTrue)
|
||||
.overrides_with(options::SQUEEZE),
|
||||
)
|
||||
.arg(
|
||||
Arg::new(options::TRUNCATE_SET1)
|
||||
.long(options::TRUNCATE_SET1)
|
||||
.short('t')
|
||||
.help("first truncate SET1 to length of SET2")
|
||||
.action(ArgAction::SetTrue)
|
||||
.overrides_with(options::TRUNCATE_SET1),
|
||||
)
|
||||
.arg(
|
||||
Arg::new(options::SETS)
|
||||
.num_args(1..)
|
||||
.value_parser(value_parser!(OsString)),
|
||||
)
|
||||
}
|
||||
+15
@@ -0,0 +1,15 @@
|
||||
// This file is part of the uutils coreutils package.
|
||||
//
|
||||
// For the full copyright and license information, please view the LICENSE
|
||||
// file that was distributed with this source code.
|
||||
|
||||
pub static BEL: u8 = 0x7;
|
||||
pub static BS: u8 = 0x8;
|
||||
pub static HT: u8 = 0x9;
|
||||
pub static LF: u8 = 0xA;
|
||||
pub static VT: u8 = 0xB;
|
||||
pub static FF: u8 = 0xC;
|
||||
pub static CR: u8 = 0xD;
|
||||
pub static SPACE: u8 = 0x20;
|
||||
pub static SPACES: &[u8] = &[HT, LF, VT, FF, CR, SPACE];
|
||||
pub static BLANK: &[u8] = &[HT, SPACE];
|
||||
Reference in New Issue
Block a user