Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions Cargo.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

1 change: 1 addition & 0 deletions fuzz/Cargo.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

1 change: 1 addition & 0 deletions src/uu/tr/Cargo.toml
Original file line number Diff line number Diff line change
Expand Up @@ -20,6 +20,7 @@ doctest = false
bytecount = { workspace = true, features = ["runtime-dispatch-simd"] }
clap = { workspace = true }
fluent = { workspace = true }
memchr = { workspace = true }
nom = { workspace = true }
uucore = { workspace = true, features = ["fs", "signals"] }

Expand Down
7 changes: 7 additions & 0 deletions src/uu/tr/benches/tr_bench.rs
Original file line number Diff line number Diff line change
Expand Up @@ -73,6 +73,13 @@ mod benches {
let data = text_data::generate_by_size(SIZE_MB, 80);
bench_tr_with_stdin(bencher, &data, &["-d", "a-z"]);
}

/// Delete a single character (the newlines).
#[divan::bench]
fn tr_delete_single_char(bencher: Bencher) {
let data = text_data::generate_by_size(SIZE_MB, 80);
bench_tr_with_stdin(bencher, &data, &["-d", "\\n"]);
}
}

fn main() {
Expand Down
64 changes: 38 additions & 26 deletions src/uu/tr/src/operation.rs
Original file line number Diff line number Diff line change
Expand Up @@ -29,7 +29,9 @@ use uucore::show_warning;

/// Common trait for operations that can process chunks of data
pub trait ChunkProcessor {
fn process_chunk(&self, input: &[u8], output: &mut Vec<u8>);
/// Return the bytes to write: `input` itself when it is left unchanged,
/// `output` otherwise.
fn process_chunk<'a>(&self, input: &'a [u8], output: &'a mut Vec<u8>) -> &'a [u8];
}

#[derive(Debug, Clone)]
Expand Down Expand Up @@ -809,48 +811,50 @@ fn set_to_bitmap(set: &[u8]) -> [bool; 256] {

#[derive(Debug)]
pub struct DeleteOperation {
pub(crate) delete_table: [bool; 256],
pub(crate) keep_table: [bool; 256],
/// The byte to delete, when it is the only one.
single_delete: Option<u8>,
}

impl DeleteOperation {
pub fn new(set: Vec<u8>) -> Self {
use crate::simd::find_single_change;

let keep_table = set_to_bitmap(&set).map(|delete| !delete);
let single_delete = find_single_change(&keep_table, |_, &keep| !keep).map(|(b, _)| b);
Self {
delete_table: set_to_bitmap(&set),
keep_table,
single_delete,
}
}
}

impl SymbolTranslator for DeleteOperation {
fn translate(&mut self, current: u8) -> Option<u8> {
// keep if not present in the delete set
(!self.delete_table[current as usize]).then_some(current)
self.keep_table[current as usize].then_some(current)
}
}

impl ChunkProcessor for DeleteOperation {
fn process_chunk(&self, input: &[u8], output: &mut Vec<u8>) {
use crate::simd::{find_single_change, process_single_delete};
fn process_chunk<'a>(&self, input: &'a [u8], output: &'a mut Vec<u8>) -> &'a [u8] {
use crate::simd::{process_delete, process_single_delete};

// Check if this is single character deletion
if let Some((delete_char, _)) =
find_single_change(&self.delete_table, |_, &should_delete| should_delete)
{
process_single_delete(input, output, delete_char);
if let Some(delete_char) = self.single_delete {
process_single_delete(input, output, delete_char, &self.keep_table)
} else {
// Standard deletion
output.extend(
input
.iter()
.filter(|&&b| !self.delete_table[b as usize])
.copied(),
);
process_delete(input, output, &self.keep_table);
output
}
}
}

#[derive(Debug)]
pub struct TranslateOperation {
pub(crate) translation_table: [u8; 256],
/// The byte to replace and its replacement, when it is the only one.
single_change: Option<(u8, u8)>,
}

impl TranslateOperation {
Expand All @@ -867,16 +871,26 @@ impl TranslateOperation {
translation_table[from as usize] = to;
}

Ok(Self { translation_table })
Ok(Self::from_table(translation_table))
} else if set1.is_empty() && set2.is_empty() {
// Identity mapping for empty sets
Ok(Self { translation_table })
Ok(Self::from_table(translation_table))
} else {
// Raised against the solved sets rather than what was typed, so
// there is nothing to point a caret at.
Err(BadSequence::EmptySet2WhenNotTruncatingSet1)
}
}

fn from_table(translation_table: [u8; 256]) -> Self {
use crate::simd::find_single_change;

let single_change = find_single_change(&translation_table, |i, &val| val != i as u8);
Self {
translation_table,
single_change,
}
}
}

impl SymbolTranslator for TranslateOperation {
Expand All @@ -886,18 +900,16 @@ impl SymbolTranslator for TranslateOperation {
}

impl ChunkProcessor for TranslateOperation {
fn process_chunk(&self, input: &[u8], output: &mut Vec<u8>) {
use crate::simd::{find_single_change, process_single_char_replace};
fn process_chunk<'a>(&self, input: &'a [u8], output: &'a mut Vec<u8>) -> &'a [u8] {
use crate::simd::process_single_char_replace;

// Check if this is a simple single-character translation
if let Some((source, target)) =
find_single_change(&self.translation_table, |i, &val| val != i as u8)
{
if let Some((source, target)) = self.single_change {
// Use SIMD-optimized single character replacement
process_single_char_replace(input, output, source, target);
process_single_char_replace(input, output, source, target)
} else {
// Standard translation using table lookup
output.extend(input.iter().map(|&b| self.translation_table[b as usize]));
output
}
}
}
Expand Down
82 changes: 69 additions & 13 deletions src/uu/tr/src/simd.rs
Original file line number Diff line number Diff line change
Expand Up @@ -28,16 +28,17 @@ where

/// SIMD-optimized single character replacement
#[inline]
pub fn process_single_char_replace(
input: &[u8],
output: &mut Vec<u8>,
pub fn process_single_char_replace<'a>(
input: &'a [u8],
output: &'a mut Vec<u8>,
source_char: u8,
target_char: u8,
) {
) -> &'a [u8] {
let count = bytecount::count(input, source_char);
if count == 0 {
output.extend_from_slice(input);
} else if count == input.len() {
return input;
}
if count == input.len() {
output.resize(output.len() + input.len(), target_char);
} else {
output.extend(
Expand All @@ -46,17 +47,70 @@ pub fn process_single_char_replace(
.map(|&b| if b == source_char { target_char } else { b }),
);
}
output
}

/// SIMD-optimized delete operation for single character
pub fn process_single_delete(input: &[u8], output: &mut Vec<u8>, delete_char: u8) {
///
/// `keep` must be false for `delete_char` only.
pub fn process_single_delete<'a>(
input: &'a [u8],
output: &'a mut Vec<u8>,
delete_char: u8,
keep: &[bool; 256],
) -> &'a [u8] {
let count = bytecount::count(input, delete_char);
if count == 0 {
output.extend_from_slice(input);
return input;
}
if count < input.len() / 128 {
// Below one match per 128 bytes, copying the runs between matches
// beats `process_delete`.
let mut start = 0;
for pos in memchr::memchr_iter(delete_char, input) {
output.extend_from_slice(&input[start..pos]);
start = pos + 1;
}
output.extend_from_slice(&input[start..]);
} else if count < input.len() {
output.extend(input.iter().filter(|&&b| b != delete_char).copied());
process_delete(input, output, keep);
}
// If count == input.len(), all deleted, output nothing
output
}

/// Append to `output` the bytes of `input` whose `keep` entry is true.
pub fn process_delete(input: &[u8], output: &mut Vec<u8>, keep: &[bool; 256]) {
// The index is always below `BLOCK` (a power of two), so the modulo is a
// mask that only serves to drop the bounds check.
const BLOCK: usize = 1024;
// Below one kept byte in `FEW`, a branch per byte is well predicted and
// stores less.
const FEW: usize = 64;
let mut block = [0; BLOCK];
// Guess for the first block from its start, then go by the previous one.
let start = &input[..input.len().min(256)];
let mut few_kept = start.iter().filter(|&&b| keep[b as usize]).count() * FEW < start.len();
for chunk in input.chunks(BLOCK) {
let mut kept = 0;
if few_kept {
// Only store the kept bytes.
for &b in chunk {
if keep[b as usize] {
block[kept % BLOCK] = b;
kept += 1;
}
}
} else {
// Store every byte and only advance past kept ones: no branch.
for &b in chunk {
block[kept % BLOCK] = b;
kept += usize::from(keep[b as usize]);
}
}
output.extend_from_slice(&block[..kept]);
few_kept = kept * FEW < chunk.len();
}
}

/// Unified I/O processing for all operations
Expand All @@ -79,18 +133,20 @@ where
};

output_buf.clear();
processor.process_chunk(&buf[..length], &mut output_buf);
let processed = processor.process_chunk(&buf[..length], &mut output_buf);

if !output_buf.is_empty() {
write_output(output, &output_buf)?;
if !processed.is_empty() {
write_output(output, processed)?;
}
}

Ok(())
}

/// Helper function to handle platform-specific write operations
#[inline]
// Kept out of line: inlined into `translate_input`, the raw `write` made the
// translator state go to memory on every byte, doubling the time of `tr -s`.
#[inline(never)]
pub fn write_output<W: Write>(output: &mut W, buf: &[u8]) -> UResult<()> {
#[cfg(not(windows))]
return output
Expand Down
19 changes: 12 additions & 7 deletions src/uu/tr/src/tr.rs
Original file line number Diff line number Diff line change
Expand Up @@ -97,7 +97,12 @@ pub fn uumain(args: impl uucore::Args) -> UResult<()> {

let stdin = stdin();
let mut locked_stdin = stdin.lock();
let mut locked_stdout = stdout().lock();
// Write straight to the file descriptor: `Stdout` is line buffered, which
// costs a search for the last newline and an extra write per chunk.
#[cfg(any(unix, target_os = "wasi"))]
let mut output = uucore::io::RawWriter(stdout());
#[cfg(not(any(unix, target_os = "wasi")))]
let mut output = stdout().lock();

// According to the man page: translating only happens if deleting or if a second set is given
let translating = !delete_flag && sets.len() > 1;
Expand Down Expand Up @@ -131,27 +136,27 @@ pub fn uumain(args: impl uucore::Args) -> UResult<()> {
let delete_op = DeleteOperation::new(set1);
let squeeze_op = SqueezeOperation::new(set2);
let op = delete_op.chain(squeeze_op);
translate_input(&mut locked_stdin, &mut locked_stdout, op)?;
translate_input(&mut locked_stdin, &mut output, op)?;
} else {
let op = DeleteOperation::new(set1);
process_input(&mut locked_stdin, &mut locked_stdout, &op)?;
process_input(&mut locked_stdin, &mut output, &op)?;
}
} else if squeeze_flag {
if sets_len == 1 {
let op = SqueezeOperation::new(set1);
translate_input(&mut locked_stdin, &mut locked_stdout, op)?;
translate_input(&mut locked_stdin, &mut output, op)?;
} else {
let translate_op = TranslateOperation::new(set1, set2.clone())?;
let squeeze_op = SqueezeOperation::new(set2);
let op = translate_op.chain(squeeze_op);
translate_input(&mut locked_stdin, &mut locked_stdout, op)?;
translate_input(&mut locked_stdin, &mut output, op)?;
}
} else {
let op = TranslateOperation::new(set1, set2)?;
process_input(&mut locked_stdin, &mut locked_stdout, &op)?;
process_input(&mut locked_stdin, &mut output, &op)?;
}

flush_output(&mut locked_stdout)?;
flush_output(&mut output)?;

Ok(())
}
Expand Down
Loading
Loading