Author SHA1 Message Date
Sylvestre Ledru be51c04c08 grep: support GREP_COLORS 'mt' and warn on deprecated GREP_COLOR
Two color-handling gaps vs GNU: the 'mt' capability in GREP_COLORS (which
sets both the selected- and context-match colors) was ignored, and the
deprecated GREP_COLOR variable produced no warning. Handle 'mt', and emit
GNU's 'GREP_COLOR=... is deprecated; use GREP_COLORS=mt=...' warning when
color output is actually produced. Fixes the GNU testsuite 'color-colors'
test.
2026-05-30 18:55:55 +02:00
Sylvestre Ledru da32a63663 grep: map invalid back-reference errors to GNU's wording
A back-reference to a non-existent group, e.g. (.)\2, makes oniguruma
fail with 'invalid backref number/name'. GNU words this per engine:
'reference to non-existent subpattern' under -P (PCRE2) and 'Invalid
back reference' for basic/extended (gnulib regex). Translate
ONIGERR_INVALID_BACKREF accordingly (gnu_error_message now takes the
regex mode). Fixes the GNU testsuite 'pcre-wx-backref' test.

Also drop pipe_in() from the compile-error tests added in the previous
commits: those patterns are rejected before stdin is read, so feeding
input raced with the child exiting and intermittently panicked the test
harness with a broken pipe under parallel execution.
2026-05-30 18:02:32 +02:00
Sylvestre Ledru 9c21a7d2f0 grep: handle regex backtracking-limit instead of panicking
A pathological pattern such as -P '((a+)*)+$' makes oniguruma exceed its
match retry limit. The onig crate's search_with_encoding/match_with_encoding
panic on that error, so uu_grep aborted with a Rust panic. Switch to the
fallible *_with_param variants and propagate the error: match_line/is_match
now return io::Result, the failure is mapped to GNU's wording ('exceeded
PCRE's backtracking limit') and surfaces as a normal exit-code-2 diagnostic.
Fixes the GNU testsuite 'pcre-abort' test.
2026-05-30 17:55:31 +02:00
Sylvestre Ledru e767a30c1a grep: emit GNU's 'Invalid range end' for reversed bracket ranges
A reversed range like [b-a] makes oniguruma fail with 'empty range in
char class', which uu_grep wrapped as 'invalid pattern "[b-a]": ...'.
GNU grep instead prints the bare POSIX diagnostic 'Invalid range end'
and exits 2. Translate the oniguruma error code
(ONIGERR_EMPTY_RANGE_IN_CHAR_CLASS) to GNU's wording, leaving other
compile errors to fall back to oniguruma's text. Fixes the GNU testsuite
'reversed-range-endpoints' test.
2026-05-30 17:50:02 +02:00
Sylvestre Ledru 1a3e8a391d grep: reject confusing [:name:] bracket syntax like GNU
GNU grep flags a bracket expression of the form [:name:] (an almost
certain misspelling of [[:name:]]) with a dedicated diagnostic and exit
code 2, whereas oniguruma silently treats it as the character set
{':','n','a','m','e'}. Port GNU's colon_warning_state logic from
parse_bracket_exp (gnulib dfa.c) so basic/extended patterns produce the
same error. Fixes the GNU testsuite 'warn-char-classes' test.
2026-05-30 17:39:29 +02:00
27 changed files with 500 additions and 3548 deletions
+1 -18
View File
@@ -9,9 +9,6 @@ on:
branches:
- '*'
permissions:
contents: write # Publish grep instead of discarding
# End the current execution if there is a new changeset in the PR.
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
@@ -50,21 +47,7 @@ jobs:
shell: bash
run: |
cd 'grep'
cargo build --release --config=profile.release.strip=true
tar -C target/release -cf - grep | zstd -19 -o ../grep-x86_64-unknown-linux-gnu.tar.zst
- name: Publish latest commit
uses: softprops/action-gh-release@v3
if: github.event_name == 'push' && github.ref == 'refs/heads/main'
with:
tag_name: latest-commit
body: |
commit: ${{ github.sha }}
draft: false
prerelease: true
files: |
grep-x86_64-unknown-linux-gnu.tar.zst
env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
cargo build --release
- name: Run GNU grep testsuite
shell: bash
-37
View File
@@ -1,37 +0,0 @@
name: CodSpeed
on:
push:
branches:
- "main"
pull_request:
# `workflow_dispatch` allows CodSpeed to trigger backtest
# performance analysis in order to generate initial data.
workflow_dispatch:
permissions:
contents: read
id-token: write
jobs:
codspeed:
name: Run benchmarks
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- name: Setup Rust toolchain, cache and cargo-codspeed binary
uses: moonrepo/setup-rust@v0
with:
channel: stable
cache-target: release
bins: cargo-codspeed
- name: Build the benchmark target(s)
run: cargo codspeed build
- name: Run the benchmarks
uses: CodSpeedHQ/action@v4
with:
mode: simulation
run: cargo codspeed run
-162
View File
@@ -1,162 +0,0 @@
name: Fuzzing
# spell-checker:ignore (people) taiki-e
# spell-checker:ignore (misc) fuzzer uufuzz
env:
CARGO_INCREMENTAL: "0"
on:
push:
branches: [ main, master ]
pull_request:
branches: [ main, master ]
permissions:
contents: read # to fetch code (actions/checkout)
# End the current execution if there is a new changeset in the PR.
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
cancel-in-progress: ${{ github.ref != 'refs/heads/main' }}
jobs:
uufuzz-examples:
name: Build and test uufuzz examples
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
with:
persist-credentials: false
- name: Install Rust
uses: dtolnay/rust-toolchain@stable
- name: Build uufuzz library
run: |
cd fuzz/uufuzz
cargo build --release
- name: Run uufuzz tests
run: |
cd fuzz/uufuzz
cargo test --lib
- name: Build and run uufuzz examples
run: |
cd fuzz/uufuzz
echo "Building all examples..."
cargo build --examples --release
# Run all examples except integration_testing (which has FD issues in CI)
for example in examples/*.rs; do
example_name=$(basename "$example" .rs)
if [ "$example_name" != "integration_testing" ]; then
cargo run --example "$example_name" --release
fi
done
fuzz-build:
name: Build the fuzzers
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
with:
persist-credentials: false
- name: Install Rust
uses: dtolnay/rust-toolchain@stable
- name: Install `cargo-fuzz`
uses: taiki-e/install-action@v2
with:
tool: cargo-fuzz
- name: Emulate a nightly toolchain
run: |
echo "RUSTC_BOOTSTRAP=1" >> "${GITHUB_ENV}"
- name: Run `cargo-fuzz build`
# Force the correct target
# https://github.com/rust-fuzz/cargo-fuzz/issues/398
run: cargo fuzz build --target $(rustc --print host-tuple)
fuzz-run:
needs: fuzz-build
name: Fuzz
runs-on: ubuntu-latest
timeout-minutes: 5
env:
RUN_FOR: 60
strategy:
fail-fast: false
matrix:
test-target:
# fuzz_grep is a differential fuzzer against GNU grep; it currently
# surfaces compatibility differences, so it is not expected to pass yet.
- { name: fuzz_grep, should_pass: false }
steps:
- uses: actions/checkout@v4
with:
persist-credentials: false
- name: Install Rust
uses: dtolnay/rust-toolchain@stable
- name: Install `cargo-fuzz`
uses: taiki-e/install-action@v2
with:
tool: cargo-fuzz
- name: Emulate a nightly toolchain
run: |
echo "RUSTC_BOOTSTRAP=1" >> "${GITHUB_ENV}"
- name: Run ${{ matrix.test-target.name }} for ${{ env.RUN_FOR }} seconds
id: run_fuzzer
shell: bash
continue-on-error: ${{ !matrix.test-target.should_pass }}
run: |
mkdir -p fuzz/stats
STATS_FILE="fuzz/stats/${{ matrix.test-target.name }}.txt"
# Force the correct target
# https://github.com/rust-fuzz/cargo-fuzz/issues/398
cargo fuzz run --target $(rustc --print host-tuple) ${{ matrix.test-target.name }} -- -max_total_time=${{ env.RUN_FOR }} -timeout=${{ env.RUN_FOR }} -detect_leaks=0 -print_final_stats=1 2>&1 | tee "$STATS_FILE"
# Save should_pass value for later inspection
echo "${{ matrix.test-target.should_pass }}" > "fuzz/stats/${{ matrix.test-target.name }}.should_pass"
# Print stats to job output for immediate visibility
echo "----------------------------------------"
echo "FUZZING STATISTICS FOR ${{ matrix.test-target.name }}"
echo "----------------------------------------"
echo "Runs: $(grep -q "stat::number_of_executed_units" "$STATS_FILE" && grep "stat::number_of_executed_units" "$STATS_FILE" | awk '{print $2}' || echo "unknown")"
echo "Execution Rate: $(grep -q "stat::average_exec_per_sec" "$STATS_FILE" && grep "stat::average_exec_per_sec" "$STATS_FILE" | awk '{print $2}' || echo "unknown") execs/sec"
echo "New Units: $(grep -q "stat::new_units_added" "$STATS_FILE" && grep "stat::new_units_added" "$STATS_FILE" | awk '{print $2}' || echo "unknown")"
echo "Expected: ${{ matrix.test-target.should_pass }}"
if grep -q "SUMMARY: " "$STATS_FILE"; then
echo "Status: $(grep "SUMMARY: " "$STATS_FILE" | head -1)"
else
echo "Status: Completed"
fi
echo "----------------------------------------"
# Add summary to GitHub step summary
echo "### Fuzzing Results for ${{ matrix.test-target.name }}" >> "$GITHUB_STEP_SUMMARY"
echo "" >> "$GITHUB_STEP_SUMMARY"
echo "| Metric | Value |" >> "$GITHUB_STEP_SUMMARY"
echo "|--------|-------|" >> "$GITHUB_STEP_SUMMARY"
if grep -q "stat::number_of_executed_units" "$STATS_FILE"; then
echo "| Runs | $(grep "stat::number_of_executed_units" "$STATS_FILE" | awk '{print $2}') |" >> "$GITHUB_STEP_SUMMARY"
fi
if grep -q "stat::average_exec_per_sec" "$STATS_FILE"; then
echo "| Execution Rate | $(grep "stat::average_exec_per_sec" "$STATS_FILE" | awk '{print $2}') execs/sec |" >> "$GITHUB_STEP_SUMMARY"
fi
if grep -q "stat::new_units_added" "$STATS_FILE"; then
echo "| New Units | $(grep "stat::new_units_added" "$STATS_FILE" | awk '{print $2}') |" >> "$GITHUB_STEP_SUMMARY"
fi
echo "| Should pass | ${{ matrix.test-target.should_pass }} |" >> "$GITHUB_STEP_SUMMARY"
if grep -q "SUMMARY: " "$STATS_FILE"; then
echo "| Status | $(grep "SUMMARY: " "$STATS_FILE" | head -1) |" >> "$GITHUB_STEP_SUMMARY"
else
echo "| Status | Completed |" >> "$GITHUB_STEP_SUMMARY"
fi
echo "" >> "$GITHUB_STEP_SUMMARY"
- name: Upload Stats
if: always()
uses: actions/upload-artifact@v4
with:
name: fuzz-stats-${{ matrix.test-target.name }}
path: |
fuzz/stats/${{ matrix.test-target.name }}.txt
fuzz/stats/${{ matrix.test-target.name }}.should_pass
retention-days: 5
-55
View File
@@ -1,55 +0,0 @@
# See https://pre-commit.com for more information
# See https://pre-commit.com/hooks.html for more hooks
exclude: ^tests/fixtures/
repos:
- repo: https://github.com/pre-commit/pre-commit-hooks
rev: v6.0.0
hooks:
- id: check-added-large-files
- id: check-executables-have-shebangs
- id: check-json
exclude: '\.vscode/(cSpell|extensions)\.json' # cSpell.json and extensions.json use comments
- id: check-shebang-scripts-are-executable
exclude: '.+\.rs' # would be triggered by #![some_attribute]
- id: check-symlinks
- id: check-toml
- id: check-yaml
args: [ --allow-multiple-documents ]
- id: destroyed-symlinks
- id: end-of-file-fixer
- id: mixed-line-ending
args: [ --fix=lf ]
- id: trailing-whitespace
- repo: local
hooks:
- id: rust-linting
name: Rust linting
description: Run cargo fmt on files included in the commit.
entry: cargo +stable fmt --
pass_filenames: true
types: [file, rust]
language: system
- id: rust-clippy
name: Rust clippy
description: Run cargo clippy on files included in the commit.
entry: cargo +stable clippy --workspace --all-targets --all-features -- -D warnings
pass_filenames: false
types: [file, rust]
language: system
- id: cargo-lock-check
name: Cargo.lock sync check
description: Ensure Cargo.lock and fuzz/Cargo.lock are up-to-date.
entry: bash -c 'for dir in . fuzz; do if [ -d "$dir" ]; then ( cd "$dir" && cargo fetch --quiet ); fi; done'
pass_filenames: false
files: 'Cargo\.(toml|lock)$'
language: system
- id: cspell
name: Code spell checker (cspell)
description: Run cspell to check for spelling errors (if available).
entry: bash -c 'if command -v cspell >/dev/null 2>&1; then cspell --no-must-find-files -- "$@"; else echo "cspell not found, skipping spell check"; exit 0; fi' --
pass_filenames: true
language: system
ci:
skip: [rust-linting, rust-clippy, cargo-lock-check, cspell]
Generated
+11 -532
View File
File diff suppressed because it is too large Load Diff
-5
View File
@@ -27,10 +27,5 @@ onig_sys = { version = "*", default-features = false }
uucore = "0.8.0"
walkdir = "2.5"
[[bench]]
name = "grep_bench"
harness = false
[dev-dependencies]
criterion = { version = "4.7.0", package = "codspeed-criterion-compat" }
uutests = "0.8.0"
-23
View File
@@ -4,30 +4,12 @@
[![dependency status](https://deps.rs/repo/github/uutils/grep/status.svg)](https://deps.rs/repo/github/uutils/grep)
[![CodeCov](https://codecov.io/gh/uutils/grep/branch/main/graph/badge.svg)](https://codecov.io/gh/uutils/grep)
[![CodSpeed](https://img.shields.io/endpoint?url=https://codspeed.io/badge.json)](https://codspeed.io/uutils/grep?utm_source=badge)
# Grep, now in Rust
A Rust implementation of [GNU Grep](https://www.gnu.org/software/grep/).
This project is an initial release and may contain bugs.
## Install
```shell
cargo install uu_grep
```
## 🚀 Try it online
You can try `grep` directly in your browser on the [uutils playground](https://uutils.github.io/playground/).
Arguments (and a full command) can be passed through the URL via the `cmd` query parameter, for example:
```shell
printf '🚀 rocket\n🛰️ satellite\n🌙 moon\n⭐ star\n' | grep 🌙
```
[Run it in the playground](https://uutils.github.io/playground/?cmd=printf%20%27%F0%9F%9A%80%20rocket%5Cn%F0%9F%9B%B0%EF%B8%8F%20satellite%5Cn%F0%9F%8C%99%20moon%5Cn%E2%AD%90%20star%5Cn%27%20%7C%20grep%20%F0%9F%8C%99)
## Building
Download Rust at: https://rustup.rs/
@@ -47,15 +29,10 @@ cargo build --release
cargo test
```
## Pre-commit hooks
This project uses [pre-commit](https://pre-commit.com); run `pre-commit install` to enable the git hooks.
## Known Issues
* Does not take `LANG`, etc., into account for handling file encodings (non-UTF8 matches are treated as binary)
* No localization support yet
* Performances need to be improved
## Contributing
-128
View File
@@ -1,128 +0,0 @@
// This file is part of the uutils grep package.
//
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
use criterion::{Criterion, black_box, criterion_group, criterion_main};
use std::ffi::OsString;
use std::path::Path;
/// Run grep end-to-end through the real `uumain` entry point. `args` are the
/// arguments after the program name (flags, pattern, paths). The exit status is
/// ignored — we only care about the work performed.
fn run(args: &[&str]) {
let mut argv: Vec<OsString> = Vec::with_capacity(args.len() + 1);
argv.push(OsString::from("grep"));
argv.extend(args.iter().map(OsString::from));
let _ = uu_grep::uumain(argv.into_iter());
}
/// Build a multi-megabyte log-like corpus plus a directory holding it alongside
/// a binary file. Every line contains `worker-<n>` and a `2024-…` timestamp; a
/// rare `RAREHIT` marker appears on a handful of lines (≈ every 10000th).
/// Returns `(dir, log_file)`.
fn build_corpus() -> (std::path::PathBuf, std::path::PathBuf) {
let mut content = String::new();
for i in 0..80_000u32 {
if i % 10_000 == 0 {
content.push_str(&format!(
"2024-01-15 10:30:{:02} RAREHIT worker-{i} special marker seen\n",
i % 60
));
} else if i % 100 == 0 {
content.push_str(&format!(
"2024-01-15 10:30:{:02} ERROR worker-{i} connection reset\n",
i % 60
));
} else {
content.push_str(&format!(
"2024-01-15 10:30:{:02} INFO worker-{i} request handled in {}ms\n",
i % 60,
i % 1000
));
}
}
assert!(content.len() > 4 * 1024 * 1024);
let dir = std::env::temp_dir().join(format!("uu_grep_bench_{}", std::process::id()));
std::fs::create_dir_all(&dir).unwrap();
let log = dir.join("app.log");
std::fs::write(&log, &content).unwrap();
// A binary file (contains NUL) that also holds the marker, so `-I` has
// something to skip while recursing.
let mut binary = vec![0u8, 1, 2, 3];
binary.extend_from_slice(b"RAREHIT in binary blob");
binary.extend(std::iter::repeat_n(0u8, 4096));
std::fs::write(dir.join("data.bin"), &binary).unwrap();
(dir, log)
}
fn bench_e2e(c: &mut Criterion) {
let (dir, log) = build_corpus();
let file = log.to_str().unwrap();
let dir_str = dir.to_str().unwrap();
// Pure scanning throughput: `-q` with a pattern that never matches forces a
// full scan and produces no output. A literal (which a buffer-at-a-time
// searcher can accelerate) versus an extended-regex control (which cannot).
{
let mut group = c.benchmark_group("scan");
group.bench_function("literal_no_match", |b| {
b.iter(|| run(black_box(&["-q", "NONEXISTENT_TOKEN_XYZ", file])))
});
group.bench_function("regex_no_match", |b| {
b.iter(|| run(black_box(&["-q", "-E", "NON[0-9]EXISTENT_TOKEN", file])))
});
group.finish();
}
// Real invocation shapes from the `grep` tldr page, each scanning the whole
// corpus. The `RAREHIT` marker matches only a handful of lines, so output
// stays small while the full-file scan dominates.
{
let mut group = c.benchmark_group("usage");
// Search for a pattern within a file.
group.bench_function("search_pattern", |b| {
b.iter(|| run(black_box(&["RAREHIT", file])))
});
// Search for an exact string (-F).
group.bench_function("fixed_string", |b| {
b.iter(|| run(black_box(&["-F", "RAREHIT", file])))
});
// Recursive search ignoring binary files (-rI).
group.bench_function("recursive_no_binary", |b| {
b.iter(|| run(black_box(&["-rI", "RAREHIT", dir_str])))
});
// Print 3 lines of context (-C 3).
group.bench_function("context", |b| {
b.iter(|| run(black_box(&["-C", "3", "RAREHIT", file])))
});
// Filename + line number with forced color (-Hn --color=always).
group.bench_function("filename_lineno_color", |b| {
b.iter(|| run(black_box(&["-Hn", "--color=always", "RAREHIT", file])))
});
// Print only the matched text (-o).
group.bench_function("only_matching", |b| {
b.iter(|| run(black_box(&["-o", "RAREHIT", file])))
});
// Invert match (-v); `worker-` is on every line, so nothing is printed
// and this measures the full inverted scan.
group.bench_function("invert_match", |b| {
b.iter(|| run(black_box(&["-v", "worker-", file])))
});
// Extended regex, case-insensitive (-Ei).
group.bench_function("extended_icase", |b| {
b.iter(|| run(black_box(&["-Ei", "rarehit", file])))
});
group.finish();
}
let _ = std::fs::remove_dir_all(Path::new(dir_str));
}
criterion_group!(benches, bench_e2e);
criterion_main!(benches);
-2
View File
@@ -1,2 +0,0 @@
[build]
rustflags = ["--cfg", "fuzzing"]
-4
View File
@@ -1,4 +0,0 @@
target
corpus
artifacts
Cargo.lock
-33
View File
@@ -1,33 +0,0 @@
[package]
name = "uu_grep-fuzz"
version = "0.0.0"
description = "uutils ~ 'grep' fuzzers"
repository = "https://github.com/microsoft/uutils-grep/tree/main/fuzz/"
edition = "2024"
rust-version = "1.88.0"
license = "MIT"
publish = false
[package.metadata]
cargo-fuzz = true
# Prevent this from interfering with the parent workspace
[workspace]
members = ["."]
# Enable debug symbols in release builds for readable backtraces
# when fuzzing discovers crashes.
[profile.release]
debug = true
[dependencies]
libfuzzer-sys = "0.4.7"
rand = { version = "0.10.1", features = ["std_rng"] }
uufuzz = { path = "uufuzz" }
uu_grep = { path = ".." }
[[bin]]
name = "fuzz_grep"
path = "fuzz_targets/fuzz_grep.rs"
test = false
doc = false
-189
View File
@@ -1,189 +0,0 @@
// This file is part of the uutils grep package.
//
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
// spell-checker:ignore uumain seedable
#![no_main]
use libfuzzer_sys::fuzz_target;
use uu_grep::uumain;
use rand::prelude::IndexedRandom;
use rand::rngs::StdRng;
use rand::{RngExt, SeedableRng};
use std::ffi::OsString;
use uufuzz::{CommandResult, compare_result, generate_and_run_uumain, run_gnu_cmd};
static CMD_PATH: &str = "grep";
/// Derive a 32-byte RNG seed from the libFuzzer input so that every run is a
/// pure function of `data`. This is what makes crash artifacts reproducible:
/// the same bytes always generate the same pattern/args/input.
fn seed_from_data(data: &[u8]) -> StdRng {
let mut seed = [0u8; 32];
for (i, b) in data.iter().enumerate() {
seed[i % 32] ^= b;
}
StdRng::from_seed(seed)
}
/// Random string mixing valid UTF-8 (incl. multi-byte) and the occasional
/// invalid byte. Driven by the caller's seeded RNG so output is deterministic.
fn gen_random_string(rng: &mut StdRng, max_length: usize) -> String {
let valid_utf8: Vec<char> =
"ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789🔩🪛🪓⚙️🔗🧰"
.chars()
.collect();
let invalid_utf8 = [0xC3u8, 0x28];
let mut result = String::new();
for _ in 0..rng.random_range(0..=max_length) {
if rng.random_bool(0.9) {
result.push(*valid_utf8.choose(rng).unwrap());
} else if let Some(c) = char::from_u32(*invalid_utf8.choose(rng).unwrap() as u32) {
result.push(c);
}
}
result
}
/// Generate a (mostly) meaningful set of grep flags, occasionally throwing in
/// garbage to exercise error handling.
fn generate_grep_args(rng: &mut StdRng) -> Vec<OsString> {
let arg_count = rng.random_range(0..=5);
let mut args = Vec::new();
for _ in 0..arg_count {
// Small chance of an invalid argument.
if rng.random_bool(0.1) {
let len = rng.random_range(1..=10);
args.push(OsString::from(gen_random_string(rng, len)));
continue;
}
match rng.random_range(0..=15) {
0 => args.push(OsString::from("-i")), // ignore case
1 => args.push(OsString::from("-v")), // invert match
2 => args.push(OsString::from("-c")), // count
3 => args.push(OsString::from("-n")), // line number
4 => args.push(OsString::from("-o")), // only matching
5 => args.push(OsString::from("-w")), // word boundaries
6 => args.push(OsString::from("-x")), // whole line match
7 => args.push(OsString::from("-F")), // fixed strings
8 => args.push(OsString::from("-E")), // extended regexp
9 => args.push(OsString::from("-G")), // basic regexp
10 => args.push(OsString::from("--null-data")),
11 => args.push(OsString::from("--byte-offset")),
12 => {
// max-count
args.push(OsString::from("-m"));
args.push(OsString::from(rng.random_range(0..=5).to_string()));
}
13 => {
// after-context
args.push(OsString::from("-A"));
args.push(OsString::from(rng.random_range(0..=3).to_string()));
}
14 => {
// before-context
args.push(OsString::from("-B"));
args.push(OsString::from(rng.random_range(0..=3).to_string()));
}
15 => args.push(OsString::from("-s")), // suppress error messages
_ => (),
}
}
args
}
/// Build a pattern. Sometimes a literal token, sometimes a small regex made of
/// random characters and metacharacters.
fn generate_pattern(rng: &mut StdRng) -> String {
match rng.random_range(0..=3) {
0 => {
let len = rng.random_range(1..=5);
gen_random_string(rng, len)
}
1 => {
// A small alternation / anchored regex.
let la = rng.random_range(1..=3);
let a = gen_random_string(rng, la);
let lb = rng.random_range(1..=3);
let b = gen_random_string(rng, lb);
format!("{a}|{b}")
}
2 => {
let lb = rng.random_range(1..=3);
let base = gen_random_string(rng, lb);
let meta = ["*", "+", "?", ".", "^", "$", ".*", "[a-z]", "\\w"];
let m = meta[rng.random_range(0..meta.len())];
format!("{base}{m}")
}
_ => {
// Pick one of a few hand-written patterns that exercise common paths.
let canned = ["a", "^", "$", ".", ".*", "[0-9]+", "\\b", "()"];
canned[rng.random_range(0..canned.len())].to_string()
}
}
}
/// Generate input text with a mix of short and long lines.
fn generate_input(rng: &mut StdRng, count: usize) -> String {
let mut lines = Vec::new();
for _ in 0..count {
if rng.random_bool(0.1) {
let len = rng.random_range(200..=500);
lines.push(gen_random_string(rng, len));
} else {
let len = rng.random_range(0..=20);
lines.push(gen_random_string(rng, len));
}
}
lines.join("\n")
}
fuzz_target!(|data: &[u8]| {
let mut rng = seed_from_data(data);
let pattern = generate_pattern(&mut rng);
// Pass the pattern through `-e` so it is never mistaken for a flag, then
// append the (possibly invalid) extra arguments.
let mut args = vec![
OsString::from("grep"),
OsString::from("-e"),
OsString::from(&pattern),
];
args.extend(generate_grep_args(&mut rng));
let input = generate_input(&mut rng, 10);
let rust_result = generate_and_run_uumain(&args, uumain, Some(&input));
let gnu_result = match run_gnu_cmd(CMD_PATH, &args[1..], false, Some(&input)) {
Ok(result) => result,
Err(error_result) => {
eprintln!("Failed to run GNU command:");
eprintln!("Stderr: {}", error_result.stderr);
eprintln!("Exit Code: {}", error_result.exit_code);
CommandResult {
stdout: String::new(),
stderr: error_result.stderr,
exit_code: error_result.exit_code,
}
}
};
compare_result(
"grep",
&format!("{:?}", &args[1..]),
Some(&input),
&rust_result,
&gnu_result,
false, // Set to true if you want to fail on stderr diff
);
});
-16
View File
@@ -1,16 +0,0 @@
[package]
name = "uufuzz"
description = "uutils ~ 'core' uutils fuzzing library"
repository = "https://github.com/uutils/coreutils/tree/main/fuzz/uufuzz"
version = "0.8.0"
edition = "2024"
rust-version = "1.88.0"
license = "MIT"
[dependencies]
console = "0.16.0"
rand = { version = "0.10.1", features = ["std_rng"] }
similar = "3.0.0"
uucore = { version = "0.8.0", features = ["parser"] }
tempfile = "3.15.0"
rustix = { version = "1.1.4", features = ["stdio", "pipe"] }
-137
View File
@@ -1,137 +0,0 @@
# uufuzz
A Rust library for **differential fuzzing** of command-line utilities. Originally designed for testing uutils coreutils against GNU coreutils, but can be used to compare any two implementations of command-line tools.
Differential fuzzing is a testing technique that compares the behavior of two implementations of the same functionality using randomly generated inputs. This helps identify bugs, inconsistencies, and security vulnerabilities by finding cases where implementations diverge unexpectedly.
## Features
- **Command Execution**: Run and capture output from both Rust and reference implementations
- **Result Comparison**: Detailed comparison of stdout, stderr, and exit codes with diff output
- **Input Generation**: Utilities for generating random strings, files, and test inputs
- **GNU Compatibility**: Built-in support for detecting and running GNU coreutils
- **Pretty Output**: Colorized and formatted test result display
## Usage
Add to your `Cargo.toml`:
```toml
[dependencies]
uufuzz = "0.1.0"
```
### Basic Example
```rust
use std::ffi::OsString;
use uufuzz::{generate_and_run_uumain, run_gnu_cmd, compare_result};
// Your utility's main function
fn my_echo_main(args: std::vec::IntoIter<OsString>) -> i32 {
// Implementation here
0
}
// Test against GNU implementation
let args = vec![OsString::from("echo"), OsString::from("hello")];
// Run your implementation
let rust_result = generate_and_run_uumain(&args, my_echo_main, None);
// Run GNU implementation
let gnu_result = run_gnu_cmd("echo", &args[1..], false, None).unwrap();
// Compare results
compare_result("echo", "hello", None, &rust_result, &gnu_result, true);
```
### With Pipe Input
```rust
let pipe_input = "test data";
let rust_result = generate_and_run_uumain(&args, my_cat_main, Some(pipe_input));
let gnu_result = run_gnu_cmd("cat", &args[1..], false, Some(pipe_input)).unwrap();
compare_result("cat", "", Some(pipe_input), &rust_result, &gnu_result, true);
```
### Random Input Generation
```rust
use uufuzz::{generate_random_string, generate_random_file};
// Generate random string up to 50 characters
let random_input = generate_random_string(50);
// Generate random temporary file
let file_path = generate_random_file().expect("Failed to create file");
```
## Use Cases
### Fuzzing Testing
Perfect for libFuzzer-based differential fuzzing:
```rust
#![no_main]
use libfuzzer_sys::fuzz_target;
use uufuzz::*;
fuzz_target!(|_data: &[u8]| {
let args = generate_test_args();
let rust_result = generate_and_run_uumain(&args, my_utility_main, None);
let gnu_result = run_gnu_cmd("utility", &args[1..], false, None).unwrap();
compare_result("utility", &format!("{:?}", args), None, &rust_result, &gnu_result, true);
});
```
### Integration Testing
Use in regular test suites to verify compatibility:
```rust
#[test]
fn test_basic_functionality() {
let args = vec![OsString::from("sort"), OsString::from("-n")];
let input = "3\n1\n2\n";
let rust_result = generate_and_run_uumain(&args, sort_main, Some(input));
let gnu_result = run_gnu_cmd("sort", &args[1..], false, Some(input)).unwrap();
assert_eq!(rust_result.stdout, gnu_result.stdout);
assert_eq!(rust_result.exit_code, gnu_result.exit_code);
}
```
## Environment Variables
- `LC_ALL=C` - Automatically set when running GNU commands for consistent behavior
## Platform Support
- **Linux**: Full support with GNU coreutils
- **macOS**: Works with GNU coreutils via Homebrew (`brew install coreutils`)
- **Windows**: Limited support (depends on available reference implementations)
## Examples
The library includes several working examples in the `examples/` directory:
### Running Examples
```bash
# Basic differential comparison
cargo run --example basic_echo
# Pipe input handling
cargo run --example pipe_input
# Simple integration testing (recommended approach)
cargo run --example simple_integration
# Complex integration testing (demonstrates file descriptor handling issues)
cargo run --example integration_testing
```
## License
Licensed under the MIT License, same as uutils coreutils.
-61
View File
@@ -1,61 +0,0 @@
// This file is part of the uutils coreutils package.
//
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
use std::ffi::OsString;
use uufuzz::{compare_result, generate_and_run_uumain, run_gnu_cmd};
// Mock echo implementation for demonstration
fn mock_echo_main(args: std::vec::IntoIter<OsString>) -> i32 {
let args: Vec<OsString> = args.collect();
// Skip the program name (first argument)
for (i, arg) in args.iter().skip(1).enumerate() {
if i > 0 {
print!(" ");
}
print!("{}", arg.to_string_lossy());
}
println!();
0
}
fn main() {
println!("=== Basic uufuzz Example ===");
// Test against GNU implementation
let args = vec![
OsString::from("echo"),
OsString::from("hello"),
OsString::from("world"),
];
println!("Running mock echo implementation...");
let rust_result = generate_and_run_uumain(&args, mock_echo_main, None);
println!("Running GNU echo...");
match run_gnu_cmd("echo", &args[1..], false, None) {
Ok(gnu_result) => {
println!("Comparing results...");
compare_result(
"echo",
"hello world",
None,
&rust_result,
&gnu_result,
false,
);
}
Err(error_result) => {
println!("Failed to run GNU echo: {}", error_result.stderr);
println!("This is expected if GNU coreutils is not installed");
// Show what our implementation produced
println!("\nOur implementation result:");
println!("Stdout: '{}'", rust_result.stdout);
println!("Stderr: '{}'", rust_result.stderr);
println!("Exit code: {}", rust_result.exit_code);
}
}
}
-163
View File
@@ -1,163 +0,0 @@
// This file is part of the uutils coreutils package.
//
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
use rand::RngExt;
use std::ffi::OsString;
use uufuzz::{generate_and_run_uumain, generate_random_string, run_gnu_cmd};
// Mock echo implementation with some bugs for demonstration
fn mock_buggy_echo_main(args: std::vec::IntoIter<OsString>) -> i32 {
let args: Vec<OsString> = args.collect();
let mut should_add_newline = true;
let mut enable_escapes = false;
let mut start_index = 1;
// Parse arguments (simplified)
for arg in args.iter().skip(1) {
let arg_str = arg.to_string_lossy();
if arg_str == "-n" {
should_add_newline = false;
start_index += 1;
} else if arg_str == "-e" {
enable_escapes = true;
start_index += 1;
} else {
break;
}
}
// Print arguments
for (i, arg) in args.iter().skip(start_index).enumerate() {
if i > 0 {
print!(" ");
}
let arg_str = arg.to_string_lossy();
if enable_escapes {
// Simulate a bug: incomplete escape sequence handling
let processed = arg_str.replace("\\n", "\n").replace("\\t", "\t");
print!("{}", processed);
} else {
print!("{}", arg_str);
}
}
if should_add_newline {
println!();
}
0
}
// Generate test arguments for echo command
fn generate_echo_args() -> Vec<OsString> {
let mut rng = rand::rng();
let mut args = vec![OsString::from("echo")];
// Randomly add flags
if rng.random_bool(0.3) {
// 30% chance
args.push(OsString::from("-n"));
}
if rng.random_bool(0.2) {
// 20% chance
args.push(OsString::from("-e"));
}
// Add 1-3 random string arguments
let num_args = rng.random_range(1..=3);
for _ in 0..num_args {
let arg = generate_random_string(rng.random_range(1..=15));
args.push(OsString::from(arg));
}
args
}
fn main() {
println!("=== Fuzzing Simulation uufuzz Example ===");
println!("This simulates how libFuzzer would test our echo implementation");
println!("against GNU echo with random inputs.\n");
let num_tests = 10;
let mut passed = 0;
let mut failed = 0;
for i in 1..=num_tests {
println!("--- Fuzz Test {} ---", i);
let args = generate_echo_args();
println!(
"Testing with args: {:?}",
args.iter().map(|s| s.to_string_lossy()).collect::<Vec<_>>()
);
// Run our implementation
let rust_result = generate_and_run_uumain(&args, mock_buggy_echo_main, None);
// Run GNU implementation
match run_gnu_cmd("echo", &args[1..], false, None) {
Ok(gnu_result) => {
// Check if results match
let stdout_match = rust_result.stdout.trim() == gnu_result.stdout.trim();
let exit_code_match = rust_result.exit_code == gnu_result.exit_code;
if stdout_match && exit_code_match {
println!("✓ PASS: Implementations match");
passed += 1;
} else {
println!("✗ FAIL: Implementations differ");
failed += 1;
// Show the difference in a controlled way (not panicking like compare_result)
if !stdout_match {
println!(" Stdout difference:");
println!(
" Ours: '{}'",
rust_result.stdout.trim().replace('\n', "\\n")
);
println!(
" GNU: '{}'",
gnu_result.stdout.trim().replace('\n', "\\n")
);
}
if !exit_code_match {
println!(
" Exit code difference: {} vs {}",
rust_result.exit_code, gnu_result.exit_code
);
}
}
}
Err(error_result) => {
println!("⚠ GNU echo not available: {}", error_result.stderr);
println!(" Our result: '{}'", rust_result.stdout.trim());
// Don't count this as pass or fail
continue;
}
}
println!();
}
println!("=== Fuzzing Results ===");
println!("Total tests: {}", num_tests);
println!("Passed: {}", passed);
println!("Failed: {}", failed);
if failed > 0 {
println!(
"\n⚠ Found {} discrepancies! In real fuzzing, these would be investigated.",
failed
);
println!("This demonstrates how differential fuzzing can find bugs in implementations.");
} else {
println!("\n✓ All tests passed! The implementations appear compatible.");
}
println!("\nIn a real libfuzzer setup, this would run thousands of iterations");
println!("automatically with more sophisticated input generation.");
}
-236
View File
@@ -1,236 +0,0 @@
// This file is part of the uutils coreutils package.
//
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
use std::ffi::OsString;
use uufuzz::{generate_and_run_uumain, run_gnu_cmd};
// Mock sort implementation for demonstration
fn mock_sort_main(args: std::vec::IntoIter<OsString>) -> i32 {
use std::io::{self, Read};
let args: Vec<OsString> = args.collect();
let mut numeric_sort = false;
let mut reverse_sort = false;
// Parse arguments
for arg in args.iter().skip(1) {
let arg_str = arg.to_string_lossy();
match arg_str.as_ref() {
"-n" | "--numeric-sort" => numeric_sort = true,
"-r" | "--reverse" => reverse_sort = true,
_ => {}
}
}
// Read from stdin
let mut input = String::new();
match io::stdin().read_to_string(&mut input) {
Ok(_) => {
let mut lines: Vec<&str> = input.lines().collect();
if numeric_sort {
// Sort numerically
lines.sort_by(|a, b| {
let a_num: f64 = a.trim().parse().unwrap_or(0.0);
let b_num: f64 = b.trim().parse().unwrap_or(0.0);
a_num.partial_cmp(&b_num).unwrap()
});
} else {
// Sort lexically
lines.sort();
}
if reverse_sort {
lines.reverse();
}
for line in lines {
println!("{}", line);
}
0
}
Err(_) => {
eprintln!("Error reading from stdin");
1
}
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_basic_sort_functionality() {
println!("Testing basic sort functionality...");
let args = vec![OsString::from("sort")];
let input = "zebra\napple\nbanana\n";
let rust_result = generate_and_run_uumain(&args, mock_sort_main, Some(input));
match run_gnu_cmd("sort", &args[1..], false, Some(input)) {
Ok(gnu_result) => {
// In test environment, stdout might not be captured properly
// Just verify the function runs without errors and exit codes match
assert_eq!(
rust_result.exit_code, gnu_result.exit_code,
"Exit codes should match"
);
println!("✓ Basic sort test passed (exit codes match)");
}
Err(_) => {
// GNU sort not available, just check our implementation runs
assert_eq!(
rust_result.exit_code, 0,
"Our sort should exit successfully"
);
println!("✓ Basic sort test passed (GNU sort not available)");
}
}
}
#[test]
fn test_numeric_sort() {
println!("Testing numeric sort...");
let args = vec![OsString::from("sort"), OsString::from("-n")];
let input = "10\n2\n1\n20\n";
let rust_result = generate_and_run_uumain(&args, mock_sort_main, Some(input));
match run_gnu_cmd("sort", &args[1..], false, Some(input)) {
Ok(gnu_result) => {
assert_eq!(
rust_result.exit_code, gnu_result.exit_code,
"Exit codes should match"
);
println!("✓ Numeric sort test passed (exit codes match)");
}
Err(_) => {
// GNU sort not available, just check our implementation runs
assert_eq!(
rust_result.exit_code, 0,
"Our numeric sort should exit successfully"
);
println!("✓ Numeric sort test passed (GNU sort not available)");
}
}
}
#[test]
fn test_reverse_sort() {
println!("Testing reverse sort...");
let args = vec![OsString::from("sort"), OsString::from("-r")];
let input = "apple\nbanana\nzebra\n";
let rust_result = generate_and_run_uumain(&args, mock_sort_main, Some(input));
match run_gnu_cmd("sort", &args[1..], false, Some(input)) {
Ok(gnu_result) => {
assert_eq!(
rust_result.exit_code, gnu_result.exit_code,
"Exit codes should match"
);
println!("✓ Reverse sort test passed (exit codes match)");
}
Err(_) => {
// GNU sort not available, just check our implementation runs
assert_eq!(
rust_result.exit_code, 0,
"Our reverse sort should exit successfully"
);
println!("✓ Reverse sort test passed (GNU sort not available)");
}
}
}
#[test]
fn test_empty_input() {
println!("Testing empty input...");
let args = vec![OsString::from("sort")];
let input = "";
let rust_result = generate_and_run_uumain(&args, mock_sort_main, Some(input));
match run_gnu_cmd("sort", &args[1..], false, Some(input)) {
Ok(gnu_result) => {
assert_eq!(
rust_result.exit_code, gnu_result.exit_code,
"Exit codes should match"
);
println!("✓ Empty input test passed (exit codes match)");
}
Err(_) => {
// GNU sort not available, just check our implementation runs
assert_eq!(
rust_result.exit_code, 0,
"Should exit successfully with empty input"
);
println!("✓ Empty input test passed (GNU sort not available)");
}
}
}
}
fn main() {
println!("=== Integration Testing uufuzz Example ===");
println!("This demonstrates how to use uufuzz in regular test suites");
println!("to verify compatibility with reference implementations.\n");
println!("Run 'cargo test --example integration_testing' to execute the tests.");
println!("Or run individual tests below for demonstration:\n");
// Demonstrate the tests manually
let test_cases = [
(
"Basic lexical sort",
vec![OsString::from("sort")],
"zebra\napple\nbanana\n",
),
(
"Numeric sort",
vec![OsString::from("sort"), OsString::from("-n")],
"10\n2\n1\n20\n",
),
(
"Reverse sort",
vec![OsString::from("sort"), OsString::from("-r")],
"apple\nbanana\nzebra\n",
),
("Empty input", vec![OsString::from("sort")], ""),
];
for (test_name, args, input) in test_cases {
println!("--- {} ---", test_name);
println!(
"Args: {:?}",
args.iter().map(|s| s.to_string_lossy()).collect::<Vec<_>>()
);
println!("Input: {:?}", input.replace('\n', "\\n"));
let rust_result = generate_and_run_uumain(&args, mock_sort_main, Some(input));
println!("Our output: {:?}", rust_result.stdout.replace('\n', "\\n"));
println!("Exit code: {}", rust_result.exit_code);
match run_gnu_cmd("sort", &args[1..], false, Some(input)) {
Ok(gnu_result) => {
println!("GNU output: {:?}", gnu_result.stdout.replace('\n', "\\n"));
if rust_result.stdout == gnu_result.stdout
&& rust_result.exit_code == gnu_result.exit_code
{
println!("✓ Outputs match!");
} else {
println!("✗ Outputs differ!");
}
}
Err(_) => {
println!("GNU sort not available for comparison");
}
}
println!();
}
println!("=== Example completed ===");
println!("In a real test suite, assertions would ensure compatibility.");
}
-68
View File
@@ -1,68 +0,0 @@
// This file is part of the uutils coreutils package.
//
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
use std::ffi::OsString;
use std::io::{self, Read};
use uufuzz::{compare_result, generate_and_run_uumain, run_gnu_cmd};
// Mock cat implementation for demonstration
fn mock_cat_main(args: std::vec::IntoIter<OsString>) -> i32 {
let _args: Vec<OsString> = args.collect();
// Read from stdin and write to stdout
let mut input = String::new();
match io::stdin().read_to_string(&mut input) {
Ok(_) => {
print!("{}", input);
0
}
Err(_) => {
eprintln!("Error reading from stdin");
1
}
}
}
fn main() {
println!("=== Pipe Input uufuzz Example ===");
let args = vec![OsString::from("cat")];
let pipe_input = "Hello from pipe!\nThis is line 2.\nAnd line 3.";
println!("Running mock cat implementation with pipe input...");
let rust_result = generate_and_run_uumain(&args, mock_cat_main, Some(pipe_input));
println!("Running GNU cat with pipe input...");
match run_gnu_cmd("cat", &args[1..], false, Some(pipe_input)) {
Ok(gnu_result) => {
println!("Comparing results...");
compare_result(
"cat",
"",
Some(pipe_input),
&rust_result,
&gnu_result,
false,
);
}
Err(error_result) => {
println!("Failed to run GNU cat: {}", error_result.stderr);
println!("This is expected if GNU coreutils is not installed");
// Show what our implementation produced
println!("\nOur implementation result:");
println!("Stdout: '{}'", rust_result.stdout);
println!("Stderr: '{}'", rust_result.stderr);
println!("Exit code: {}", rust_result.exit_code);
// Verify our mock implementation works
if rust_result.stdout.trim() == pipe_input.trim() {
println!("✓ Our mock cat implementation correctly echoed the pipe input");
} else {
println!("✗ Our mock cat implementation failed to echo the pipe input correctly");
}
}
}
}
-105
View File
@@ -1,105 +0,0 @@
// This file is part of the uutils coreutils package.
//
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
use std::ffi::OsString;
use uufuzz::{CommandResult, run_gnu_cmd};
fn main() {
println!("=== Simple Integration Testing uufuzz Example ===");
println!("This demonstrates how to use uufuzz to compare against GNU tools");
println!("without the complex file descriptor manipulation.\n");
// Test cases that work well with external command comparison
let test_cases = [
(
"echo test",
"echo",
vec![OsString::from("hello"), OsString::from("world")],
None,
),
(
"echo with flag",
"echo",
vec![OsString::from("-n"), OsString::from("no-newline")],
None,
),
(
"cat with input",
"cat",
vec![],
Some("Hello from cat!\nLine 2\n"),
),
("sort basic", "sort", vec![], Some("zebra\napple\nbanana\n")),
(
"sort numeric",
"sort",
vec![OsString::from("-n")],
Some("10\n2\n1\n20\n"),
),
];
for (test_name, cmd, args, input) in test_cases {
println!("--- {} ---", test_name);
// Run GNU command
match run_gnu_cmd(cmd, &args, false, input) {
Ok(gnu_result) => {
println!("✓ GNU {} succeeded", cmd);
println!(
" Stdout: {:?}",
gnu_result.stdout.trim().replace('\n', "\\n")
);
println!(" Exit code: {}", gnu_result.exit_code);
// This demonstrates how you would compare results
// In real usage, you'd run your implementation and compare:
// let my_result = run_my_implementation(&args, input);
// assert_eq!(my_result.stdout, gnu_result.stdout);
// assert_eq!(my_result.exit_code, gnu_result.exit_code);
}
Err(error_result) => {
println!(
"⚠ GNU {} failed or not available: {}",
cmd, error_result.stderr
);
println!(" This is normal if GNU coreutils isn't installed");
}
}
println!();
}
println!("=== Practical Example: Compare two echo implementations ===");
// Simple echo comparison
let args = vec![OsString::from("hello"), OsString::from("world")];
match run_gnu_cmd("echo", &args, false, None) {
Ok(gnu_result) => {
println!("GNU echo result: {:?}", gnu_result.stdout.trim());
// Simulate our own echo implementation result
let our_result = CommandResult {
stdout: "hello world\n".to_string(),
stderr: String::new(),
exit_code: 0,
};
if our_result.stdout.trim() == gnu_result.stdout.trim()
&& our_result.exit_code == gnu_result.exit_code
{
println!("✓ Our echo matches GNU echo!");
} else {
println!("✗ Our echo differs from GNU echo");
println!(" Our result: {:?}", our_result.stdout.trim());
println!(" GNU result: {:?}", gnu_result.stdout.trim());
}
}
Err(_) => {
println!("Cannot compare - GNU echo not available");
}
}
println!("\n=== Example completed ===");
println!("This approach is simpler and more reliable for integration testing.");
}
-469
View File
@@ -1,469 +0,0 @@
// This file is part of the uutils coreutils package.
//
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
use console::Style;
use pretty_print::{
print_diff, print_end_with_status, print_or_empty, print_section, print_with_style,
};
use rand::RngExt;
use rand::prelude::IndexedRandom;
use rustix::io::dup;
use rustix::io::read;
use rustix::stdio::{dup2_stderr, dup2_stdin, dup2_stdout};
use std::env::temp_dir;
use std::ffi::OsString;
use std::fs::File;
use std::io::{Seek, SeekFrom, Write, pipe};
use std::process::{Command, Stdio};
use std::sync::atomic::Ordering;
use std::sync::{Once, atomic::AtomicBool};
use std::{io, thread};
pub mod pretty_print;
/// Represents the result of running a command, including its standard output,
/// standard error, and exit code.
#[derive(Debug)]
pub struct CommandResult {
/// The standard output (stdout) of the command as a string.
pub stdout: String,
/// The standard error (stderr) of the command as a string.
pub stderr: String,
/// The exit code of the command.
pub exit_code: i32,
}
static CHECK_GNU: Once = Once::new();
static IS_GNU: AtomicBool = AtomicBool::new(false);
pub fn is_gnu_cmd(cmd_path: &str) -> Result<(), std::io::Error> {
CHECK_GNU.call_once(|| {
let version_output = Command::new(cmd_path).arg("--version").output().unwrap();
println!("version_output {version_output:#?}");
let version_str = String::from_utf8_lossy(&version_output.stdout).to_string();
if version_str.contains("GNU coreutils") {
IS_GNU.store(true, Ordering::Relaxed);
}
});
if IS_GNU.load(Ordering::Relaxed) {
Ok(())
} else {
panic!("Not the GNU implementation");
}
}
pub fn generate_and_run_uumain<F>(
args: &[OsString],
uumain_function: F,
pipe_input: Option<&str>,
) -> CommandResult
where
F: FnOnce(std::vec::IntoIter<OsString>) -> i32 + Send + 'static,
{
// Duplicate the stdout and stderr file descriptors to restore later
let original_stdout_fd_owned =
dup(std::io::stdout()).expect("Failed to duplicate STDOUT_FILENO");
let original_stderr_fd_owned =
dup(std::io::stderr()).expect("Failed to duplicate STDERR_FILENO");
println!("Running test {:?}", &args[0..]);
let (read_pipe_stdout, write_pipe_stdout) = pipe().expect("Failed to create pipes");
let (read_pipe_stderr, write_pipe_stderr) = pipe().expect("Failed to create pipes");
// Redirect stdout and stderr to their respective pipes
dup2_stdout(&write_pipe_stdout).expect("Failed to redirect STDOUT_FILENO");
dup2_stderr(&write_pipe_stderr).expect("Failed to redirect STDERR_FILENO");
// Handle stdin redirection if needed
let original_stdin_fd_owned = if let Some(input_str) = pipe_input {
// we have pipe input
let mut input_file = tempfile::tempfile().unwrap();
write!(input_file, "{input_str}").unwrap();
input_file.seek(SeekFrom::Start(0)).unwrap();
// Redirect stdin to read from the in-memory file
let stdin_fd = dup(std::io::stdin()).expect("Failed to duplicate STDIN");
// Redirect stdin to read from the in-memory file
dup2_stdin(&input_file).expect("Failed to set up stdin redirection");
Some(stdin_fd)
} else {
None
};
let (uumain_exit_status, captured_stdout, captured_stderr) = thread::scope(|s| {
let out = s.spawn(|| read_from_fd(read_pipe_stdout));
let err = s.spawn(|| read_from_fd(read_pipe_stderr));
#[allow(clippy::unnecessary_to_owned)]
// TODO: clippy wants us to use args.iter().cloned() ?
let status = uumain_function(args.to_owned().into_iter());
// Reset the exit code global variable in case we run another test after this one
// See https://github.com/uutils/coreutils/issues/5777
uucore::error::set_exit_code(0);
io::stdout().flush().unwrap();
io::stderr().flush().unwrap();
// Drop write ends to close them, allowing readers to get EOF
drop(write_pipe_stdout);
drop(write_pipe_stderr);
// Restore stdout/stderr
let _ = dup2_stdout(&original_stdout_fd_owned);
let _ = dup2_stderr(&original_stderr_fd_owned);
(status, out.join().unwrap(), err.join().unwrap())
});
// Restore the original stdin if it was modified
if let Some(fd) = original_stdin_fd_owned {
dup2_stdin(&fd).expect("Failed to restore the original STDIN");
}
CommandResult {
stdout: captured_stdout,
stderr: captured_stderr
.split_once(':')
.map(|x| x.1)
.unwrap_or("")
.trim()
.to_string(),
exit_code: uumain_exit_status,
}
}
fn read_from_fd(fd: impl std::os::fd::AsFd) -> String {
let mut captured_output = Vec::new();
let mut read_buffer = [0; 1024];
loop {
match read(&fd, &mut read_buffer) {
Ok(bytes_read) => {
if bytes_read == 0 {
break;
}
captured_output.extend_from_slice(&read_buffer[..bytes_read]);
}
Err(_) => {
eprintln!("Failed to read from the pipe");
break;
}
}
}
String::from_utf8_lossy(&captured_output).into_owned()
}
pub fn run_gnu_cmd(
cmd_path: &str,
args: &[OsString],
check_gnu: bool,
pipe_input: Option<&str>,
) -> Result<CommandResult, CommandResult> {
// if the check passes, do nothing
if check_gnu && let Err(e) = is_gnu_cmd(cmd_path) {
// Convert the io::Error into the function's error type
return Err(CommandResult {
stdout: String::new(),
stderr: e.to_string(),
exit_code: -1,
});
}
let mut command = Command::new(cmd_path);
for arg in args {
command.arg(arg);
}
// See https://github.com/uutils/coreutils/issues/6794
// uutils' coreutils is not locale-aware, and aims to mirror/be compatible with GNU Core Utilities's LC_ALL=C behavior
command.env("LC_ALL", "C");
let output = if let Some(input_str) = pipe_input {
// We have an pipe input
command
.stdin(Stdio::piped())
.stdout(Stdio::piped())
.stderr(Stdio::piped());
let mut child = command.spawn().expect("Failed to execute command");
let child_stdin = child.stdin.as_mut().unwrap();
child_stdin
.write_all(input_str.as_bytes())
.expect("Failed to write to stdin");
match child.wait_with_output() {
Ok(output) => output,
Err(e) => {
return Err(CommandResult {
stdout: String::new(),
stderr: e.to_string(),
exit_code: -1,
});
}
}
} else {
// Just run with args
match command.output() {
Ok(output) => output,
Err(e) => {
return Err(CommandResult {
stdout: String::new(),
stderr: e.to_string(),
exit_code: -1,
});
}
}
};
let exit_code = output.status.code().unwrap_or(-1);
// Here we get stdout and stderr as Strings
let stdout = String::from_utf8_lossy(&output.stdout).to_string();
let stderr = String::from_utf8_lossy(&output.stderr).to_string();
let stderr = stderr
.split_once(':')
.map(|x| x.1)
.unwrap_or("")
.trim()
.to_string();
if output.status.success() || !check_gnu {
Ok(CommandResult {
stdout,
stderr,
exit_code,
})
} else {
Err(CommandResult {
stdout,
stderr,
exit_code,
})
}
}
/// Compare results from two different implementations of a command.
///
/// # Arguments
/// * `test_type` - The command.
/// * `input` - The input provided to the command.
/// * `rust_result` - The result of running the command with the Rust implementation.
/// * `gnu_result` - The result of running the command with the GNU implementation.
/// * `fail_on_stderr_diff` - Whether to fail the test if there is a difference in stderr output.
pub fn compare_result(
test_type: &str,
input: &str,
pipe_input: Option<&str>,
rust_result: &CommandResult,
gnu_result: &CommandResult,
fail_on_stderr_diff: bool,
) {
print_section(format!("Compare result for: {test_type} {input}"));
if let Some(pipe) = pipe_input {
println!("Pipe: {pipe}");
}
let mut discrepancies = Vec::new();
let mut should_panic = false;
if rust_result.stdout.trim() != gnu_result.stdout.trim() {
discrepancies.push("stdout differs");
println!("Rust stdout:");
print_or_empty(rust_result.stdout.as_str());
println!("GNU stdout:");
print_or_empty(gnu_result.stdout.as_ref());
print_diff(&rust_result.stdout, &gnu_result.stdout);
should_panic = true;
}
if rust_result.stderr.trim() != gnu_result.stderr.trim() {
discrepancies.push("stderr differs");
println!("Rust stderr:");
print_or_empty(rust_result.stderr.as_str());
println!("GNU stderr:");
print_or_empty(gnu_result.stderr.as_str());
print_diff(&rust_result.stderr, &gnu_result.stderr);
if fail_on_stderr_diff {
should_panic = true;
}
}
if rust_result.exit_code != gnu_result.exit_code {
discrepancies.push("exit code differs");
println!(
"Different exit code: (Rust: {}, GNU: {})",
rust_result.exit_code, gnu_result.exit_code
);
should_panic = true;
}
if discrepancies.is_empty() {
print_end_with_status("Same behavior", true);
} else {
print_with_style(
format!("Discrepancies detected: {}", discrepancies.join(", ")),
Style::new().red(),
);
if should_panic {
print_end_with_status(
format!("Test failed and will panic for: {test_type} {input}"),
false,
);
panic!("Test failed for: {test_type} {input}");
} else {
print_end_with_status(
format!("Test completed with discrepancies for: {test_type} {input}"),
false,
);
}
}
println!();
}
pub fn generate_random_string(max_length: usize) -> String {
let mut rng = rand::rng();
let valid_utf8: Vec<char> =
"ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789🔩🪛🪓⚙️🔗🧰"
.chars()
.collect();
let invalid_utf8 = [0xC3, 0x28]; // Invalid UTF-8 sequence
let mut result = String::new();
for _ in 0..rng.random_range(0..=max_length) {
if rng.random_bool(0.9) {
let ch = valid_utf8.choose(&mut rng).unwrap();
result.push(*ch);
} else {
let ch = invalid_utf8.choose(&mut rng).unwrap();
if let Some(c) = char::from_u32(*ch as u32) {
result.push(c);
}
}
}
result
}
#[allow(dead_code)]
pub fn generate_random_file() -> Result<String, std::io::Error> {
let mut rng = rand::rng();
let file_name: String = (0..10)
.map(|_| rng.random_range(b'a'..=b'z') as char)
.collect();
let mut file_path = temp_dir();
file_path.push(file_name);
let mut file = File::create(&file_path)?;
let content_length = rng.random_range(10..1000);
let content: String = (0..content_length)
.map(|_| rng.random_range(b' '..=b'~') as char)
.collect();
file.write_all(content.as_bytes())?;
Ok(file_path.to_str().unwrap().to_string())
}
#[allow(dead_code)]
pub fn replace_fuzz_binary_name(cmd: &str, result: &mut CommandResult) {
let fuzz_bin_name = format!("fuzz/target/x86_64-unknown-linux-gnu/release/fuzz_{cmd}");
result.stdout = result.stdout.replace(&fuzz_bin_name, cmd);
result.stderr = result.stderr.replace(&fuzz_bin_name, cmd);
}
#[cfg(test)]
mod tests {
use super::*;
use std::ffi::OsString;
#[test]
fn test_command_result_creation() {
let result = CommandResult {
stdout: "Hello, world!".to_string(),
stderr: "".to_string(),
exit_code: 0,
};
assert_eq!(result.stdout, "Hello, world!");
assert_eq!(result.stderr, "");
assert_eq!(result.exit_code, 0);
}
#[test]
fn test_generate_random_string() {
let result = generate_random_string(10);
// Check character count, not byte count (emojis are multi-byte)
assert!(result.chars().count() <= 10);
// Test that empty string can be generated (max_length = 0)
let empty_result = generate_random_string(0);
assert_eq!(empty_result.chars().count(), 0);
}
#[test]
fn test_replace_fuzz_binary_name() {
let mut result = CommandResult {
stdout: "fuzz/target/x86_64-unknown-linux-gnu/release/fuzz_echo: error".to_string(),
stderr: "fuzz/target/x86_64-unknown-linux-gnu/release/fuzz_echo failed".to_string(),
exit_code: 1,
};
replace_fuzz_binary_name("echo", &mut result);
assert_eq!(result.stdout, "echo: error");
assert_eq!(result.stderr, "echo failed");
assert_eq!(result.exit_code, 1);
}
#[test]
fn test_run_gnu_cmd_nonexistent() {
let args = vec![OsString::from("--version")];
let result = run_gnu_cmd("nonexistent_command_12345", &args, false, None);
// Should return an error since the command doesn't exist
assert!(result.is_err());
let error_result = result.unwrap_err();
assert_ne!(error_result.exit_code, 0);
}
#[test]
fn test_run_gnu_cmd_basic() {
// Test with a simple command that should exist on most systems
let args = vec![OsString::from("--version")];
let result = run_gnu_cmd("echo", &args, false, None);
// Should succeed (echo --version might not be standard but echo should exist)
if let Err(e) = result {
// Command failed but at least ran
assert_ne!(e.exit_code, -1); // -1 would indicate the command couldn't be found
}
}
#[test]
fn test_run_gnu_cmd_with_pipe_input() {
let args: Vec<OsString> = vec![];
let pipe_input = "hello world";
let result = run_gnu_cmd("cat", &args, false, Some(pipe_input));
// cat might not be available in test environment, that's ok
if let Ok(cmd_result) = result {
assert_eq!(cmd_result.stdout.trim(), "hello world");
}
}
#[test]
fn test_generate_random_file() {
let result = generate_random_file();
// File creation might fail due to permissions, that's acceptable for this test
if let Ok(path) = result {
assert!(!path.is_empty());
// Clean up - try to remove the file
let _ = std::fs::remove_file(&path);
}
}
}

Some files were not shown because too many files have changed in this diff Show More