mirror of
https://github.com/uutils/grep.git
synced 2026-06-10 16:15:11 -07:00
Compare commits
12
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
b0700b1d78 | ||
|
|
bc416c6d8b | ||
|
|
c614a57a05 | ||
|
|
6e6db248f1 | ||
|
|
b5816820ed | ||
|
|
ede1676d1a | ||
|
|
b46a86d48a | ||
|
|
b0164440e3 | ||
|
|
96762f26ca | ||
|
|
c8dfef6563 | ||
|
|
ddac723054 | ||
|
|
079619ee44 |
@@ -9,6 +9,9 @@ on:
|
|||||||
branches:
|
branches:
|
||||||
- '*'
|
- '*'
|
||||||
|
|
||||||
|
permissions:
|
||||||
|
contents: write # Publish grep instead of discarding
|
||||||
|
|
||||||
# End the current execution if there is a new changeset in the PR.
|
# End the current execution if there is a new changeset in the PR.
|
||||||
concurrency:
|
concurrency:
|
||||||
group: ${{ github.workflow }}-${{ github.ref }}
|
group: ${{ github.workflow }}-${{ github.ref }}
|
||||||
@@ -47,7 +50,21 @@ jobs:
|
|||||||
shell: bash
|
shell: bash
|
||||||
run: |
|
run: |
|
||||||
cd 'grep'
|
cd 'grep'
|
||||||
cargo build --release
|
cargo build --release --config=profile.release.strip=true
|
||||||
|
tar -C target/release -cf - grep | zstd -19 -o ../grep-x86_64-unknown-linux-gnu.tar.zst
|
||||||
|
- name: Publish latest commit
|
||||||
|
uses: softprops/action-gh-release@v3
|
||||||
|
if: github.event_name == 'push' && github.ref == 'refs/heads/main'
|
||||||
|
with:
|
||||||
|
tag_name: latest-commit
|
||||||
|
body: |
|
||||||
|
commit: ${{ github.sha }}
|
||||||
|
draft: false
|
||||||
|
prerelease: true
|
||||||
|
files: |
|
||||||
|
grep-x86_64-unknown-linux-gnu.tar.zst
|
||||||
|
env:
|
||||||
|
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||||
|
|
||||||
- name: Run GNU grep testsuite
|
- name: Run GNU grep testsuite
|
||||||
shell: bash
|
shell: bash
|
||||||
|
|||||||
@@ -1,50 +0,0 @@
|
|||||||
name: Benchmarks
|
|
||||||
|
|
||||||
# spell-checker:ignore codspeed dtolnay Swatinem sccache
|
|
||||||
|
|
||||||
on:
|
|
||||||
push:
|
|
||||||
branches: [ main, master ]
|
|
||||||
pull_request:
|
|
||||||
branches: [ main, master ]
|
|
||||||
|
|
||||||
permissions:
|
|
||||||
contents: read # to fetch code (actions/checkout)
|
|
||||||
|
|
||||||
# End the current execution if there is a new changeset in the PR.
|
|
||||||
concurrency:
|
|
||||||
group: ${{ github.workflow }}-${{ github.ref }}
|
|
||||||
cancel-in-progress: ${{ github.ref != 'refs/heads/main' }}
|
|
||||||
|
|
||||||
jobs:
|
|
||||||
benchmarks:
|
|
||||||
name: Run benchmarks (CodSpeed)
|
|
||||||
runs-on: ubuntu-latest
|
|
||||||
steps:
|
|
||||||
- uses: actions/checkout@v4
|
|
||||||
with:
|
|
||||||
persist-credentials: false
|
|
||||||
|
|
||||||
- name: Install Rust
|
|
||||||
uses: dtolnay/rust-toolchain@stable
|
|
||||||
- uses: Swatinem/rust-cache@v2
|
|
||||||
|
|
||||||
- name: Run sccache-cache
|
|
||||||
uses: mozilla-actions/sccache-action@v0.0.10
|
|
||||||
|
|
||||||
- name: Install cargo-codspeed
|
|
||||||
uses: taiki-e/install-action@v2
|
|
||||||
with:
|
|
||||||
tool: cargo-codspeed
|
|
||||||
|
|
||||||
- name: Build benchmarks
|
|
||||||
run: cargo codspeed build -p uu_grep
|
|
||||||
|
|
||||||
- name: Run benchmarks
|
|
||||||
uses: CodSpeedHQ/action@v4
|
|
||||||
env:
|
|
||||||
CODSPEED_LOG: debug
|
|
||||||
with:
|
|
||||||
mode: simulation
|
|
||||||
run: cargo codspeed run -p uu_grep > /dev/null
|
|
||||||
token: ${{ secrets.CODSPEED_TOKEN }}
|
|
||||||
@@ -0,0 +1,37 @@
|
|||||||
|
name: CodSpeed
|
||||||
|
|
||||||
|
on:
|
||||||
|
push:
|
||||||
|
branches:
|
||||||
|
- "main"
|
||||||
|
pull_request:
|
||||||
|
# `workflow_dispatch` allows CodSpeed to trigger backtest
|
||||||
|
# performance analysis in order to generate initial data.
|
||||||
|
workflow_dispatch:
|
||||||
|
|
||||||
|
permissions:
|
||||||
|
contents: read
|
||||||
|
id-token: write
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
codspeed:
|
||||||
|
name: Run benchmarks
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Setup Rust toolchain, cache and cargo-codspeed binary
|
||||||
|
uses: moonrepo/setup-rust@v0
|
||||||
|
with:
|
||||||
|
channel: stable
|
||||||
|
cache-target: release
|
||||||
|
bins: cargo-codspeed
|
||||||
|
|
||||||
|
- name: Build the benchmark target(s)
|
||||||
|
run: cargo codspeed build
|
||||||
|
|
||||||
|
- name: Run the benchmarks
|
||||||
|
uses: CodSpeedHQ/action@v4
|
||||||
|
with:
|
||||||
|
mode: simulation
|
||||||
|
run: cargo codspeed run
|
||||||
@@ -0,0 +1,55 @@
|
|||||||
|
# See https://pre-commit.com for more information
|
||||||
|
# See https://pre-commit.com/hooks.html for more hooks
|
||||||
|
exclude: ^tests/fixtures/
|
||||||
|
repos:
|
||||||
|
- repo: https://github.com/pre-commit/pre-commit-hooks
|
||||||
|
rev: v6.0.0
|
||||||
|
hooks:
|
||||||
|
- id: check-added-large-files
|
||||||
|
- id: check-executables-have-shebangs
|
||||||
|
- id: check-json
|
||||||
|
exclude: '\.vscode/(cSpell|extensions)\.json' # cSpell.json and extensions.json use comments
|
||||||
|
- id: check-shebang-scripts-are-executable
|
||||||
|
exclude: '.+\.rs' # would be triggered by #![some_attribute]
|
||||||
|
- id: check-symlinks
|
||||||
|
- id: check-toml
|
||||||
|
- id: check-yaml
|
||||||
|
args: [ --allow-multiple-documents ]
|
||||||
|
- id: destroyed-symlinks
|
||||||
|
- id: end-of-file-fixer
|
||||||
|
- id: mixed-line-ending
|
||||||
|
args: [ --fix=lf ]
|
||||||
|
- id: trailing-whitespace
|
||||||
|
|
||||||
|
- repo: local
|
||||||
|
hooks:
|
||||||
|
- id: rust-linting
|
||||||
|
name: Rust linting
|
||||||
|
description: Run cargo fmt on files included in the commit.
|
||||||
|
entry: cargo +stable fmt --
|
||||||
|
pass_filenames: true
|
||||||
|
types: [file, rust]
|
||||||
|
language: system
|
||||||
|
- id: rust-clippy
|
||||||
|
name: Rust clippy
|
||||||
|
description: Run cargo clippy on files included in the commit.
|
||||||
|
entry: cargo +stable clippy --workspace --all-targets --all-features -- -D warnings
|
||||||
|
pass_filenames: false
|
||||||
|
types: [file, rust]
|
||||||
|
language: system
|
||||||
|
- id: cargo-lock-check
|
||||||
|
name: Cargo.lock sync check
|
||||||
|
description: Ensure Cargo.lock and fuzz/Cargo.lock are up-to-date.
|
||||||
|
entry: bash -c 'for dir in . fuzz; do if [ -d "$dir" ]; then ( cd "$dir" && cargo fetch --quiet ); fi; done'
|
||||||
|
pass_filenames: false
|
||||||
|
files: 'Cargo\.(toml|lock)$'
|
||||||
|
language: system
|
||||||
|
- id: cspell
|
||||||
|
name: Code spell checker (cspell)
|
||||||
|
description: Run cspell to check for spelling errors (if available).
|
||||||
|
entry: bash -c 'if command -v cspell >/dev/null 2>&1; then cspell --no-must-find-files -- "$@"; else echo "cspell not found, skipping spell check"; exit 0; fi' --
|
||||||
|
pass_filenames: true
|
||||||
|
language: system
|
||||||
|
|
||||||
|
ci:
|
||||||
|
skip: [rust-linting, rust-clippy, cargo-lock-check, cspell]
|
||||||
Generated
+328
-98
File diff suppressed because it is too large
Load Diff
+4
-6
@@ -27,12 +27,10 @@ onig_sys = { version = "*", default-features = false }
|
|||||||
uucore = "0.8.0"
|
uucore = "0.8.0"
|
||||||
walkdir = "2.5"
|
walkdir = "2.5"
|
||||||
|
|
||||||
[dev-dependencies]
|
|
||||||
divan = { package = "codspeed-divan-compat", version = "4.0.5" }
|
|
||||||
tempfile = "3.10.1"
|
|
||||||
uucore = { version = "0.8.0", features = ["benchmark"] }
|
|
||||||
uutests = "0.8.0"
|
|
||||||
|
|
||||||
[[bench]]
|
[[bench]]
|
||||||
name = "grep_bench"
|
name = "grep_bench"
|
||||||
harness = false
|
harness = false
|
||||||
|
|
||||||
|
[dev-dependencies]
|
||||||
|
criterion = { version = "4.7.0", package = "codspeed-criterion-compat" }
|
||||||
|
uutests = "0.8.0"
|
||||||
|
|||||||
@@ -4,6 +4,7 @@
|
|||||||
[](https://deps.rs/repo/github/uutils/grep)
|
[](https://deps.rs/repo/github/uutils/grep)
|
||||||
|
|
||||||
[](https://codecov.io/gh/uutils/grep)
|
[](https://codecov.io/gh/uutils/grep)
|
||||||
|
[](https://codspeed.io/uutils/grep?utm_source=badge)
|
||||||
|
|
||||||
# Grep, now in Rust
|
# Grep, now in Rust
|
||||||
|
|
||||||
@@ -29,6 +30,10 @@ cargo build --release
|
|||||||
cargo test
|
cargo test
|
||||||
```
|
```
|
||||||
|
|
||||||
|
## Pre-commit hooks
|
||||||
|
|
||||||
|
This project uses [pre-commit](https://pre-commit.com); run `pre-commit install` to enable the git hooks.
|
||||||
|
|
||||||
## Known Issues
|
## Known Issues
|
||||||
|
|
||||||
* Does not take `LANG`, etc., into account for handling file encodings (non-UTF8 matches are treated as binary)
|
* Does not take `LANG`, etc., into account for handling file encodings (non-UTF8 matches are treated as binary)
|
||||||
|
|||||||
+104
-231
@@ -1,255 +1,128 @@
|
|||||||
// Benchmarks for the grep utility
|
|
||||||
//
|
|
||||||
// SPDX-License-Identifier: MIT
|
|
||||||
//
|
|
||||||
// This file is part of the uutils grep package.
|
// This file is part of the uutils grep package.
|
||||||
// It is licensed under the MIT License.
|
//
|
||||||
// For the full copyright and license information, please view the LICENSE
|
// For the full copyright and license information, please view the LICENSE
|
||||||
// file that was distributed with this source code.
|
// file that was distributed with this source code.
|
||||||
|
|
||||||
use divan::{Bencher, black_box};
|
use criterion::{Criterion, black_box, criterion_group, criterion_main};
|
||||||
use uu_grep::uumain;
|
use std::ffi::OsString;
|
||||||
use uucore::benchmark::{create_test_file, run_util_function};
|
use std::path::Path;
|
||||||
|
|
||||||
/// Build an access-log-like data set with `n` lines.
|
/// Run grep end-to-end through the real `uumain` entry point. `args` are the
|
||||||
///
|
/// arguments after the program name (flags, pattern, paths). The exit status is
|
||||||
/// Roughly a quarter of the lines use a non-default HTTP method / status /
|
/// ignored — we only care about the work performed.
|
||||||
/// user-agent so that selective patterns match a realistic subset rather than
|
fn run(args: &[&str]) {
|
||||||
/// every line or no line at all.
|
let mut argv: Vec<OsString> = Vec::with_capacity(args.len() + 1);
|
||||||
fn access_log(n: usize) -> Vec<u8> {
|
argv.push(OsString::from("grep"));
|
||||||
let mut data = Vec::new();
|
argv.extend(args.iter().map(OsString::from));
|
||||||
for i in 0..n {
|
let _ = uu_grep::uumain(argv.into_iter());
|
||||||
let method = if i % 4 == 0 { "POST" } else { "GET" };
|
}
|
||||||
let status = if i % 7 == 0 { 404 } else { 200 };
|
|
||||||
let agent = if i % 3 == 0 {
|
/// Build a multi-megabyte log-like corpus plus a directory holding it alongside
|
||||||
"Mozilla/5.0 (X11; Linux x86_64) Chrome/120.0"
|
/// a binary file. Every line contains `worker-<n>` and a `2024-…` timestamp; a
|
||||||
|
/// rare `RAREHIT` marker appears on a handful of lines (≈ every 10000th).
|
||||||
|
/// Returns `(dir, log_file)`.
|
||||||
|
fn build_corpus() -> (std::path::PathBuf, std::path::PathBuf) {
|
||||||
|
let mut content = String::new();
|
||||||
|
for i in 0..80_000u32 {
|
||||||
|
if i % 10_000 == 0 {
|
||||||
|
content.push_str(&format!(
|
||||||
|
"2024-01-15 10:30:{:02} RAREHIT worker-{i} special marker seen\n",
|
||||||
|
i % 60
|
||||||
|
));
|
||||||
|
} else if i % 100 == 0 {
|
||||||
|
content.push_str(&format!(
|
||||||
|
"2024-01-15 10:30:{:02} ERROR worker-{i} connection reset\n",
|
||||||
|
i % 60
|
||||||
|
));
|
||||||
} else {
|
} else {
|
||||||
"curl/8.5.0"
|
content.push_str(&format!(
|
||||||
};
|
"2024-01-15 10:30:{:02} INFO worker-{i} request handled in {}ms\n",
|
||||||
let line = format!(
|
i % 60,
|
||||||
"192.168.{}.{} - - [01/Jan/2024:00:00:00 +0000] \"{} /index.html HTTP/1.1\" {} 1234 \"-\" \"{}\"\n",
|
i % 1000
|
||||||
(i / 256) % 256,
|
|
||||||
i % 256,
|
|
||||||
method,
|
|
||||||
status,
|
|
||||||
agent,
|
|
||||||
);
|
|
||||||
data.extend_from_slice(line.as_bytes());
|
|
||||||
}
|
|
||||||
data
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Benchmark a literal search that matches nothing.
|
|
||||||
///
|
|
||||||
/// This is the purest measure of raw scan throughput: the whole file is read
|
|
||||||
/// and searched but no output is produced.
|
|
||||||
#[divan::bench]
|
|
||||||
fn literal_no_match(bencher: Bencher) {
|
|
||||||
let temp_dir = tempfile::tempdir().unwrap();
|
|
||||||
let file_path = create_test_file(&access_log(2_000_000), temp_dir.path());
|
|
||||||
let file_path_str = file_path.to_str().unwrap();
|
|
||||||
|
|
||||||
bencher.bench(|| {
|
|
||||||
black_box(run_util_function(
|
|
||||||
uumain,
|
|
||||||
&["ZZZ_NONEXISTENT_PATTERN_ZZZ", file_path_str],
|
|
||||||
));
|
));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
assert!(content.len() > 4 * 1024 * 1024);
|
||||||
|
|
||||||
|
let dir = std::env::temp_dir().join(format!("uu_grep_bench_{}", std::process::id()));
|
||||||
|
std::fs::create_dir_all(&dir).unwrap();
|
||||||
|
let log = dir.join("app.log");
|
||||||
|
std::fs::write(&log, &content).unwrap();
|
||||||
|
|
||||||
|
// A binary file (contains NUL) that also holds the marker, so `-I` has
|
||||||
|
// something to skip while recursing.
|
||||||
|
let mut binary = vec![0u8, 1, 2, 3];
|
||||||
|
binary.extend_from_slice(b"RAREHIT in binary blob");
|
||||||
|
binary.extend(std::iter::repeat_n(0u8, 4096));
|
||||||
|
std::fs::write(dir.join("data.bin"), &binary).unwrap();
|
||||||
|
|
||||||
|
(dir, log)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn bench_e2e(c: &mut Criterion) {
|
||||||
|
let (dir, log) = build_corpus();
|
||||||
|
let file = log.to_str().unwrap();
|
||||||
|
let dir_str = dir.to_str().unwrap();
|
||||||
|
|
||||||
|
// Pure scanning throughput: `-q` with a pattern that never matches forces a
|
||||||
|
// full scan and produces no output. A literal (which a buffer-at-a-time
|
||||||
|
// searcher can accelerate) versus an extended-regex control (which cannot).
|
||||||
|
{
|
||||||
|
let mut group = c.benchmark_group("scan");
|
||||||
|
group.bench_function("literal_no_match", |b| {
|
||||||
|
b.iter(|| run(black_box(&["-q", "NONEXISTENT_TOKEN_XYZ", file])))
|
||||||
});
|
});
|
||||||
}
|
group.bench_function("regex_no_match", |b| {
|
||||||
|
b.iter(|| run(black_box(&["-q", "-E", "NON[0-9]EXISTENT_TOKEN", file])))
|
||||||
/// Benchmark a literal search that matches a subset of lines.
|
|
||||||
#[divan::bench]
|
|
||||||
fn literal_match_some(bencher: Bencher) {
|
|
||||||
let temp_dir = tempfile::tempdir().unwrap();
|
|
||||||
let file_path = create_test_file(&access_log(2_000_000), temp_dir.path());
|
|
||||||
let file_path_str = file_path.to_str().unwrap();
|
|
||||||
|
|
||||||
bencher.bench(|| {
|
|
||||||
black_box(run_util_function(uumain, &["POST", file_path_str]));
|
|
||||||
});
|
});
|
||||||
|
group.finish();
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Benchmark a literal search that matches every line (counting only).
|
// Real invocation shapes from the `grep` tldr page, each scanning the whole
|
||||||
///
|
// corpus. The `RAREHIT` marker matches only a handful of lines, so output
|
||||||
/// `-c` keeps the output bounded so the benchmark measures matching rather than
|
// stays small while the full-file scan dominates.
|
||||||
/// terminal I/O.
|
{
|
||||||
#[divan::bench]
|
let mut group = c.benchmark_group("usage");
|
||||||
fn literal_match_all_count(bencher: Bencher) {
|
|
||||||
let temp_dir = tempfile::tempdir().unwrap();
|
|
||||||
let file_path = create_test_file(&access_log(2_000_000), temp_dir.path());
|
|
||||||
let file_path_str = file_path.to_str().unwrap();
|
|
||||||
|
|
||||||
bencher.bench(|| {
|
// Search for a pattern within a file.
|
||||||
black_box(run_util_function(uumain, &["-c", "HTTP", file_path_str]));
|
group.bench_function("search_pattern", |b| {
|
||||||
|
b.iter(|| run(black_box(&["RAREHIT", file])))
|
||||||
});
|
});
|
||||||
}
|
// Search for an exact string (-F).
|
||||||
|
group.bench_function("fixed_string", |b| {
|
||||||
/// Benchmark a fixed-string search (`-F`).
|
b.iter(|| run(black_box(&["-F", "RAREHIT", file])))
|
||||||
#[divan::bench]
|
|
||||||
fn fixed_string(bencher: Bencher) {
|
|
||||||
let temp_dir = tempfile::tempdir().unwrap();
|
|
||||||
let file_path = create_test_file(&access_log(2_000_000), temp_dir.path());
|
|
||||||
let file_path_str = file_path.to_str().unwrap();
|
|
||||||
|
|
||||||
bencher.bench(|| {
|
|
||||||
black_box(run_util_function(
|
|
||||||
uumain,
|
|
||||||
&["-F", "Chrome/120.0", file_path_str],
|
|
||||||
));
|
|
||||||
});
|
});
|
||||||
}
|
// Recursive search ignoring binary files (-rI).
|
||||||
|
group.bench_function("recursive_no_binary", |b| {
|
||||||
/// Benchmark a case-insensitive search (`-i`).
|
b.iter(|| run(black_box(&["-rI", "RAREHIT", dir_str])))
|
||||||
#[divan::bench]
|
|
||||||
fn case_insensitive(bencher: Bencher) {
|
|
||||||
let temp_dir = tempfile::tempdir().unwrap();
|
|
||||||
let file_path = create_test_file(&access_log(2_000_000), temp_dir.path());
|
|
||||||
let file_path_str = file_path.to_str().unwrap();
|
|
||||||
|
|
||||||
bencher.bench(|| {
|
|
||||||
black_box(run_util_function(uumain, &["-i", "mozilla", file_path_str]));
|
|
||||||
});
|
});
|
||||||
}
|
// Print 3 lines of context (-C 3).
|
||||||
|
group.bench_function("context", |b| {
|
||||||
/// Benchmark counting matches (`-c`).
|
b.iter(|| run(black_box(&["-C", "3", "RAREHIT", file])))
|
||||||
#[divan::bench]
|
|
||||||
fn count(bencher: Bencher) {
|
|
||||||
let temp_dir = tempfile::tempdir().unwrap();
|
|
||||||
let file_path = create_test_file(&access_log(2_000_000), temp_dir.path());
|
|
||||||
let file_path_str = file_path.to_str().unwrap();
|
|
||||||
|
|
||||||
bencher.bench(|| {
|
|
||||||
black_box(run_util_function(uumain, &["-c", "POST", file_path_str]));
|
|
||||||
});
|
});
|
||||||
}
|
// Filename + line number with forced color (-Hn --color=always).
|
||||||
|
group.bench_function("filename_lineno_color", |b| {
|
||||||
/// Benchmark an inverted match (`-v`).
|
b.iter(|| run(black_box(&["-Hn", "--color=always", "RAREHIT", file])))
|
||||||
///
|
|
||||||
/// Most lines do not contain "POST", so this selects the majority of lines;
|
|
||||||
/// `-c` bounds the output.
|
|
||||||
#[divan::bench]
|
|
||||||
fn invert_match_count(bencher: Bencher) {
|
|
||||||
let temp_dir = tempfile::tempdir().unwrap();
|
|
||||||
let file_path = create_test_file(&access_log(2_000_000), temp_dir.path());
|
|
||||||
let file_path_str = file_path.to_str().unwrap();
|
|
||||||
|
|
||||||
bencher.bench(|| {
|
|
||||||
black_box(run_util_function(uumain, &["-vc", "POST", file_path_str]));
|
|
||||||
});
|
});
|
||||||
}
|
// Print only the matched text (-o).
|
||||||
|
group.bench_function("only_matching", |b| {
|
||||||
/// Benchmark printing line numbers (`-n`).
|
b.iter(|| run(black_box(&["-o", "RAREHIT", file])))
|
||||||
#[divan::bench]
|
|
||||||
fn line_number(bencher: Bencher) {
|
|
||||||
let temp_dir = tempfile::tempdir().unwrap();
|
|
||||||
let file_path = create_test_file(&access_log(1_000_000), temp_dir.path());
|
|
||||||
let file_path_str = file_path.to_str().unwrap();
|
|
||||||
|
|
||||||
bencher.bench(|| {
|
|
||||||
black_box(run_util_function(uumain, &["-nc", "POST", file_path_str]));
|
|
||||||
});
|
});
|
||||||
}
|
// Invert match (-v); `worker-` is on every line, so nothing is printed
|
||||||
|
// and this measures the full inverted scan.
|
||||||
/// Benchmark word-boundary matching (`-w`).
|
group.bench_function("invert_match", |b| {
|
||||||
#[divan::bench]
|
b.iter(|| run(black_box(&["-v", "worker-", file])))
|
||||||
fn word_match(bencher: Bencher) {
|
|
||||||
let temp_dir = tempfile::tempdir().unwrap();
|
|
||||||
let file_path = create_test_file(&access_log(2_000_000), temp_dir.path());
|
|
||||||
let file_path_str = file_path.to_str().unwrap();
|
|
||||||
|
|
||||||
bencher.bench(|| {
|
|
||||||
black_box(run_util_function(uumain, &["-wc", "GET", file_path_str]));
|
|
||||||
});
|
});
|
||||||
}
|
// Extended regex, case-insensitive (-Ei).
|
||||||
|
group.bench_function("extended_icase", |b| {
|
||||||
/// Benchmark an extended regular expression with alternation (`-E`).
|
b.iter(|| run(black_box(&["-Ei", "rarehit", file])))
|
||||||
#[divan::bench]
|
|
||||||
fn extended_regex(bencher: Bencher) {
|
|
||||||
let temp_dir = tempfile::tempdir().unwrap();
|
|
||||||
let file_path = create_test_file(&access_log(2_000_000), temp_dir.path());
|
|
||||||
let file_path_str = file_path.to_str().unwrap();
|
|
||||||
|
|
||||||
bencher.bench(|| {
|
|
||||||
black_box(run_util_function(
|
|
||||||
uumain,
|
|
||||||
&["-Ec", "(POST|DELETE|PUT)", file_path_str],
|
|
||||||
));
|
|
||||||
});
|
});
|
||||||
|
|
||||||
|
group.finish();
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Benchmark a basic regular expression with an anchor and character class.
|
let _ = std::fs::remove_dir_all(Path::new(dir_str));
|
||||||
#[divan::bench]
|
|
||||||
fn basic_regex(bencher: Bencher) {
|
|
||||||
let temp_dir = tempfile::tempdir().unwrap();
|
|
||||||
let file_path = create_test_file(&access_log(2_000_000), temp_dir.path());
|
|
||||||
let file_path_str = file_path.to_str().unwrap();
|
|
||||||
|
|
||||||
bencher.bench(|| {
|
|
||||||
black_box(run_util_function(
|
|
||||||
uumain,
|
|
||||||
&["-c", "^192\\.168\\.[0-9]*\\.0 ", file_path_str],
|
|
||||||
));
|
|
||||||
});
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Benchmark a Perl-compatible regular expression (`-P`).
|
criterion_group!(benches, bench_e2e);
|
||||||
#[divan::bench]
|
criterion_main!(benches);
|
||||||
fn perl_regex(bencher: Bencher) {
|
|
||||||
let temp_dir = tempfile::tempdir().unwrap();
|
|
||||||
let file_path = create_test_file(&access_log(2_000_000), temp_dir.path());
|
|
||||||
let file_path_str = file_path.to_str().unwrap();
|
|
||||||
|
|
||||||
bencher.bench(|| {
|
|
||||||
black_box(run_util_function(
|
|
||||||
uumain,
|
|
||||||
&["-Pc", "\"\\d{3}\" \\d+", file_path_str],
|
|
||||||
));
|
|
||||||
});
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Benchmark `--only-matching` (`-o`) extracting a substring from each line.
|
|
||||||
#[divan::bench]
|
|
||||||
fn only_matching(bencher: Bencher) {
|
|
||||||
let temp_dir = tempfile::tempdir().unwrap();
|
|
||||||
let file_path = create_test_file(&access_log(1_000_000), temp_dir.path());
|
|
||||||
let file_path_str = file_path.to_str().unwrap();
|
|
||||||
|
|
||||||
bencher.bench(|| {
|
|
||||||
black_box(run_util_function(
|
|
||||||
uumain,
|
|
||||||
&["-Eoc", "[0-9]+\\.[0-9]+\\.[0-9]+\\.[0-9]+", file_path_str],
|
|
||||||
));
|
|
||||||
});
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Benchmark quiet mode (`-q`), which can stop at the first match.
|
|
||||||
#[divan::bench]
|
|
||||||
fn quiet_first_match(bencher: Bencher) {
|
|
||||||
let temp_dir = tempfile::tempdir().unwrap();
|
|
||||||
let file_path = create_test_file(&access_log(2_000_000), temp_dir.path());
|
|
||||||
let file_path_str = file_path.to_str().unwrap();
|
|
||||||
|
|
||||||
bencher.bench(|| {
|
|
||||||
black_box(run_util_function(uumain, &["-q", "POST", file_path_str]));
|
|
||||||
});
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Benchmark searching short numeric lines (many small lines).
|
|
||||||
#[divan::bench]
|
|
||||||
fn short_lines(bencher: Bencher) {
|
|
||||||
let temp_dir = tempfile::tempdir().unwrap();
|
|
||||||
let mut data = Vec::new();
|
|
||||||
for i in 0..10_000_000 {
|
|
||||||
data.extend_from_slice(format!("{i}\n").as_bytes());
|
|
||||||
}
|
|
||||||
let file_path = create_test_file(&data, temp_dir.path());
|
|
||||||
let file_path_str = file_path.to_str().unwrap();
|
|
||||||
|
|
||||||
bencher.bench(|| {
|
|
||||||
black_box(run_util_function(uumain, &["-c", "999", file_path_str]));
|
|
||||||
});
|
|
||||||
}
|
|
||||||
|
|
||||||
fn main() {
|
|
||||||
divan::main();
|
|
||||||
}
|
|
||||||
|
|||||||
+79
-56
@@ -3,9 +3,12 @@
|
|||||||
// For the full copyright and license information, please view the LICENSE
|
// For the full copyright and license information, please view the LICENSE
|
||||||
// file that was distributed with this source code.
|
// file that was distributed with this source code.
|
||||||
|
|
||||||
mod context_buffer;
|
#[doc(hidden)]
|
||||||
mod line_buffer;
|
pub mod context_buffer;
|
||||||
mod matcher;
|
#[doc(hidden)]
|
||||||
|
pub mod line_buffer;
|
||||||
|
#[doc(hidden)]
|
||||||
|
pub mod matcher;
|
||||||
mod output;
|
mod output;
|
||||||
mod searcher;
|
mod searcher;
|
||||||
|
|
||||||
@@ -20,7 +23,8 @@ use std::path::Path;
|
|||||||
use uucore::error::{FromIo, UResult, USimpleError};
|
use uucore::error::{FromIo, UResult, USimpleError};
|
||||||
|
|
||||||
#[derive(Clone, Copy, PartialEq, Eq)]
|
#[derive(Clone, Copy, PartialEq, Eq)]
|
||||||
enum RegexMode {
|
#[doc(hidden)]
|
||||||
|
pub enum RegexMode {
|
||||||
Fixed,
|
Fixed,
|
||||||
Basic,
|
Basic,
|
||||||
Extended,
|
Extended,
|
||||||
@@ -28,7 +32,8 @@ enum RegexMode {
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Clone, Copy, PartialEq, Eq)]
|
#[derive(Clone, Copy, PartialEq, Eq)]
|
||||||
enum BinaryMode {
|
#[doc(hidden)]
|
||||||
|
pub enum BinaryMode {
|
||||||
Binary,
|
Binary,
|
||||||
Text,
|
Text,
|
||||||
WithoutMatch,
|
WithoutMatch,
|
||||||
@@ -42,79 +47,84 @@ enum ColorMode {
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Clone, Copy, PartialEq, Eq)]
|
#[derive(Clone, Copy, PartialEq, Eq)]
|
||||||
enum DirectoryMode {
|
#[doc(hidden)]
|
||||||
|
pub enum DirectoryMode {
|
||||||
Read,
|
Read,
|
||||||
Skip,
|
Skip,
|
||||||
Recurse,
|
Recurse,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Clone, Copy, PartialEq, Eq)]
|
#[derive(Clone, Copy, PartialEq, Eq)]
|
||||||
enum DeviceMode {
|
#[doc(hidden)]
|
||||||
|
pub enum DeviceMode {
|
||||||
Default,
|
Default,
|
||||||
Read,
|
Read,
|
||||||
Skip,
|
Skip,
|
||||||
}
|
}
|
||||||
|
|
||||||
struct ColorConfig<'a> {
|
#[doc(hidden)]
|
||||||
matched_selected: &'a str,
|
pub struct ColorConfig<'a> {
|
||||||
matched_context: &'a str,
|
pub matched_selected: &'a str,
|
||||||
filename: &'a str,
|
pub matched_context: &'a str,
|
||||||
line_number: &'a str,
|
pub filename: &'a str,
|
||||||
byte_offset: &'a str,
|
pub line_number: &'a str,
|
||||||
separator: &'a str,
|
pub byte_offset: &'a str,
|
||||||
selected_line: &'a str,
|
pub separator: &'a str,
|
||||||
context_line: &'a str,
|
pub selected_line: &'a str,
|
||||||
|
pub context_line: &'a str,
|
||||||
|
|
||||||
reverse_video: bool,
|
pub reverse_video: bool,
|
||||||
no_erase: bool,
|
pub no_erase: bool,
|
||||||
}
|
}
|
||||||
|
|
||||||
struct GlobSet {
|
#[doc(hidden)]
|
||||||
|
pub struct GlobSet {
|
||||||
patterns: Vec<glob::Pattern>,
|
patterns: Vec<glob::Pattern>,
|
||||||
}
|
}
|
||||||
|
|
||||||
struct Config<'a> {
|
#[doc(hidden)]
|
||||||
|
pub struct Config<'a> {
|
||||||
// Searcher
|
// Searcher
|
||||||
directory_mode: DirectoryMode,
|
pub directory_mode: DirectoryMode,
|
||||||
device_mode: DeviceMode,
|
pub device_mode: DeviceMode,
|
||||||
follow_symlinks: bool,
|
pub follow_symlinks: bool,
|
||||||
include_globs: GlobSet,
|
pub include_globs: GlobSet,
|
||||||
exclude_globs: GlobSet,
|
pub exclude_globs: GlobSet,
|
||||||
exclude_dir_globs: GlobSet,
|
pub exclude_dir_globs: GlobSet,
|
||||||
label: &'a str,
|
pub label: &'a str,
|
||||||
#[cfg(windows)]
|
#[cfg(windows)]
|
||||||
strip_cr: bool,
|
pub strip_cr: bool,
|
||||||
binary_mode: BinaryMode,
|
pub binary_mode: BinaryMode,
|
||||||
max_count: Option<u64>,
|
pub max_count: Option<u64>,
|
||||||
before_context: usize,
|
pub before_context: usize,
|
||||||
after_context: usize,
|
pub after_context: usize,
|
||||||
has_context: bool,
|
pub has_context: bool,
|
||||||
|
|
||||||
// Matcher
|
// Matcher
|
||||||
patterns: &'a [&'a str],
|
pub patterns: &'a [&'a str],
|
||||||
regex_mode: RegexMode,
|
pub regex_mode: RegexMode,
|
||||||
ignore_case: bool,
|
pub ignore_case: bool,
|
||||||
invert_match: bool,
|
pub invert_match: bool,
|
||||||
word_regexp: bool,
|
pub word_regexp: bool,
|
||||||
line_regexp: bool,
|
pub line_regexp: bool,
|
||||||
|
|
||||||
// Output
|
// Output
|
||||||
quiet: bool,
|
pub quiet: bool,
|
||||||
count: bool,
|
pub count: bool,
|
||||||
show_filename: bool,
|
pub show_filename: bool,
|
||||||
files_with_matches: bool,
|
pub files_with_matches: bool,
|
||||||
files_without_match: bool,
|
pub files_without_match: bool,
|
||||||
only_matching: bool,
|
pub only_matching: bool,
|
||||||
byte_offset: bool,
|
pub byte_offset: bool,
|
||||||
line_number: bool,
|
pub line_number: bool,
|
||||||
initial_tab: bool,
|
pub initial_tab: bool,
|
||||||
null_separator: bool,
|
pub null_separator: bool,
|
||||||
null_data: bool,
|
pub null_data: bool,
|
||||||
line_buffered: bool,
|
pub line_buffered: bool,
|
||||||
no_messages: bool,
|
pub no_messages: bool,
|
||||||
group_separator: Option<&'a str>,
|
pub group_separator: Option<&'a str>,
|
||||||
use_color: bool,
|
pub use_color: bool,
|
||||||
color_config: ColorConfig<'a>,
|
pub color_config: ColorConfig<'a>,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[uucore::main(no_signals)]
|
#[uucore::main(no_signals)]
|
||||||
@@ -854,8 +864,21 @@ fn expand_num_shorthand(args: impl Iterator<Item = OsString>) -> Vec<OsString> {
|
|||||||
out
|
out
|
||||||
}
|
}
|
||||||
|
|
||||||
|
impl Default for GlobSet {
|
||||||
|
fn default() -> Self {
|
||||||
|
Self::new()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
impl GlobSet {
|
impl GlobSet {
|
||||||
fn with_capacity(capacity: usize) -> Self {
|
/// Create an empty GlobSet.
|
||||||
|
pub fn new() -> Self {
|
||||||
|
Self {
|
||||||
|
patterns: Vec::new(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn with_capacity(capacity: usize) -> Self {
|
||||||
Self {
|
Self {
|
||||||
patterns: Vec::with_capacity(capacity),
|
patterns: Vec::with_capacity(capacity),
|
||||||
}
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user