Author SHA1 Message Date
Sylvestre LedruandGitHub b0700b1d78 Merge pull request #21 from oech3/pub
GnuTests: publish binary from main
2026-06-02 21:04:50 +02:00
oech3 bc416c6d8b GnuTests: publish binary from main 2026-06-03 00:05:24 +09:00
Sylvestre LedruandGitHub c614a57a05 Merge pull request #17 from uutils/e2e-bench-only
E2e bench only
2026-05-31 11:17:38 +02:00
Sylvestre Ledru 6e6db248f1 bench: add e2e benchmarks for the grep tldr invocations
Cover the real-world grep usage shapes from the tldr page end-to-end
through uumain over a shared multi-MB corpus (plus a directory with a
binary file for -rI):

  search pattern, -F fixed string, -rI recursive ignoring binary,
  -C 3 context, -Hn --color=always, -o only-matching, -v invert,
  -Ei extended + ignore-case.

Kept alongside the pure-scan throughput benches (literal vs regex, no
match). A rare marker keeps matched output small so the full-file scan
dominates the timing.
2026-05-31 11:10:11 +02:00
Sylvestre Ledru b5816820ed bench: end-to-end search throughput only
Replace the matcher micro-benchmarks with a single end-to-end 'search'
group driven through uumain over a multi-megabyte file: a literal
pattern (which a buffer-at-a-time searcher can accelerate) and an
extended-regex control (which cannot). Matching pre-split lines in
isolation cannot reveal how the searcher feeds data to the matcher;
this does.
2026-05-31 11:05:28 +02:00
Sylvestre LedruandGitHub ede1676d1a Merge pull request #15 from uutils/bench-literal-throughput
bench: end-to-end search throughput via uumain
2026-05-31 10:55:17 +02:00
Sylvestre Ledru b46a86d48a bench: end-to-end search throughput via uumain
The existing match/throughput benches call Matcher::match_line on
pre-split lines, so they only measure matching in isolation and cannot
observe how the searcher feeds data to the matcher. Add a 'search'
group that drives the whole pipeline through uumain over a multi-MB
file: a literal pattern (which a buffer-at-a-time searcher can speed
up) and an extended-regex control (which it cannot). Uses -q with a
non-matching pattern for a silent full-file scan.
2026-05-31 10:41:52 +02:00
Sylvestre LedruandGitHub b0164440e3 Merge pull request #14 from uutils/fix
Add Default impl for GlobSet to satisfy clippy
2026-05-31 10:34:25 +02:00
Sylvestre Ledru 96762f26ca Add Default impl for GlobSet to satisfy clippy 2026-05-31 10:18:32 +02:00
Sylvestre Ledru c8dfef6563 Add pre-commit configuration 2026-05-31 10:11:57 +02:00
Sylvestre LedruandGitHub ddac723054 Merge pull request #13 from uutils/codspeed-wizard-1780213675759
Add CodSpeed performance benchmarks
2026-05-31 10:11:23 +02:00
codspeed-hq[bot]andGitHub 079619ee44 Add CodSpeed performance benchmarks 2026-05-31 07:59:22 +00:00
Sylvestre LedruandGitHub 2a6a3aba7e Add note about performance improvements needed 2026-05-30 19:25:27 +02:00
11 changed files with 939 additions and 498 deletions
+18 -1
View File
@@ -9,6 +9,9 @@ on:
branches: branches:
- '*' - '*'
permissions:
contents: write # Publish grep instead of discarding
# End the current execution if there is a new changeset in the PR. # End the current execution if there is a new changeset in the PR.
concurrency: concurrency:
group: ${{ github.workflow }}-${{ github.ref }} group: ${{ github.workflow }}-${{ github.ref }}
@@ -47,7 +50,21 @@ jobs:
shell: bash shell: bash
run: | run: |
cd 'grep' cd 'grep'
cargo build --release cargo build --release --config=profile.release.strip=true
tar -C target/release -cf - grep | zstd -19 -o ../grep-x86_64-unknown-linux-gnu.tar.zst
- name: Publish latest commit
uses: softprops/action-gh-release@v3
if: github.event_name == 'push' && github.ref == 'refs/heads/main'
with:
tag_name: latest-commit
body: |
commit: ${{ github.sha }}
draft: false
prerelease: true
files: |
grep-x86_64-unknown-linux-gnu.tar.zst
env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
- name: Run GNU grep testsuite - name: Run GNU grep testsuite
shell: bash shell: bash
+37
View File
@@ -0,0 +1,37 @@
name: CodSpeed
on:
push:
branches:
- "main"
pull_request:
# `workflow_dispatch` allows CodSpeed to trigger backtest
# performance analysis in order to generate initial data.
workflow_dispatch:
permissions:
contents: read
id-token: write
jobs:
codspeed:
name: Run benchmarks
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- name: Setup Rust toolchain, cache and cargo-codspeed binary
uses: moonrepo/setup-rust@v0
with:
channel: stable
cache-target: release
bins: cargo-codspeed
- name: Build the benchmark target(s)
run: cargo codspeed build
- name: Run the benchmarks
uses: CodSpeedHQ/action@v4
with:
mode: simulation
run: cargo codspeed run
+55
View File
@@ -0,0 +1,55 @@
# See https://pre-commit.com for more information
# See https://pre-commit.com/hooks.html for more hooks
exclude: ^tests/fixtures/
repos:
- repo: https://github.com/pre-commit/pre-commit-hooks
rev: v6.0.0
hooks:
- id: check-added-large-files
- id: check-executables-have-shebangs
- id: check-json
exclude: '\.vscode/(cSpell|extensions)\.json' # cSpell.json and extensions.json use comments
- id: check-shebang-scripts-are-executable
exclude: '.+\.rs' # would be triggered by #![some_attribute]
- id: check-symlinks
- id: check-toml
- id: check-yaml
args: [ --allow-multiple-documents ]
- id: destroyed-symlinks
- id: end-of-file-fixer
- id: mixed-line-ending
args: [ --fix=lf ]
- id: trailing-whitespace
- repo: local
hooks:
- id: rust-linting
name: Rust linting
description: Run cargo fmt on files included in the commit.
entry: cargo +stable fmt --
pass_filenames: true
types: [file, rust]
language: system
- id: rust-clippy
name: Rust clippy
description: Run cargo clippy on files included in the commit.
entry: cargo +stable clippy --workspace --all-targets --all-features -- -D warnings
pass_filenames: false
types: [file, rust]
language: system
- id: cargo-lock-check
name: Cargo.lock sync check
description: Ensure Cargo.lock and fuzz/Cargo.lock are up-to-date.
entry: bash -c 'for dir in . fuzz; do if [ -d "$dir" ]; then ( cd "$dir" && cargo fetch --quiet ); fi; done'
pass_filenames: false
files: 'Cargo\.(toml|lock)$'
language: system
- id: cspell
name: Code spell checker (cspell)
description: Run cspell to check for spelling errors (if available).
entry: bash -c 'if command -v cspell >/dev/null 2>&1; then cspell --no-must-find-files -- "$@"; else echo "cspell not found, skipping spell check"; exit 0; fi' --
pass_filenames: true
language: system
ci:
skip: [rust-linting, rust-clippy, cargo-lock-check, cspell]
Generated
+532 -11
View File
File diff suppressed because it is too large Load Diff
+5
View File
@@ -27,5 +27,10 @@ onig_sys = { version = "*", default-features = false }
uucore = "0.8.0" uucore = "0.8.0"
walkdir = "2.5" walkdir = "2.5"
[[bench]]
name = "grep_bench"
harness = false
[dev-dependencies] [dev-dependencies]
criterion = { version = "4.7.0", package = "codspeed-criterion-compat" }
uutests = "0.8.0" uutests = "0.8.0"
+6
View File
@@ -4,6 +4,7 @@
[![dependency status](https://deps.rs/repo/github/uutils/grep/status.svg)](https://deps.rs/repo/github/uutils/grep) [![dependency status](https://deps.rs/repo/github/uutils/grep/status.svg)](https://deps.rs/repo/github/uutils/grep)
[![CodeCov](https://codecov.io/gh/uutils/grep/branch/main/graph/badge.svg)](https://codecov.io/gh/uutils/grep) [![CodeCov](https://codecov.io/gh/uutils/grep/branch/main/graph/badge.svg)](https://codecov.io/gh/uutils/grep)
[![CodSpeed](https://img.shields.io/endpoint?url=https://codspeed.io/badge.json)](https://codspeed.io/uutils/grep?utm_source=badge)
# Grep, now in Rust # Grep, now in Rust
@@ -29,10 +30,15 @@ cargo build --release
cargo test cargo test
``` ```
## Pre-commit hooks
This project uses [pre-commit](https://pre-commit.com); run `pre-commit install` to enable the git hooks.
## Known Issues ## Known Issues
* Does not take `LANG`, etc., into account for handling file encodings (non-UTF8 matches are treated as binary) * Does not take `LANG`, etc., into account for handling file encodings (non-UTF8 matches are treated as binary)
* No localization support yet * No localization support yet
* Performances need to be improved
## Contributing ## Contributing
+128
View File
@@ -0,0 +1,128 @@
// This file is part of the uutils grep package.
//
// For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code.
use criterion::{Criterion, black_box, criterion_group, criterion_main};
use std::ffi::OsString;
use std::path::Path;
/// Run grep end-to-end through the real `uumain` entry point. `args` are the
/// arguments after the program name (flags, pattern, paths). The exit status is
/// ignored — we only care about the work performed.
fn run(args: &[&str]) {
let mut argv: Vec<OsString> = Vec::with_capacity(args.len() + 1);
argv.push(OsString::from("grep"));
argv.extend(args.iter().map(OsString::from));
let _ = uu_grep::uumain(argv.into_iter());
}
/// Build a multi-megabyte log-like corpus plus a directory holding it alongside
/// a binary file. Every line contains `worker-<n>` and a `2024-…` timestamp; a
/// rare `RAREHIT` marker appears on a handful of lines (≈ every 10000th).
/// Returns `(dir, log_file)`.
fn build_corpus() -> (std::path::PathBuf, std::path::PathBuf) {
let mut content = String::new();
for i in 0..80_000u32 {
if i % 10_000 == 0 {
content.push_str(&format!(
"2024-01-15 10:30:{:02} RAREHIT worker-{i} special marker seen\n",
i % 60
));
} else if i % 100 == 0 {
content.push_str(&format!(
"2024-01-15 10:30:{:02} ERROR worker-{i} connection reset\n",
i % 60
));
} else {
content.push_str(&format!(
"2024-01-15 10:30:{:02} INFO worker-{i} request handled in {}ms\n",
i % 60,
i % 1000
));
}
}
assert!(content.len() > 4 * 1024 * 1024);
let dir = std::env::temp_dir().join(format!("uu_grep_bench_{}", std::process::id()));
std::fs::create_dir_all(&dir).unwrap();
let log = dir.join("app.log");
std::fs::write(&log, &content).unwrap();
// A binary file (contains NUL) that also holds the marker, so `-I` has
// something to skip while recursing.
let mut binary = vec![0u8, 1, 2, 3];
binary.extend_from_slice(b"RAREHIT in binary blob");
binary.extend(std::iter::repeat_n(0u8, 4096));
std::fs::write(dir.join("data.bin"), &binary).unwrap();
(dir, log)
}
fn bench_e2e(c: &mut Criterion) {
let (dir, log) = build_corpus();
let file = log.to_str().unwrap();
let dir_str = dir.to_str().unwrap();
// Pure scanning throughput: `-q` with a pattern that never matches forces a
// full scan and produces no output. A literal (which a buffer-at-a-time
// searcher can accelerate) versus an extended-regex control (which cannot).
{
let mut group = c.benchmark_group("scan");
group.bench_function("literal_no_match", |b| {
b.iter(|| run(black_box(&["-q", "NONEXISTENT_TOKEN_XYZ", file])))
});
group.bench_function("regex_no_match", |b| {
b.iter(|| run(black_box(&["-q", "-E", "NON[0-9]EXISTENT_TOKEN", file])))
});
group.finish();
}
// Real invocation shapes from the `grep` tldr page, each scanning the whole
// corpus. The `RAREHIT` marker matches only a handful of lines, so output
// stays small while the full-file scan dominates.
{
let mut group = c.benchmark_group("usage");
// Search for a pattern within a file.
group.bench_function("search_pattern", |b| {
b.iter(|| run(black_box(&["RAREHIT", file])))
});
// Search for an exact string (-F).
group.bench_function("fixed_string", |b| {
b.iter(|| run(black_box(&["-F", "RAREHIT", file])))
});
// Recursive search ignoring binary files (-rI).
group.bench_function("recursive_no_binary", |b| {
b.iter(|| run(black_box(&["-rI", "RAREHIT", dir_str])))
});
// Print 3 lines of context (-C 3).
group.bench_function("context", |b| {
b.iter(|| run(black_box(&["-C", "3", "RAREHIT", file])))
});
// Filename + line number with forced color (-Hn --color=always).
group.bench_function("filename_lineno_color", |b| {
b.iter(|| run(black_box(&["-Hn", "--color=always", "RAREHIT", file])))
});
// Print only the matched text (-o).
group.bench_function("only_matching", |b| {
b.iter(|| run(black_box(&["-o", "RAREHIT", file])))
});
// Invert match (-v); `worker-` is on every line, so nothing is printed
// and this measures the full inverted scan.
group.bench_function("invert_match", |b| {
b.iter(|| run(black_box(&["-v", "worker-", file])))
});
// Extended regex, case-insensitive (-Ei).
group.bench_function("extended_icase", |b| {
b.iter(|| run(black_box(&["-Ei", "rarehit", file])))
});
group.finish();
}
let _ = std::fs::remove_dir_all(Path::new(dir_str));
}
criterion_group!(benches, bench_e2e);
criterion_main!(benches);
+79 -68
View File
@@ -3,9 +3,12 @@
// For the full copyright and license information, please view the LICENSE // For the full copyright and license information, please view the LICENSE
// file that was distributed with this source code. // file that was distributed with this source code.
mod context_buffer; #[doc(hidden)]
mod line_buffer; pub mod context_buffer;
mod matcher; #[doc(hidden)]
pub mod line_buffer;
#[doc(hidden)]
pub mod matcher;
mod output; mod output;
mod searcher; mod searcher;
@@ -20,7 +23,8 @@ use std::path::Path;
use uucore::error::{FromIo, UResult, USimpleError}; use uucore::error::{FromIo, UResult, USimpleError};
#[derive(Clone, Copy, PartialEq, Eq)] #[derive(Clone, Copy, PartialEq, Eq)]
enum RegexMode { #[doc(hidden)]
pub enum RegexMode {
Fixed, Fixed,
Basic, Basic,
Extended, Extended,
@@ -28,7 +32,8 @@ enum RegexMode {
} }
#[derive(Clone, Copy, PartialEq, Eq)] #[derive(Clone, Copy, PartialEq, Eq)]
enum BinaryMode { #[doc(hidden)]
pub enum BinaryMode {
Binary, Binary,
Text, Text,
WithoutMatch, WithoutMatch,
@@ -42,79 +47,84 @@ enum ColorMode {
} }
#[derive(Clone, Copy, PartialEq, Eq)] #[derive(Clone, Copy, PartialEq, Eq)]
enum DirectoryMode { #[doc(hidden)]
pub enum DirectoryMode {
Read, Read,
Skip, Skip,
Recurse, Recurse,
} }
#[derive(Clone, Copy, PartialEq, Eq)] #[derive(Clone, Copy, PartialEq, Eq)]
enum DeviceMode { #[doc(hidden)]
pub enum DeviceMode {
Default, Default,
Read, Read,
Skip, Skip,
} }
struct ColorConfig<'a> { #[doc(hidden)]
matched_selected: &'a str, pub struct ColorConfig<'a> {
matched_context: &'a str, pub matched_selected: &'a str,
filename: &'a str, pub matched_context: &'a str,
line_number: &'a str, pub filename: &'a str,
byte_offset: &'a str, pub line_number: &'a str,
separator: &'a str, pub byte_offset: &'a str,
selected_line: &'a str, pub separator: &'a str,
context_line: &'a str, pub selected_line: &'a str,
pub context_line: &'a str,
reverse_video: bool, pub reverse_video: bool,
no_erase: bool, pub no_erase: bool,
} }
struct GlobSet { #[doc(hidden)]
pub struct GlobSet {
patterns: Vec<glob::Pattern>, patterns: Vec<glob::Pattern>,
} }
struct Config<'a> { #[doc(hidden)]
pub struct Config<'a> {
// Searcher // Searcher
directory_mode: DirectoryMode, pub directory_mode: DirectoryMode,
device_mode: DeviceMode, pub device_mode: DeviceMode,
follow_symlinks: bool, pub follow_symlinks: bool,
include_globs: GlobSet, pub include_globs: GlobSet,
exclude_globs: GlobSet, pub exclude_globs: GlobSet,
exclude_dir_globs: GlobSet, pub exclude_dir_globs: GlobSet,
label: &'a str, pub label: &'a str,
#[cfg(windows)] #[cfg(windows)]
strip_cr: bool, pub strip_cr: bool,
binary_mode: BinaryMode, pub binary_mode: BinaryMode,
max_count: Option<u64>, pub max_count: Option<u64>,
before_context: usize, pub before_context: usize,
after_context: usize, pub after_context: usize,
has_context: bool, pub has_context: bool,
// Matcher // Matcher
patterns: &'a [&'a str], pub patterns: &'a [&'a str],
regex_mode: RegexMode, pub regex_mode: RegexMode,
ignore_case: bool, pub ignore_case: bool,
invert_match: bool, pub invert_match: bool,
word_regexp: bool, pub word_regexp: bool,
line_regexp: bool, pub line_regexp: bool,
// Output // Output
quiet: bool, pub quiet: bool,
count: bool, pub count: bool,
show_filename: bool, pub show_filename: bool,
files_with_matches: bool, pub files_with_matches: bool,
files_without_match: bool, pub files_without_match: bool,
only_matching: bool, pub only_matching: bool,
byte_offset: bool, pub byte_offset: bool,
line_number: bool, pub line_number: bool,
initial_tab: bool, pub initial_tab: bool,
null_separator: bool, pub null_separator: bool,
null_data: bool, pub null_data: bool,
line_buffered: bool, pub line_buffered: bool,
no_messages: bool, pub no_messages: bool,
group_separator: Option<&'a str>, pub group_separator: Option<&'a str>,
use_color: bool, pub use_color: bool,
color_config: ColorConfig<'a>, pub color_config: ColorConfig<'a>,
} }
#[uucore::main(no_signals)] #[uucore::main(no_signals)]
@@ -342,13 +352,6 @@ pub fn uumain(args: impl uucore::Args) -> UResult<()> {
ColorMode::Never => false, ColorMode::Never => false,
ColorMode::Auto => std::io::stdout().is_terminal(), ColorMode::Auto => std::io::stdout().is_terminal(),
}; };
// GREP_COLOR is deprecated in favour of GREP_COLORS' `mt` capability;
// GNU warns about it, but only when color output is actually produced.
if use_color && !grep_color.is_empty() {
eprintln!(
"grep: warning: GREP_COLOR='{grep_color}' is deprecated; use GREP_COLORS='mt={grep_color}'"
);
}
let color_config = ColorConfig::from_env(&grep_color, &grep_colors); let color_config = ColorConfig::from_env(&grep_color, &grep_colors);
let config = Config { let config = Config {
@@ -861,8 +864,21 @@ fn expand_num_shorthand(args: impl Iterator<Item = OsString>) -> Vec<OsString> {
out out
} }
impl Default for GlobSet {
fn default() -> Self {
Self::new()
}
}
impl GlobSet { impl GlobSet {
fn with_capacity(capacity: usize) -> Self { /// Create an empty GlobSet.
pub fn new() -> Self {
Self {
patterns: Vec::new(),
}
}
pub fn with_capacity(capacity: usize) -> Self {
Self { Self {
patterns: Vec::with_capacity(capacity), patterns: Vec::with_capacity(capacity),
} }
@@ -910,11 +926,6 @@ impl<'a> ColorConfig<'a> {
for item in grep_colors.split(':') { for item in grep_colors.split(':') {
if let Some((key, value)) = item.split_once('=') { if let Some((key, value)) = item.split_once('=') {
match key { match key {
// `mt` sets both the selected- and context-match colors.
"mt" => {
config.matched_selected = value;
config.matched_context = value;
}
"ms" => config.matched_selected = value, "ms" => config.matched_selected = value,
"mc" => config.matched_context = value, "mc" => config.matched_context = value,
"fn" => config.filename = value, "fn" => config.filename = value,
+60 -263
View File
@@ -5,14 +5,10 @@
use crate::{Config, RegexMode}; use crate::{Config, RegexMode};
use onig::{ use onig::{
EncodedBytes, Error, MatchParam, Regex, RegexOptions, Region, SearchOptions, Syntax, EncodedBytes, Regex, RegexOptions, Region, SearchOptions, Syntax, SyntaxBehavior,
SyntaxBehavior, SyntaxOperator, SyntaxOperator,
}; };
use onig_sys::{ use onig_sys::{OnigEncCtype_ONIGENC_CTYPE_WORD, OnigEncodingUTF8};
ONIGERR_EMPTY_RANGE_IN_CHAR_CLASS, ONIGERR_INVALID_BACKREF, ONIGERR_RETRY_LIMIT_IN_MATCH_OVER,
ONIGERR_RETRY_LIMIT_IN_SEARCH_OVER, OnigEncCtype_ONIGENC_CTYPE_WORD, OnigEncodingUTF8,
};
use std::io;
use uucore::error::{UResult, USimpleError}; use uucore::error::{UResult, USimpleError};
pub struct Matcher<'a> { pub struct Matcher<'a> {
@@ -30,30 +26,26 @@ impl<'a> Matcher<'a> {
} }
/// Decide whether `line` matches and return the positions to highlight. /// Decide whether `line` matches and return the positions to highlight.
/// pub fn match_line(&self, line: &[u8]) -> Option<Vec<(usize, usize)>> {
/// Returns an error if the regex engine bails out (e.g. it exceeds its
/// backtracking retry limit on a pathological pattern); the caller turns
/// that into a GNU-style diagnostic and exit code 2 rather than aborting.
pub fn match_line(&self, line: &[u8]) -> io::Result<Option<Vec<(usize, usize)>>> {
let mut any_seen = false; let mut any_seen = false;
let mut positions = Vec::new(); let positions: Vec<_> = MatchIter::new(&self.patterns, line)
let mut iter = MatchIter::new(&self.patterns, line).map_err(match_error)?; .filter(|&(start, end)| {
while let Some((start, end)) = iter.next_match().map_err(match_error)? {
any_seen = true; any_seen = true;
// Drop zero-length matches from the output. // Drop zero-length matches from the output.
if start == end { if start == end {
continue; return false;
} }
// Drop matches that don't span the whole line if `-x` was requested. // Drop matches that don't span the whole line if `-x` was requested.
if self.config.line_regexp && !(start == 0 && end == line.len()) { if self.config.line_regexp && !(start == 0 && end == line.len()) {
continue; return false;
} }
// Drop matches that aren't word matches if `-w` was requested. // Drop matches that aren't word matches if `-w` was requested.
if self.config.word_regexp && !Self::is_word_match(line, start, end) { if self.config.word_regexp && !Self::is_word_match(line, start, end) {
continue; return false;
}
positions.push((start, end));
} }
true
})
.collect();
let raw_matched = if self.config.line_regexp || self.config.word_regexp { let raw_matched = if self.config.line_regexp || self.config.word_regexp {
// -w / -x are authoritative once positions are filtered. // -w / -x are authoritative once positions are filtered.
@@ -62,25 +54,23 @@ impl<'a> Matcher<'a> {
any_seen any_seen
}; };
Ok((raw_matched != self.config.invert_match).then_some(positions)) if raw_matched != self.config.invert_match {
Some(positions)
} else {
None
}
} }
/// Cheap match check that doesn't enumerate positions. /// Cheap match check that doesn't enumerate positions.
pub fn is_match(&self, line: &[u8]) -> io::Result<Option<Vec<(usize, usize)>>> { pub fn is_match(&self, line: &[u8]) -> Option<Vec<(usize, usize)>> {
// `-w` / `-x` need positions to filter, so we fall back to `match_line`. // `-w` / `-x` need positions to filter, so we fall back to `match_line`.
let matched = if self.config.line_regexp || self.config.word_regexp { let matched = if self.config.line_regexp || self.config.word_regexp {
self.match_line(line)?.is_some() self.match_line(line).is_some()
} else { } else {
let mut raw_matched = false; let raw_matched = self.patterns.iter().any(|p| p.is_match(line));
for p in &self.patterns {
if p.is_match(line).map_err(match_error)? {
raw_matched = true;
break;
}
}
raw_matched != self.config.invert_match raw_matched != self.config.invert_match
}; };
Ok(matched.then(Vec::new)) matched.then(Vec::new)
} }
/// Word-boundary check `-w`. /// Word-boundary check `-w`.
@@ -122,31 +112,35 @@ struct MatchIter<'a> {
} }
impl<'a> MatchIter<'a> { impl<'a> MatchIter<'a> {
fn new(patterns: &'a [CompiledPattern], line: &'a [u8]) -> Result<Self, Error> { fn new(patterns: &'a [CompiledPattern], line: &'a [u8]) -> Self {
let mut cursors = Vec::with_capacity(patterns.len()); Self {
for pattern in patterns { cursors: patterns
.iter()
.map(|pattern| {
let mut c = Cursor { let mut c = Cursor {
pattern, pattern,
line, line,
offset: 0, offset: 0,
pending: None, pending: None,
}; };
c.refill()?; c.refill();
cursors.push(c); c
}
Ok(Self {
cursors,
last_end: 0,
}) })
.collect(),
last_end: 0,
}
}
} }
/// Yield the next match across all patterns, or `None` when exhausted. impl<'a> Iterator for MatchIter<'a> {
fn next_match(&mut self) -> Result<Option<(usize, usize)>, Error> { type Item = (usize, usize);
fn next(&mut self) -> Option<Self::Item> {
// Discard stale pendings that fall before the last emit. // Discard stale pendings that fall before the last emit.
for cursor in &mut self.cursors { for cursor in &mut self.cursors {
if matches!(cursor.pending, Some((s, _)) if s < self.last_end) { if matches!(cursor.pending, Some((s, _)) if s < self.last_end) {
cursor.offset = self.last_end; cursor.offset = self.last_end;
cursor.refill()?; cursor.refill();
} }
} }
@@ -159,15 +153,12 @@ impl<'a> MatchIter<'a> {
.enumerate() .enumerate()
.filter_map(|(i, c)| c.pending.map(|p| (i, p))) .filter_map(|(i, c)| c.pending.map(|p| (i, p)))
.min_by_key(|&(_, (s, e))| (s, std::cmp::Reverse(e))) .min_by_key(|&(_, (s, e))| (s, std::cmp::Reverse(e)))
.map(|(i, _)| i); .map(|(i, _)| i)?;
let Some(best_idx) = best_idx else {
return Ok(None);
};
let (start, end) = self.cursors[best_idx].pending.unwrap(); let (start, end) = self.cursors[best_idx].pending.unwrap();
self.cursors[best_idx].refill()?; self.cursors[best_idx].refill();
self.last_end = end; self.last_end = end;
Ok(Some((start, end))) Some((start, end))
} }
} }
@@ -182,25 +173,24 @@ struct Cursor<'a> {
} }
impl Cursor<'_> { impl Cursor<'_> {
fn refill(&mut self) -> Result<(), Error> { fn refill(&mut self) {
if self.offset >= self.line.len() { if self.offset >= self.line.len() {
self.pending = None; self.pending = None;
return Ok(()); return;
} }
let Some((start, leftmost_end)) = self.pattern.search_leftmost(self.line, self.offset)? let Some((start, leftmost_end)) = self.pattern.search_leftmost(self.line, self.offset)
else { else {
self.pending = None; self.pending = None;
return Ok(()); return;
}; };
let end = self let end = self
.pattern .pattern
.longest_end_at(self.line, start)? .longest_end_at(self.line, start)
.unwrap_or(leftmost_end); .unwrap_or(leftmost_end);
// Advance the next search past the match we just found. // Advance the next search past the match we just found.
// Zero-length matches need a +1 nudge to avoid spinning forever. // Zero-length matches need a +1 nudge to avoid spinning forever.
self.offset = end.max(start + 1); self.offset = end.max(start + 1);
self.pending = Some((start, end)); self.pending = Some((start, end));
Ok(())
} }
} }
@@ -215,12 +205,6 @@ struct CompiledPattern {
impl CompiledPattern { impl CompiledPattern {
fn compile(pattern: &str, config: &Config) -> UResult<Self> { fn compile(pattern: &str, config: &Config) -> UResult<Self> {
// GNU grep rejects the confusing `[:name:]` bracket form (a misspelled
// `[[:name:]]`) in basic/extended modes; oniguruma accepts it silently.
if matches!(config.regex_mode, RegexMode::Basic | RegexMode::Extended) {
check_confusing_bracket(pattern)?;
}
let mut syntax = *match config.regex_mode { let mut syntax = *match config.regex_mode {
RegexMode::Fixed => Syntax::asis(), RegexMode::Fixed => Syntax::asis(),
RegexMode::Basic => Syntax::grep(), RegexMode::Basic => Syntax::grep(),
@@ -248,29 +232,17 @@ impl CompiledPattern {
options |= RegexOptions::REGEX_OPTION_IGNORECASE; options |= RegexOptions::REGEX_OPTION_IGNORECASE;
} }
let mode = config.regex_mode; fn compile_with(pattern: &str, syntax: &Syntax, options: RegexOptions) -> UResult<Regex> {
fn compile_with(
pattern: &str,
syntax: &Syntax,
options: RegexOptions,
mode: RegexMode,
) -> UResult<Regex> {
Regex::with_options_and_encoding(pattern, options, syntax).map_err(|err| { Regex::with_options_and_encoding(pattern, options, syntax).map_err(|err| {
// Prefer GNU grep's wording for the errors it has a dedicated USimpleError::new(2, format!("invalid pattern \"{pattern}\": {err}"))
// message for; fall back to oniguruma's text otherwise.
match gnu_error_message(err.code(), mode) {
Some(msg) => USimpleError::new(2, msg.to_string()),
None => USimpleError::new(2, format!("invalid pattern \"{pattern}\": {err}")),
}
}) })
} }
let leftmost = compile_with(pattern, &syntax, options, mode)?; let leftmost = compile_with(pattern, &syntax, options)?;
let longest_anchored = compile_with( let longest_anchored = compile_with(
pattern, pattern,
&syntax, &syntax,
options | RegexOptions::REGEX_OPTION_FIND_LONGEST, options | RegexOptions::REGEX_OPTION_FIND_LONGEST,
mode,
)?; )?;
Ok(Self { Ok(Self {
leftmost, leftmost,
@@ -279,216 +251,41 @@ impl CompiledPattern {
} }
/// Find the leftmost match starting at or after `offset`. /// Find the leftmost match starting at or after `offset`.
fn search_leftmost(&self, line: &[u8], offset: usize) -> Result<Option<(usize, usize)>, Error> { fn search_leftmost(&self, line: &[u8], offset: usize) -> Option<(usize, usize)> {
let mut region = Region::new(); let mut region = Region::new();
let found = self.leftmost.search_with_param( self.leftmost.search_with_encoding(
EncodedBytes::from_parts(line, &raw mut OnigEncodingUTF8), EncodedBytes::from_parts(line, &raw mut OnigEncodingUTF8),
offset, offset,
line.len(), line.len(),
SearchOptions::SEARCH_OPTION_NONE, SearchOptions::SEARCH_OPTION_NONE,
Some(&mut region), Some(&mut region),
MatchParam::default(),
)?; )?;
Ok(found.and_then(|_| region.pos(0))) region.pos(0)
} }
/// Given a known leftmost start `start`, return the longest extent /// Given a known leftmost start `start`, return the longest extent
/// of a match anchored exactly there = POSIX leftmost-longest end. /// of a match anchored exactly there = POSIX leftmost-longest end.
fn longest_end_at(&self, line: &[u8], start: usize) -> Result<Option<usize>, Error> { fn longest_end_at(&self, line: &[u8], start: usize) -> Option<usize> {
let mut region = Region::new(); let mut region = Region::new();
self.longest_anchored.match_with_param( self.longest_anchored.match_with_encoding(
EncodedBytes::from_parts(line, &raw mut OnigEncodingUTF8), EncodedBytes::from_parts(line, &raw mut OnigEncodingUTF8),
start, start,
SearchOptions::SEARCH_OPTION_NONE, SearchOptions::SEARCH_OPTION_NONE,
Some(&mut region), Some(&mut region),
MatchParam::default(), );
)?; region.pos(0).map(|(_, end)| end)
Ok(region.pos(0).map(|(_, end)| end))
} }
/// True if any match exists in `line` (including zero-length). /// True if any match exists in `line` (including zero-length).
fn is_match(&self, line: &[u8]) -> Result<bool, Error> { fn is_match(&self, line: &[u8]) -> bool {
Ok(self self.leftmost
.leftmost .search_with_encoding(
.search_with_param(
EncodedBytes::from_parts(line, &raw mut OnigEncodingUTF8), EncodedBytes::from_parts(line, &raw mut OnigEncodingUTF8),
0, 0,
line.len(), line.len(),
SearchOptions::SEARCH_OPTION_NONE, SearchOptions::SEARCH_OPTION_NONE,
None, None,
MatchParam::default(), )
)? .is_some()
.is_some())
} }
} }
/// Convert a regex-engine match-time error into an I/O error carrying GNU
/// grep's wording. The only error we expect in practice is the backtracking
/// retry limit being exceeded on a pathological pattern; GNU reports this as
/// `exceeded PCRE's backtracking limit` and exits 2 instead of aborting.
fn match_error(err: Error) -> io::Error {
let message = if matches!(
err.code(),
ONIGERR_RETRY_LIMIT_IN_MATCH_OVER | ONIGERR_RETRY_LIMIT_IN_SEARCH_OVER
) {
"exceeded PCRE's backtracking limit".to_string()
} else {
err.description().to_string()
};
io::Error::other(message)
}
/// Map an oniguruma compile-error code to GNU grep's wording for the same
/// condition, when one exists. GNU emits a bare POSIX-style diagnostic (e.g.
/// `Invalid range end`) rather than oniguruma's phrasing, so translating keeps
/// us byte-compatible. Returns `None` for errors with no GNU equivalent, where
/// the caller falls back to oniguruma's own message.
fn gnu_error_message(code: i32, mode: RegexMode) -> Option<&'static str> {
match code {
// e.g. `[b-a]`: a range whose end precedes its start.
ONIGERR_EMPTY_RANGE_IN_CHAR_CLASS => Some("Invalid range end"),
// e.g. `(.)\2`: a back-reference to a group that does not exist. GNU
// (via PCRE2) and gnulib's regex word this differently.
ONIGERR_INVALID_BACKREF if mode == RegexMode::Perl => {
Some("reference to non-existent subpattern")
}
ONIGERR_INVALID_BACKREF => Some("Invalid back reference"),
_ => None,
}
}
/// Reject the confusing `[:name:]` bracket form the way GNU grep does.
///
/// A bracket expression like `[:space:]` is almost always a misspelled
/// `[[:space:]]`; GNU grep flags it with a dedicated diagnostic and exits 2,
/// whereas oniguruma silently treats it as the set `{':','s','p',…}`. This
/// scans the pattern for that form and returns the same error.
fn check_confusing_bracket(pattern: &str) -> UResult<()> {
let bytes = pattern.as_bytes();
let mut i = 0;
while i < bytes.len() {
match bytes[i] {
// Outside a bracket a backslash escapes the next character, so
// `\[` does not open a bracket expression.
b'\\' => i += 2,
b'[' => {
i += 1;
if bracket_warns(bytes, &mut i) {
return Err(USimpleError::new(
2,
"character class syntax is [[:space:]], not [:space:]".to_string(),
));
}
}
_ => i += 1,
}
}
Ok(())
}
/// Consume a single bracket expression starting just past its opening `[` and
/// report whether GNU grep's colon warning fires for it.
///
/// This is a faithful port of the `colon_warning_state` logic in GNU grep's
/// `parse_bracket_exp` (gnulib `dfa.c`). The state is a bitmask:
/// bit 0 — first character is a colon
/// bit 1 — last character is a colon
/// bit 2 — includes some other (non-colon) character
/// bit 3 — includes a range, char/equivalence class, or collating element
/// The warning fires exactly when the state ends equal to `7` (bits 02 set,
/// bit 3 clear). On the way it advances `i` past the closing `]`.
fn bracket_warns(bytes: &[u8], i: &mut usize) -> bool {
fn fetch(bytes: &[u8], i: &mut usize) -> Option<u8> {
let b = bytes.get(*i).copied();
if b.is_some() {
*i += 1;
}
b
}
let Some(first) = fetch(bytes, i) else {
return false;
};
let mut c = first;
if c == b'^' {
match fetch(bytes, i) {
Some(x) => c = x,
None => return false,
}
}
let mut state: u8 = u8::from(c == b':');
'scan: loop {
state &= !2;
let mut c1: Option<u8> = None;
if c == b'[' {
let Some(nc1) = fetch(bytes, i) else {
return false;
};
// `[:`, `[.` and `[=` introduce a class / collating / equivalence
// element; consume it whole and mark bit 3.
if nc1 == b':' || nc1 == b'.' || nc1 == b'=' {
loop {
match fetch(bytes, i) {
None => break,
Some(cc) if cc == nc1 && bytes.get(*i).copied() == Some(b']') => break,
Some(_) => {}
}
}
if fetch(bytes, i).is_none() {
return false; // consumes the `]`
}
state |= 8;
match fetch(bytes, i) {
Some(b']') => break 'scan,
Some(x) => {
c = x;
continue 'scan;
}
None => return false,
}
}
// Otherwise `[` is an ordinary character; `nc1` is the lookahead.
c1 = Some(nc1);
}
if c1.is_none() {
c1 = fetch(bytes, i);
}
if c1 == Some(b'-') {
let Some(mut c2) = fetch(bytes, i) else {
return false;
};
if c2 == b'[' && bytes.get(*i).copied() == Some(b'.') {
c2 = b']';
}
if c2 == b']' {
// `[x-]`: the hyphen is a literal; put the `]` back so the
// loop terminator sees it next.
*i -= 1;
} else {
state |= 8;
match fetch(bytes, i) {
Some(b']') => break 'scan,
Some(x) => {
c = x;
continue 'scan;
}
None => return false,
}
}
}
state |= if c == b':' { 2 } else { 4 };
match c1 {
Some(b']') => break 'scan,
Some(x) => c = x,
None => return false,
}
}
state == 7
}
+3 -3
View File
@@ -281,7 +281,7 @@ impl<'a> Searcher<'a> {
return Ok(false); return Ok(false);
} }
if let Some(positions) = self.session_match_line(line)? { if let Some(positions) = self.session_match_line(line) {
// TODO: GNU grep respects LANG. Here, I'm always checking for valid UTF-8. // TODO: GNU grep respects LANG. Here, I'm always checking for valid UTF-8.
if !self.session_mark_binary_if(|| std::str::from_utf8(line).is_err()) { if !self.session_mark_binary_if(|| std::str::from_utf8(line).is_err()) {
return Ok(false); return Ok(false);
@@ -321,9 +321,9 @@ impl<'a> Searcher<'a> {
self.config.binary_mode != BinaryMode::WithoutMatch self.config.binary_mode != BinaryMode::WithoutMatch
} }
fn session_match_line(&self, line: &[u8]) -> io::Result<Option<Vec<(usize, usize)>>> { fn session_match_line(&self, line: &[u8]) -> Option<Vec<(usize, usize)>> {
if !self.session_can_match() { if !self.session_can_match() {
Ok(None) None
} else if self.session_needs_match_positions() { } else if self.session_needs_match_positions() {
self.matcher.match_line(line) self.matcher.match_line(line)
} else { } else {
-136
View File
@@ -126,142 +126,6 @@ fn ere_invalid_pattern_is_error() {
.stderr_contains("invalid pattern"); .stderr_contains("invalid pattern");
} }
#[test]
fn confusing_bracket_class_is_error() {
// GNU grep rejects the misspelled `[:name:]` form (meant to be
// `[[:name:]]`) with a dedicated diagnostic and exit code 2.
// No piped input: the pattern is rejected at compile time, before stdin is
// read, so feeding stdin would race with the child exiting (broken pipe).
for pattern in ["[:space:]", "[:digit:]", "[^:space:]", "x[:space:]y"] {
let (_s, mut c) = ucmd();
c.args(&[pattern])
.fails_with_code(2)
.stderr_is("grep: character class syntax is [[:space:]], not [:space:]\n");
}
// The same diagnostic applies in extended mode.
let (_s, mut c) = ucmd();
c.args(&["-E", "[:space:]"])
.fails_with_code(2)
.stderr_is("grep: character class syntax is [[:space:]], not [:space:]\n");
}
#[test]
fn lookalike_brackets_are_not_confusing() {
// Patterns that are NOT the confusing `[:name:]` form must compile
// normally (no diagnostic). A proper class, a colon set, a range, a
// trailing colon set, and `-F` literal text all stay valid.
for pattern in [
"[[:space:]]",
"[::]",
"[:space]",
"[:spac-e:]",
"[a:space:]",
] {
let (_s, mut c) = ucmd();
c.args(&[pattern])
.pipe_in("z\n")
.fails_with_code(1)
.no_output();
}
// `\[` does not open a bracket expression.
let (_s, mut c) = ucmd();
c.args(&["\\[:space:]"])
.pipe_in("z\n")
.fails_with_code(1)
.no_output();
// `-F` treats the text literally, so no diagnostic.
let (_s, mut c) = ucmd();
c.args(&["-F", "[:space:]"])
.pipe_in("x\n")
.fails_with_code(1)
.no_output();
}
#[test]
fn reversed_range_uses_gnu_wording() {
// A range like `[b-a]` is an error; GNU prints the bare POSIX diagnostic
// "Invalid range end" (not oniguruma's phrasing) and exits 2.
// No piped input: the pattern is rejected before stdin is read.
for args in [&["[b-a]"][..], &["-E", "[b-a]"][..]] {
let (_s, mut c) = ucmd();
c.args(args)
.fails_with_code(2)
.stderr_is("grep: Invalid range end\n");
}
}
#[test]
fn pcre_backtracking_limit_does_not_abort() {
// A pathological PCRE pattern can exceed oniguruma's retry limit. GNU
// grep reports this and exits 2 (it must not crash); stdout stays empty.
let (_s, mut c) = ucmd();
c.args(&["-P", "((a+)*)+$"])
.pipe_in("aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaab\n")
.fails_with_code(2)
.stdout_is("")
.stderr_contains("backtracking limit");
}
#[test]
fn invalid_backreference_uses_gnu_wording() {
// A back-reference to a non-existent group is worded differently by GNU
// depending on the engine: PCRE (-P) vs gnulib regex (BRE/ERE).
// No piped input: these patterns are rejected before stdin is read.
let (_s, mut c) = ucmd();
c.args(&["-P", r"(.)\2"])
.fails_with_code(2)
.stderr_is("grep: reference to non-existent subpattern\n");
for args in [&["-E", r"(.)\2"][..], &[r"\(.\)\2"][..]] {
let (_s, mut c) = ucmd();
c.args(args)
.fails_with_code(2)
.stderr_is("grep: Invalid back reference\n");
}
// A valid back-reference with -Pw / -Px must still match.
for flag in ["-Pw", "-Px"] {
let (_s, mut c) = ucmd();
c.args(&[flag, r"(.)\1"])
.pipe_in("aa\n")
.succeeds()
.stdout_is("aa\n");
}
}
#[test]
fn grep_colors_mt_and_grep_color_deprecation() {
// GREP_COLORS `mt` sets the match color (both selected and context).
let (_s, mut c) = ucmd();
c.args(&["--color=always", "."])
.env("GREP_COLORS", "mt=36")
.pipe_in("x\n")
.succeeds()
.stdout_is("\u{1b}[36m\u{1b}[Kx\u{1b}[m\u{1b}[K\n");
// GREP_COLOR is deprecated: it still sets the match color, but emits a
// warning when color is actually produced.
let (_s, mut c) = ucmd();
c.args(&["--color=always", "."])
.env("GREP_COLOR", "36")
.pipe_in("x\n")
.succeeds()
.stdout_is("\u{1b}[36m\u{1b}[Kx\u{1b}[m\u{1b}[K\n")
.stderr_is("grep: warning: GREP_COLOR='36' is deprecated; use GREP_COLORS='mt=36'\n");
// No warning when color output is disabled.
let (_s, mut c) = ucmd();
c.args(&["--color=never", "."])
.env("GREP_COLOR", "36")
.pipe_in("x\n")
.succeeds()
.stdout_is("x\n")
.no_stderr();
}
#[test] #[test]
fn fixed_string_is_literal() { fn fixed_string_is_literal() {
// Metacharacters are not interpreted. // Metacharacters are not interpreted.