mirror of
https://github.com/uutils/grep.git
synced 2026-06-10 16:15:11 -07:00
Compare commits
5
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
be51c04c08 | ||
|
|
da32a63663 | ||
|
|
9c21a7d2f0 | ||
|
|
e767a30c1a | ||
|
|
1a3e8a391d |
@@ -1,37 +0,0 @@
|
|||||||
name: CodSpeed
|
|
||||||
|
|
||||||
on:
|
|
||||||
push:
|
|
||||||
branches:
|
|
||||||
- "main"
|
|
||||||
pull_request:
|
|
||||||
# `workflow_dispatch` allows CodSpeed to trigger backtest
|
|
||||||
# performance analysis in order to generate initial data.
|
|
||||||
workflow_dispatch:
|
|
||||||
|
|
||||||
permissions:
|
|
||||||
contents: read
|
|
||||||
id-token: write
|
|
||||||
|
|
||||||
jobs:
|
|
||||||
codspeed:
|
|
||||||
name: Run benchmarks
|
|
||||||
runs-on: ubuntu-latest
|
|
||||||
steps:
|
|
||||||
- uses: actions/checkout@v4
|
|
||||||
|
|
||||||
- name: Setup Rust toolchain, cache and cargo-codspeed binary
|
|
||||||
uses: moonrepo/setup-rust@v0
|
|
||||||
with:
|
|
||||||
channel: stable
|
|
||||||
cache-target: release
|
|
||||||
bins: cargo-codspeed
|
|
||||||
|
|
||||||
- name: Build the benchmark target(s)
|
|
||||||
run: cargo codspeed build
|
|
||||||
|
|
||||||
- name: Run the benchmarks
|
|
||||||
uses: CodSpeedHQ/action@v4
|
|
||||||
with:
|
|
||||||
mode: simulation
|
|
||||||
run: cargo codspeed run
|
|
||||||
@@ -1,55 +0,0 @@
|
|||||||
# See https://pre-commit.com for more information
|
|
||||||
# See https://pre-commit.com/hooks.html for more hooks
|
|
||||||
exclude: ^tests/fixtures/
|
|
||||||
repos:
|
|
||||||
- repo: https://github.com/pre-commit/pre-commit-hooks
|
|
||||||
rev: v6.0.0
|
|
||||||
hooks:
|
|
||||||
- id: check-added-large-files
|
|
||||||
- id: check-executables-have-shebangs
|
|
||||||
- id: check-json
|
|
||||||
exclude: '\.vscode/(cSpell|extensions)\.json' # cSpell.json and extensions.json use comments
|
|
||||||
- id: check-shebang-scripts-are-executable
|
|
||||||
exclude: '.+\.rs' # would be triggered by #![some_attribute]
|
|
||||||
- id: check-symlinks
|
|
||||||
- id: check-toml
|
|
||||||
- id: check-yaml
|
|
||||||
args: [ --allow-multiple-documents ]
|
|
||||||
- id: destroyed-symlinks
|
|
||||||
- id: end-of-file-fixer
|
|
||||||
- id: mixed-line-ending
|
|
||||||
args: [ --fix=lf ]
|
|
||||||
- id: trailing-whitespace
|
|
||||||
|
|
||||||
- repo: local
|
|
||||||
hooks:
|
|
||||||
- id: rust-linting
|
|
||||||
name: Rust linting
|
|
||||||
description: Run cargo fmt on files included in the commit.
|
|
||||||
entry: cargo +stable fmt --
|
|
||||||
pass_filenames: true
|
|
||||||
types: [file, rust]
|
|
||||||
language: system
|
|
||||||
- id: rust-clippy
|
|
||||||
name: Rust clippy
|
|
||||||
description: Run cargo clippy on files included in the commit.
|
|
||||||
entry: cargo +stable clippy --workspace --all-targets --all-features -- -D warnings
|
|
||||||
pass_filenames: false
|
|
||||||
types: [file, rust]
|
|
||||||
language: system
|
|
||||||
- id: cargo-lock-check
|
|
||||||
name: Cargo.lock sync check
|
|
||||||
description: Ensure Cargo.lock and fuzz/Cargo.lock are up-to-date.
|
|
||||||
entry: bash -c 'for dir in . fuzz; do if [ -d "$dir" ]; then ( cd "$dir" && cargo fetch --quiet ); fi; done'
|
|
||||||
pass_filenames: false
|
|
||||||
files: 'Cargo\.(toml|lock)$'
|
|
||||||
language: system
|
|
||||||
- id: cspell
|
|
||||||
name: Code spell checker (cspell)
|
|
||||||
description: Run cspell to check for spelling errors (if available).
|
|
||||||
entry: bash -c 'if command -v cspell >/dev/null 2>&1; then cspell --no-must-find-files -- "$@"; else echo "cspell not found, skipping spell check"; exit 0; fi' --
|
|
||||||
pass_filenames: true
|
|
||||||
language: system
|
|
||||||
|
|
||||||
ci:
|
|
||||||
skip: [rust-linting, rust-clippy, cargo-lock-check, cspell]
|
|
||||||
Generated
+11
-532
File diff suppressed because it is too large
Load Diff
@@ -27,10 +27,5 @@ onig_sys = { version = "*", default-features = false }
|
|||||||
uucore = "0.8.0"
|
uucore = "0.8.0"
|
||||||
walkdir = "2.5"
|
walkdir = "2.5"
|
||||||
|
|
||||||
[[bench]]
|
|
||||||
name = "grep_bench"
|
|
||||||
harness = false
|
|
||||||
|
|
||||||
[dev-dependencies]
|
[dev-dependencies]
|
||||||
criterion = { version = "4.7.0", package = "codspeed-criterion-compat" }
|
|
||||||
uutests = "0.8.0"
|
uutests = "0.8.0"
|
||||||
|
|||||||
@@ -4,7 +4,6 @@
|
|||||||
[](https://deps.rs/repo/github/uutils/grep)
|
[](https://deps.rs/repo/github/uutils/grep)
|
||||||
|
|
||||||
[](https://codecov.io/gh/uutils/grep)
|
[](https://codecov.io/gh/uutils/grep)
|
||||||
[](https://codspeed.io/uutils/grep?utm_source=badge)
|
|
||||||
|
|
||||||
# Grep, now in Rust
|
# Grep, now in Rust
|
||||||
|
|
||||||
@@ -30,15 +29,10 @@ cargo build --release
|
|||||||
cargo test
|
cargo test
|
||||||
```
|
```
|
||||||
|
|
||||||
## Pre-commit hooks
|
|
||||||
|
|
||||||
This project uses [pre-commit](https://pre-commit.com); run `pre-commit install` to enable the git hooks.
|
|
||||||
|
|
||||||
## Known Issues
|
## Known Issues
|
||||||
|
|
||||||
* Does not take `LANG`, etc., into account for handling file encodings (non-UTF8 matches are treated as binary)
|
* Does not take `LANG`, etc., into account for handling file encodings (non-UTF8 matches are treated as binary)
|
||||||
* No localization support yet
|
* No localization support yet
|
||||||
* Performances need to be improved
|
|
||||||
|
|
||||||
## Contributing
|
## Contributing
|
||||||
|
|
||||||
|
|||||||
@@ -1,128 +0,0 @@
|
|||||||
// This file is part of the uutils grep package.
|
|
||||||
//
|
|
||||||
// For the full copyright and license information, please view the LICENSE
|
|
||||||
// file that was distributed with this source code.
|
|
||||||
|
|
||||||
use criterion::{Criterion, black_box, criterion_group, criterion_main};
|
|
||||||
use std::ffi::OsString;
|
|
||||||
use std::path::Path;
|
|
||||||
|
|
||||||
/// Run grep end-to-end through the real `uumain` entry point. `args` are the
|
|
||||||
/// arguments after the program name (flags, pattern, paths). The exit status is
|
|
||||||
/// ignored — we only care about the work performed.
|
|
||||||
fn run(args: &[&str]) {
|
|
||||||
let mut argv: Vec<OsString> = Vec::with_capacity(args.len() + 1);
|
|
||||||
argv.push(OsString::from("grep"));
|
|
||||||
argv.extend(args.iter().map(OsString::from));
|
|
||||||
let _ = uu_grep::uumain(argv.into_iter());
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Build a multi-megabyte log-like corpus plus a directory holding it alongside
|
|
||||||
/// a binary file. Every line contains `worker-<n>` and a `2024-…` timestamp; a
|
|
||||||
/// rare `RAREHIT` marker appears on a handful of lines (≈ every 10000th).
|
|
||||||
/// Returns `(dir, log_file)`.
|
|
||||||
fn build_corpus() -> (std::path::PathBuf, std::path::PathBuf) {
|
|
||||||
let mut content = String::new();
|
|
||||||
for i in 0..80_000u32 {
|
|
||||||
if i % 10_000 == 0 {
|
|
||||||
content.push_str(&format!(
|
|
||||||
"2024-01-15 10:30:{:02} RAREHIT worker-{i} special marker seen\n",
|
|
||||||
i % 60
|
|
||||||
));
|
|
||||||
} else if i % 100 == 0 {
|
|
||||||
content.push_str(&format!(
|
|
||||||
"2024-01-15 10:30:{:02} ERROR worker-{i} connection reset\n",
|
|
||||||
i % 60
|
|
||||||
));
|
|
||||||
} else {
|
|
||||||
content.push_str(&format!(
|
|
||||||
"2024-01-15 10:30:{:02} INFO worker-{i} request handled in {}ms\n",
|
|
||||||
i % 60,
|
|
||||||
i % 1000
|
|
||||||
));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
assert!(content.len() > 4 * 1024 * 1024);
|
|
||||||
|
|
||||||
let dir = std::env::temp_dir().join(format!("uu_grep_bench_{}", std::process::id()));
|
|
||||||
std::fs::create_dir_all(&dir).unwrap();
|
|
||||||
let log = dir.join("app.log");
|
|
||||||
std::fs::write(&log, &content).unwrap();
|
|
||||||
|
|
||||||
// A binary file (contains NUL) that also holds the marker, so `-I` has
|
|
||||||
// something to skip while recursing.
|
|
||||||
let mut binary = vec![0u8, 1, 2, 3];
|
|
||||||
binary.extend_from_slice(b"RAREHIT in binary blob");
|
|
||||||
binary.extend(std::iter::repeat_n(0u8, 4096));
|
|
||||||
std::fs::write(dir.join("data.bin"), &binary).unwrap();
|
|
||||||
|
|
||||||
(dir, log)
|
|
||||||
}
|
|
||||||
|
|
||||||
fn bench_e2e(c: &mut Criterion) {
|
|
||||||
let (dir, log) = build_corpus();
|
|
||||||
let file = log.to_str().unwrap();
|
|
||||||
let dir_str = dir.to_str().unwrap();
|
|
||||||
|
|
||||||
// Pure scanning throughput: `-q` with a pattern that never matches forces a
|
|
||||||
// full scan and produces no output. A literal (which a buffer-at-a-time
|
|
||||||
// searcher can accelerate) versus an extended-regex control (which cannot).
|
|
||||||
{
|
|
||||||
let mut group = c.benchmark_group("scan");
|
|
||||||
group.bench_function("literal_no_match", |b| {
|
|
||||||
b.iter(|| run(black_box(&["-q", "NONEXISTENT_TOKEN_XYZ", file])))
|
|
||||||
});
|
|
||||||
group.bench_function("regex_no_match", |b| {
|
|
||||||
b.iter(|| run(black_box(&["-q", "-E", "NON[0-9]EXISTENT_TOKEN", file])))
|
|
||||||
});
|
|
||||||
group.finish();
|
|
||||||
}
|
|
||||||
|
|
||||||
// Real invocation shapes from the `grep` tldr page, each scanning the whole
|
|
||||||
// corpus. The `RAREHIT` marker matches only a handful of lines, so output
|
|
||||||
// stays small while the full-file scan dominates.
|
|
||||||
{
|
|
||||||
let mut group = c.benchmark_group("usage");
|
|
||||||
|
|
||||||
// Search for a pattern within a file.
|
|
||||||
group.bench_function("search_pattern", |b| {
|
|
||||||
b.iter(|| run(black_box(&["RAREHIT", file])))
|
|
||||||
});
|
|
||||||
// Search for an exact string (-F).
|
|
||||||
group.bench_function("fixed_string", |b| {
|
|
||||||
b.iter(|| run(black_box(&["-F", "RAREHIT", file])))
|
|
||||||
});
|
|
||||||
// Recursive search ignoring binary files (-rI).
|
|
||||||
group.bench_function("recursive_no_binary", |b| {
|
|
||||||
b.iter(|| run(black_box(&["-rI", "RAREHIT", dir_str])))
|
|
||||||
});
|
|
||||||
// Print 3 lines of context (-C 3).
|
|
||||||
group.bench_function("context", |b| {
|
|
||||||
b.iter(|| run(black_box(&["-C", "3", "RAREHIT", file])))
|
|
||||||
});
|
|
||||||
// Filename + line number with forced color (-Hn --color=always).
|
|
||||||
group.bench_function("filename_lineno_color", |b| {
|
|
||||||
b.iter(|| run(black_box(&["-Hn", "--color=always", "RAREHIT", file])))
|
|
||||||
});
|
|
||||||
// Print only the matched text (-o).
|
|
||||||
group.bench_function("only_matching", |b| {
|
|
||||||
b.iter(|| run(black_box(&["-o", "RAREHIT", file])))
|
|
||||||
});
|
|
||||||
// Invert match (-v); `worker-` is on every line, so nothing is printed
|
|
||||||
// and this measures the full inverted scan.
|
|
||||||
group.bench_function("invert_match", |b| {
|
|
||||||
b.iter(|| run(black_box(&["-v", "worker-", file])))
|
|
||||||
});
|
|
||||||
// Extended regex, case-insensitive (-Ei).
|
|
||||||
group.bench_function("extended_icase", |b| {
|
|
||||||
b.iter(|| run(black_box(&["-Ei", "rarehit", file])))
|
|
||||||
});
|
|
||||||
|
|
||||||
group.finish();
|
|
||||||
}
|
|
||||||
|
|
||||||
let _ = std::fs::remove_dir_all(Path::new(dir_str));
|
|
||||||
}
|
|
||||||
|
|
||||||
criterion_group!(benches, bench_e2e);
|
|
||||||
criterion_main!(benches);
|
|
||||||
+68
-79
@@ -3,12 +3,9 @@
|
|||||||
// For the full copyright and license information, please view the LICENSE
|
// For the full copyright and license information, please view the LICENSE
|
||||||
// file that was distributed with this source code.
|
// file that was distributed with this source code.
|
||||||
|
|
||||||
#[doc(hidden)]
|
mod context_buffer;
|
||||||
pub mod context_buffer;
|
mod line_buffer;
|
||||||
#[doc(hidden)]
|
mod matcher;
|
||||||
pub mod line_buffer;
|
|
||||||
#[doc(hidden)]
|
|
||||||
pub mod matcher;
|
|
||||||
mod output;
|
mod output;
|
||||||
mod searcher;
|
mod searcher;
|
||||||
|
|
||||||
@@ -23,8 +20,7 @@ use std::path::Path;
|
|||||||
use uucore::error::{FromIo, UResult, USimpleError};
|
use uucore::error::{FromIo, UResult, USimpleError};
|
||||||
|
|
||||||
#[derive(Clone, Copy, PartialEq, Eq)]
|
#[derive(Clone, Copy, PartialEq, Eq)]
|
||||||
#[doc(hidden)]
|
enum RegexMode {
|
||||||
pub enum RegexMode {
|
|
||||||
Fixed,
|
Fixed,
|
||||||
Basic,
|
Basic,
|
||||||
Extended,
|
Extended,
|
||||||
@@ -32,8 +28,7 @@ pub enum RegexMode {
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Clone, Copy, PartialEq, Eq)]
|
#[derive(Clone, Copy, PartialEq, Eq)]
|
||||||
#[doc(hidden)]
|
enum BinaryMode {
|
||||||
pub enum BinaryMode {
|
|
||||||
Binary,
|
Binary,
|
||||||
Text,
|
Text,
|
||||||
WithoutMatch,
|
WithoutMatch,
|
||||||
@@ -47,84 +42,79 @@ enum ColorMode {
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Clone, Copy, PartialEq, Eq)]
|
#[derive(Clone, Copy, PartialEq, Eq)]
|
||||||
#[doc(hidden)]
|
enum DirectoryMode {
|
||||||
pub enum DirectoryMode {
|
|
||||||
Read,
|
Read,
|
||||||
Skip,
|
Skip,
|
||||||
Recurse,
|
Recurse,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Clone, Copy, PartialEq, Eq)]
|
#[derive(Clone, Copy, PartialEq, Eq)]
|
||||||
#[doc(hidden)]
|
enum DeviceMode {
|
||||||
pub enum DeviceMode {
|
|
||||||
Default,
|
Default,
|
||||||
Read,
|
Read,
|
||||||
Skip,
|
Skip,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[doc(hidden)]
|
struct ColorConfig<'a> {
|
||||||
pub struct ColorConfig<'a> {
|
matched_selected: &'a str,
|
||||||
pub matched_selected: &'a str,
|
matched_context: &'a str,
|
||||||
pub matched_context: &'a str,
|
filename: &'a str,
|
||||||
pub filename: &'a str,
|
line_number: &'a str,
|
||||||
pub line_number: &'a str,
|
byte_offset: &'a str,
|
||||||
pub byte_offset: &'a str,
|
separator: &'a str,
|
||||||
pub separator: &'a str,
|
selected_line: &'a str,
|
||||||
pub selected_line: &'a str,
|
context_line: &'a str,
|
||||||
pub context_line: &'a str,
|
|
||||||
|
|
||||||
pub reverse_video: bool,
|
reverse_video: bool,
|
||||||
pub no_erase: bool,
|
no_erase: bool,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[doc(hidden)]
|
struct GlobSet {
|
||||||
pub struct GlobSet {
|
|
||||||
patterns: Vec<glob::Pattern>,
|
patterns: Vec<glob::Pattern>,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[doc(hidden)]
|
struct Config<'a> {
|
||||||
pub struct Config<'a> {
|
|
||||||
// Searcher
|
// Searcher
|
||||||
pub directory_mode: DirectoryMode,
|
directory_mode: DirectoryMode,
|
||||||
pub device_mode: DeviceMode,
|
device_mode: DeviceMode,
|
||||||
pub follow_symlinks: bool,
|
follow_symlinks: bool,
|
||||||
pub include_globs: GlobSet,
|
include_globs: GlobSet,
|
||||||
pub exclude_globs: GlobSet,
|
exclude_globs: GlobSet,
|
||||||
pub exclude_dir_globs: GlobSet,
|
exclude_dir_globs: GlobSet,
|
||||||
pub label: &'a str,
|
label: &'a str,
|
||||||
#[cfg(windows)]
|
#[cfg(windows)]
|
||||||
pub strip_cr: bool,
|
strip_cr: bool,
|
||||||
pub binary_mode: BinaryMode,
|
binary_mode: BinaryMode,
|
||||||
pub max_count: Option<u64>,
|
max_count: Option<u64>,
|
||||||
pub before_context: usize,
|
before_context: usize,
|
||||||
pub after_context: usize,
|
after_context: usize,
|
||||||
pub has_context: bool,
|
has_context: bool,
|
||||||
|
|
||||||
// Matcher
|
// Matcher
|
||||||
pub patterns: &'a [&'a str],
|
patterns: &'a [&'a str],
|
||||||
pub regex_mode: RegexMode,
|
regex_mode: RegexMode,
|
||||||
pub ignore_case: bool,
|
ignore_case: bool,
|
||||||
pub invert_match: bool,
|
invert_match: bool,
|
||||||
pub word_regexp: bool,
|
word_regexp: bool,
|
||||||
pub line_regexp: bool,
|
line_regexp: bool,
|
||||||
|
|
||||||
// Output
|
// Output
|
||||||
pub quiet: bool,
|
quiet: bool,
|
||||||
pub count: bool,
|
count: bool,
|
||||||
pub show_filename: bool,
|
show_filename: bool,
|
||||||
pub files_with_matches: bool,
|
files_with_matches: bool,
|
||||||
pub files_without_match: bool,
|
files_without_match: bool,
|
||||||
pub only_matching: bool,
|
only_matching: bool,
|
||||||
pub byte_offset: bool,
|
byte_offset: bool,
|
||||||
pub line_number: bool,
|
line_number: bool,
|
||||||
pub initial_tab: bool,
|
initial_tab: bool,
|
||||||
pub null_separator: bool,
|
null_separator: bool,
|
||||||
pub null_data: bool,
|
null_data: bool,
|
||||||
pub line_buffered: bool,
|
line_buffered: bool,
|
||||||
pub no_messages: bool,
|
no_messages: bool,
|
||||||
pub group_separator: Option<&'a str>,
|
group_separator: Option<&'a str>,
|
||||||
pub use_color: bool,
|
use_color: bool,
|
||||||
pub color_config: ColorConfig<'a>,
|
color_config: ColorConfig<'a>,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[uucore::main(no_signals)]
|
#[uucore::main(no_signals)]
|
||||||
@@ -352,6 +342,13 @@ pub fn uumain(args: impl uucore::Args) -> UResult<()> {
|
|||||||
ColorMode::Never => false,
|
ColorMode::Never => false,
|
||||||
ColorMode::Auto => std::io::stdout().is_terminal(),
|
ColorMode::Auto => std::io::stdout().is_terminal(),
|
||||||
};
|
};
|
||||||
|
// GREP_COLOR is deprecated in favour of GREP_COLORS' `mt` capability;
|
||||||
|
// GNU warns about it, but only when color output is actually produced.
|
||||||
|
if use_color && !grep_color.is_empty() {
|
||||||
|
eprintln!(
|
||||||
|
"grep: warning: GREP_COLOR='{grep_color}' is deprecated; use GREP_COLORS='mt={grep_color}'"
|
||||||
|
);
|
||||||
|
}
|
||||||
let color_config = ColorConfig::from_env(&grep_color, &grep_colors);
|
let color_config = ColorConfig::from_env(&grep_color, &grep_colors);
|
||||||
|
|
||||||
let config = Config {
|
let config = Config {
|
||||||
@@ -864,21 +861,8 @@ fn expand_num_shorthand(args: impl Iterator<Item = OsString>) -> Vec<OsString> {
|
|||||||
out
|
out
|
||||||
}
|
}
|
||||||
|
|
||||||
impl Default for GlobSet {
|
|
||||||
fn default() -> Self {
|
|
||||||
Self::new()
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
impl GlobSet {
|
impl GlobSet {
|
||||||
/// Create an empty GlobSet.
|
fn with_capacity(capacity: usize) -> Self {
|
||||||
pub fn new() -> Self {
|
|
||||||
Self {
|
|
||||||
patterns: Vec::new(),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn with_capacity(capacity: usize) -> Self {
|
|
||||||
Self {
|
Self {
|
||||||
patterns: Vec::with_capacity(capacity),
|
patterns: Vec::with_capacity(capacity),
|
||||||
}
|
}
|
||||||
@@ -926,6 +910,11 @@ impl<'a> ColorConfig<'a> {
|
|||||||
for item in grep_colors.split(':') {
|
for item in grep_colors.split(':') {
|
||||||
if let Some((key, value)) = item.split_once('=') {
|
if let Some((key, value)) = item.split_once('=') {
|
||||||
match key {
|
match key {
|
||||||
|
// `mt` sets both the selected- and context-match colors.
|
||||||
|
"mt" => {
|
||||||
|
config.matched_selected = value;
|
||||||
|
config.matched_context = value;
|
||||||
|
}
|
||||||
"ms" => config.matched_selected = value,
|
"ms" => config.matched_selected = value,
|
||||||
"mc" => config.matched_context = value,
|
"mc" => config.matched_context = value,
|
||||||
"fn" => config.filename = value,
|
"fn" => config.filename = value,
|
||||||
|
|||||||
+279
-76
@@ -5,10 +5,14 @@
|
|||||||
|
|
||||||
use crate::{Config, RegexMode};
|
use crate::{Config, RegexMode};
|
||||||
use onig::{
|
use onig::{
|
||||||
EncodedBytes, Regex, RegexOptions, Region, SearchOptions, Syntax, SyntaxBehavior,
|
EncodedBytes, Error, MatchParam, Regex, RegexOptions, Region, SearchOptions, Syntax,
|
||||||
SyntaxOperator,
|
SyntaxBehavior, SyntaxOperator,
|
||||||
};
|
};
|
||||||
use onig_sys::{OnigEncCtype_ONIGENC_CTYPE_WORD, OnigEncodingUTF8};
|
use onig_sys::{
|
||||||
|
ONIGERR_EMPTY_RANGE_IN_CHAR_CLASS, ONIGERR_INVALID_BACKREF, ONIGERR_RETRY_LIMIT_IN_MATCH_OVER,
|
||||||
|
ONIGERR_RETRY_LIMIT_IN_SEARCH_OVER, OnigEncCtype_ONIGENC_CTYPE_WORD, OnigEncodingUTF8,
|
||||||
|
};
|
||||||
|
use std::io;
|
||||||
use uucore::error::{UResult, USimpleError};
|
use uucore::error::{UResult, USimpleError};
|
||||||
|
|
||||||
pub struct Matcher<'a> {
|
pub struct Matcher<'a> {
|
||||||
@@ -26,26 +30,30 @@ impl<'a> Matcher<'a> {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Decide whether `line` matches and return the positions to highlight.
|
/// Decide whether `line` matches and return the positions to highlight.
|
||||||
pub fn match_line(&self, line: &[u8]) -> Option<Vec<(usize, usize)>> {
|
///
|
||||||
|
/// Returns an error if the regex engine bails out (e.g. it exceeds its
|
||||||
|
/// backtracking retry limit on a pathological pattern); the caller turns
|
||||||
|
/// that into a GNU-style diagnostic and exit code 2 rather than aborting.
|
||||||
|
pub fn match_line(&self, line: &[u8]) -> io::Result<Option<Vec<(usize, usize)>>> {
|
||||||
let mut any_seen = false;
|
let mut any_seen = false;
|
||||||
let positions: Vec<_> = MatchIter::new(&self.patterns, line)
|
let mut positions = Vec::new();
|
||||||
.filter(|&(start, end)| {
|
let mut iter = MatchIter::new(&self.patterns, line).map_err(match_error)?;
|
||||||
any_seen = true;
|
while let Some((start, end)) = iter.next_match().map_err(match_error)? {
|
||||||
// Drop zero-length matches from the output.
|
any_seen = true;
|
||||||
if start == end {
|
// Drop zero-length matches from the output.
|
||||||
return false;
|
if start == end {
|
||||||
}
|
continue;
|
||||||
// Drop matches that don't span the whole line if `-x` was requested.
|
}
|
||||||
if self.config.line_regexp && !(start == 0 && end == line.len()) {
|
// Drop matches that don't span the whole line if `-x` was requested.
|
||||||
return false;
|
if self.config.line_regexp && !(start == 0 && end == line.len()) {
|
||||||
}
|
continue;
|
||||||
// Drop matches that aren't word matches if `-w` was requested.
|
}
|
||||||
if self.config.word_regexp && !Self::is_word_match(line, start, end) {
|
// Drop matches that aren't word matches if `-w` was requested.
|
||||||
return false;
|
if self.config.word_regexp && !Self::is_word_match(line, start, end) {
|
||||||
}
|
continue;
|
||||||
true
|
}
|
||||||
})
|
positions.push((start, end));
|
||||||
.collect();
|
}
|
||||||
|
|
||||||
let raw_matched = if self.config.line_regexp || self.config.word_regexp {
|
let raw_matched = if self.config.line_regexp || self.config.word_regexp {
|
||||||
// -w / -x are authoritative once positions are filtered.
|
// -w / -x are authoritative once positions are filtered.
|
||||||
@@ -54,23 +62,25 @@ impl<'a> Matcher<'a> {
|
|||||||
any_seen
|
any_seen
|
||||||
};
|
};
|
||||||
|
|
||||||
if raw_matched != self.config.invert_match {
|
Ok((raw_matched != self.config.invert_match).then_some(positions))
|
||||||
Some(positions)
|
|
||||||
} else {
|
|
||||||
None
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Cheap match check that doesn't enumerate positions.
|
/// Cheap match check that doesn't enumerate positions.
|
||||||
pub fn is_match(&self, line: &[u8]) -> Option<Vec<(usize, usize)>> {
|
pub fn is_match(&self, line: &[u8]) -> io::Result<Option<Vec<(usize, usize)>>> {
|
||||||
// `-w` / `-x` need positions to filter, so we fall back to `match_line`.
|
// `-w` / `-x` need positions to filter, so we fall back to `match_line`.
|
||||||
let matched = if self.config.line_regexp || self.config.word_regexp {
|
let matched = if self.config.line_regexp || self.config.word_regexp {
|
||||||
self.match_line(line).is_some()
|
self.match_line(line)?.is_some()
|
||||||
} else {
|
} else {
|
||||||
let raw_matched = self.patterns.iter().any(|p| p.is_match(line));
|
let mut raw_matched = false;
|
||||||
|
for p in &self.patterns {
|
||||||
|
if p.is_match(line).map_err(match_error)? {
|
||||||
|
raw_matched = true;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
raw_matched != self.config.invert_match
|
raw_matched != self.config.invert_match
|
||||||
};
|
};
|
||||||
matched.then(Vec::new)
|
Ok(matched.then(Vec::new))
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Word-boundary check `-w`.
|
/// Word-boundary check `-w`.
|
||||||
@@ -112,35 +122,31 @@ struct MatchIter<'a> {
|
|||||||
}
|
}
|
||||||
|
|
||||||
impl<'a> MatchIter<'a> {
|
impl<'a> MatchIter<'a> {
|
||||||
fn new(patterns: &'a [CompiledPattern], line: &'a [u8]) -> Self {
|
fn new(patterns: &'a [CompiledPattern], line: &'a [u8]) -> Result<Self, Error> {
|
||||||
Self {
|
let mut cursors = Vec::with_capacity(patterns.len());
|
||||||
cursors: patterns
|
for pattern in patterns {
|
||||||
.iter()
|
let mut c = Cursor {
|
||||||
.map(|pattern| {
|
pattern,
|
||||||
let mut c = Cursor {
|
line,
|
||||||
pattern,
|
offset: 0,
|
||||||
line,
|
pending: None,
|
||||||
offset: 0,
|
};
|
||||||
pending: None,
|
c.refill()?;
|
||||||
};
|
cursors.push(c);
|
||||||
c.refill();
|
|
||||||
c
|
|
||||||
})
|
|
||||||
.collect(),
|
|
||||||
last_end: 0,
|
|
||||||
}
|
}
|
||||||
|
Ok(Self {
|
||||||
|
cursors,
|
||||||
|
last_end: 0,
|
||||||
|
})
|
||||||
}
|
}
|
||||||
}
|
|
||||||
|
|
||||||
impl<'a> Iterator for MatchIter<'a> {
|
/// Yield the next match across all patterns, or `None` when exhausted.
|
||||||
type Item = (usize, usize);
|
fn next_match(&mut self) -> Result<Option<(usize, usize)>, Error> {
|
||||||
|
|
||||||
fn next(&mut self) -> Option<Self::Item> {
|
|
||||||
// Discard stale pendings that fall before the last emit.
|
// Discard stale pendings that fall before the last emit.
|
||||||
for cursor in &mut self.cursors {
|
for cursor in &mut self.cursors {
|
||||||
if matches!(cursor.pending, Some((s, _)) if s < self.last_end) {
|
if matches!(cursor.pending, Some((s, _)) if s < self.last_end) {
|
||||||
cursor.offset = self.last_end;
|
cursor.offset = self.last_end;
|
||||||
cursor.refill();
|
cursor.refill()?;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -153,12 +159,15 @@ impl<'a> Iterator for MatchIter<'a> {
|
|||||||
.enumerate()
|
.enumerate()
|
||||||
.filter_map(|(i, c)| c.pending.map(|p| (i, p)))
|
.filter_map(|(i, c)| c.pending.map(|p| (i, p)))
|
||||||
.min_by_key(|&(_, (s, e))| (s, std::cmp::Reverse(e)))
|
.min_by_key(|&(_, (s, e))| (s, std::cmp::Reverse(e)))
|
||||||
.map(|(i, _)| i)?;
|
.map(|(i, _)| i);
|
||||||
|
let Some(best_idx) = best_idx else {
|
||||||
|
return Ok(None);
|
||||||
|
};
|
||||||
|
|
||||||
let (start, end) = self.cursors[best_idx].pending.unwrap();
|
let (start, end) = self.cursors[best_idx].pending.unwrap();
|
||||||
self.cursors[best_idx].refill();
|
self.cursors[best_idx].refill()?;
|
||||||
self.last_end = end;
|
self.last_end = end;
|
||||||
Some((start, end))
|
Ok(Some((start, end)))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -173,24 +182,25 @@ struct Cursor<'a> {
|
|||||||
}
|
}
|
||||||
|
|
||||||
impl Cursor<'_> {
|
impl Cursor<'_> {
|
||||||
fn refill(&mut self) {
|
fn refill(&mut self) -> Result<(), Error> {
|
||||||
if self.offset >= self.line.len() {
|
if self.offset >= self.line.len() {
|
||||||
self.pending = None;
|
self.pending = None;
|
||||||
return;
|
return Ok(());
|
||||||
}
|
}
|
||||||
let Some((start, leftmost_end)) = self.pattern.search_leftmost(self.line, self.offset)
|
let Some((start, leftmost_end)) = self.pattern.search_leftmost(self.line, self.offset)?
|
||||||
else {
|
else {
|
||||||
self.pending = None;
|
self.pending = None;
|
||||||
return;
|
return Ok(());
|
||||||
};
|
};
|
||||||
let end = self
|
let end = self
|
||||||
.pattern
|
.pattern
|
||||||
.longest_end_at(self.line, start)
|
.longest_end_at(self.line, start)?
|
||||||
.unwrap_or(leftmost_end);
|
.unwrap_or(leftmost_end);
|
||||||
// Advance the next search past the match we just found.
|
// Advance the next search past the match we just found.
|
||||||
// Zero-length matches need a +1 nudge to avoid spinning forever.
|
// Zero-length matches need a +1 nudge to avoid spinning forever.
|
||||||
self.offset = end.max(start + 1);
|
self.offset = end.max(start + 1);
|
||||||
self.pending = Some((start, end));
|
self.pending = Some((start, end));
|
||||||
|
Ok(())
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -205,6 +215,12 @@ struct CompiledPattern {
|
|||||||
|
|
||||||
impl CompiledPattern {
|
impl CompiledPattern {
|
||||||
fn compile(pattern: &str, config: &Config) -> UResult<Self> {
|
fn compile(pattern: &str, config: &Config) -> UResult<Self> {
|
||||||
|
// GNU grep rejects the confusing `[:name:]` bracket form (a misspelled
|
||||||
|
// `[[:name:]]`) in basic/extended modes; oniguruma accepts it silently.
|
||||||
|
if matches!(config.regex_mode, RegexMode::Basic | RegexMode::Extended) {
|
||||||
|
check_confusing_bracket(pattern)?;
|
||||||
|
}
|
||||||
|
|
||||||
let mut syntax = *match config.regex_mode {
|
let mut syntax = *match config.regex_mode {
|
||||||
RegexMode::Fixed => Syntax::asis(),
|
RegexMode::Fixed => Syntax::asis(),
|
||||||
RegexMode::Basic => Syntax::grep(),
|
RegexMode::Basic => Syntax::grep(),
|
||||||
@@ -232,17 +248,29 @@ impl CompiledPattern {
|
|||||||
options |= RegexOptions::REGEX_OPTION_IGNORECASE;
|
options |= RegexOptions::REGEX_OPTION_IGNORECASE;
|
||||||
}
|
}
|
||||||
|
|
||||||
fn compile_with(pattern: &str, syntax: &Syntax, options: RegexOptions) -> UResult<Regex> {
|
let mode = config.regex_mode;
|
||||||
|
fn compile_with(
|
||||||
|
pattern: &str,
|
||||||
|
syntax: &Syntax,
|
||||||
|
options: RegexOptions,
|
||||||
|
mode: RegexMode,
|
||||||
|
) -> UResult<Regex> {
|
||||||
Regex::with_options_and_encoding(pattern, options, syntax).map_err(|err| {
|
Regex::with_options_and_encoding(pattern, options, syntax).map_err(|err| {
|
||||||
USimpleError::new(2, format!("invalid pattern \"{pattern}\": {err}"))
|
// Prefer GNU grep's wording for the errors it has a dedicated
|
||||||
|
// message for; fall back to oniguruma's text otherwise.
|
||||||
|
match gnu_error_message(err.code(), mode) {
|
||||||
|
Some(msg) => USimpleError::new(2, msg.to_string()),
|
||||||
|
None => USimpleError::new(2, format!("invalid pattern \"{pattern}\": {err}")),
|
||||||
|
}
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
let leftmost = compile_with(pattern, &syntax, options)?;
|
let leftmost = compile_with(pattern, &syntax, options, mode)?;
|
||||||
let longest_anchored = compile_with(
|
let longest_anchored = compile_with(
|
||||||
pattern,
|
pattern,
|
||||||
&syntax,
|
&syntax,
|
||||||
options | RegexOptions::REGEX_OPTION_FIND_LONGEST,
|
options | RegexOptions::REGEX_OPTION_FIND_LONGEST,
|
||||||
|
mode,
|
||||||
)?;
|
)?;
|
||||||
Ok(Self {
|
Ok(Self {
|
||||||
leftmost,
|
leftmost,
|
||||||
@@ -251,41 +279,216 @@ impl CompiledPattern {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Find the leftmost match starting at or after `offset`.
|
/// Find the leftmost match starting at or after `offset`.
|
||||||
fn search_leftmost(&self, line: &[u8], offset: usize) -> Option<(usize, usize)> {
|
fn search_leftmost(&self, line: &[u8], offset: usize) -> Result<Option<(usize, usize)>, Error> {
|
||||||
let mut region = Region::new();
|
let mut region = Region::new();
|
||||||
self.leftmost.search_with_encoding(
|
let found = self.leftmost.search_with_param(
|
||||||
EncodedBytes::from_parts(line, &raw mut OnigEncodingUTF8),
|
EncodedBytes::from_parts(line, &raw mut OnigEncodingUTF8),
|
||||||
offset,
|
offset,
|
||||||
line.len(),
|
line.len(),
|
||||||
SearchOptions::SEARCH_OPTION_NONE,
|
SearchOptions::SEARCH_OPTION_NONE,
|
||||||
Some(&mut region),
|
Some(&mut region),
|
||||||
|
MatchParam::default(),
|
||||||
)?;
|
)?;
|
||||||
region.pos(0)
|
Ok(found.and_then(|_| region.pos(0)))
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Given a known leftmost start `start`, return the longest extent
|
/// Given a known leftmost start `start`, return the longest extent
|
||||||
/// of a match anchored exactly there = POSIX leftmost-longest end.
|
/// of a match anchored exactly there = POSIX leftmost-longest end.
|
||||||
fn longest_end_at(&self, line: &[u8], start: usize) -> Option<usize> {
|
fn longest_end_at(&self, line: &[u8], start: usize) -> Result<Option<usize>, Error> {
|
||||||
let mut region = Region::new();
|
let mut region = Region::new();
|
||||||
self.longest_anchored.match_with_encoding(
|
self.longest_anchored.match_with_param(
|
||||||
EncodedBytes::from_parts(line, &raw mut OnigEncodingUTF8),
|
EncodedBytes::from_parts(line, &raw mut OnigEncodingUTF8),
|
||||||
start,
|
start,
|
||||||
SearchOptions::SEARCH_OPTION_NONE,
|
SearchOptions::SEARCH_OPTION_NONE,
|
||||||
Some(&mut region),
|
Some(&mut region),
|
||||||
);
|
MatchParam::default(),
|
||||||
region.pos(0).map(|(_, end)| end)
|
)?;
|
||||||
|
Ok(region.pos(0).map(|(_, end)| end))
|
||||||
}
|
}
|
||||||
|
|
||||||
/// True if any match exists in `line` (including zero-length).
|
/// True if any match exists in `line` (including zero-length).
|
||||||
fn is_match(&self, line: &[u8]) -> bool {
|
fn is_match(&self, line: &[u8]) -> Result<bool, Error> {
|
||||||
self.leftmost
|
Ok(self
|
||||||
.search_with_encoding(
|
.leftmost
|
||||||
|
.search_with_param(
|
||||||
EncodedBytes::from_parts(line, &raw mut OnigEncodingUTF8),
|
EncodedBytes::from_parts(line, &raw mut OnigEncodingUTF8),
|
||||||
0,
|
0,
|
||||||
line.len(),
|
line.len(),
|
||||||
SearchOptions::SEARCH_OPTION_NONE,
|
SearchOptions::SEARCH_OPTION_NONE,
|
||||||
None,
|
None,
|
||||||
)
|
MatchParam::default(),
|
||||||
.is_some()
|
)?
|
||||||
|
.is_some())
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Convert a regex-engine match-time error into an I/O error carrying GNU
|
||||||
|
/// grep's wording. The only error we expect in practice is the backtracking
|
||||||
|
/// retry limit being exceeded on a pathological pattern; GNU reports this as
|
||||||
|
/// `exceeded PCRE's backtracking limit` and exits 2 instead of aborting.
|
||||||
|
fn match_error(err: Error) -> io::Error {
|
||||||
|
let message = if matches!(
|
||||||
|
err.code(),
|
||||||
|
ONIGERR_RETRY_LIMIT_IN_MATCH_OVER | ONIGERR_RETRY_LIMIT_IN_SEARCH_OVER
|
||||||
|
) {
|
||||||
|
"exceeded PCRE's backtracking limit".to_string()
|
||||||
|
} else {
|
||||||
|
err.description().to_string()
|
||||||
|
};
|
||||||
|
io::Error::other(message)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Map an oniguruma compile-error code to GNU grep's wording for the same
|
||||||
|
/// condition, when one exists. GNU emits a bare POSIX-style diagnostic (e.g.
|
||||||
|
/// `Invalid range end`) rather than oniguruma's phrasing, so translating keeps
|
||||||
|
/// us byte-compatible. Returns `None` for errors with no GNU equivalent, where
|
||||||
|
/// the caller falls back to oniguruma's own message.
|
||||||
|
fn gnu_error_message(code: i32, mode: RegexMode) -> Option<&'static str> {
|
||||||
|
match code {
|
||||||
|
// e.g. `[b-a]`: a range whose end precedes its start.
|
||||||
|
ONIGERR_EMPTY_RANGE_IN_CHAR_CLASS => Some("Invalid range end"),
|
||||||
|
// e.g. `(.)\2`: a back-reference to a group that does not exist. GNU
|
||||||
|
// (via PCRE2) and gnulib's regex word this differently.
|
||||||
|
ONIGERR_INVALID_BACKREF if mode == RegexMode::Perl => {
|
||||||
|
Some("reference to non-existent subpattern")
|
||||||
|
}
|
||||||
|
ONIGERR_INVALID_BACKREF => Some("Invalid back reference"),
|
||||||
|
_ => None,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Reject the confusing `[:name:]` bracket form the way GNU grep does.
|
||||||
|
///
|
||||||
|
/// A bracket expression like `[:space:]` is almost always a misspelled
|
||||||
|
/// `[[:space:]]`; GNU grep flags it with a dedicated diagnostic and exits 2,
|
||||||
|
/// whereas oniguruma silently treats it as the set `{':','s','p',…}`. This
|
||||||
|
/// scans the pattern for that form and returns the same error.
|
||||||
|
fn check_confusing_bracket(pattern: &str) -> UResult<()> {
|
||||||
|
let bytes = pattern.as_bytes();
|
||||||
|
let mut i = 0;
|
||||||
|
while i < bytes.len() {
|
||||||
|
match bytes[i] {
|
||||||
|
// Outside a bracket a backslash escapes the next character, so
|
||||||
|
// `\[` does not open a bracket expression.
|
||||||
|
b'\\' => i += 2,
|
||||||
|
b'[' => {
|
||||||
|
i += 1;
|
||||||
|
if bracket_warns(bytes, &mut i) {
|
||||||
|
return Err(USimpleError::new(
|
||||||
|
2,
|
||||||
|
"character class syntax is [[:space:]], not [:space:]".to_string(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
_ => i += 1,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Consume a single bracket expression starting just past its opening `[` and
|
||||||
|
/// report whether GNU grep's colon warning fires for it.
|
||||||
|
///
|
||||||
|
/// This is a faithful port of the `colon_warning_state` logic in GNU grep's
|
||||||
|
/// `parse_bracket_exp` (gnulib `dfa.c`). The state is a bitmask:
|
||||||
|
/// bit 0 — first character is a colon
|
||||||
|
/// bit 1 — last character is a colon
|
||||||
|
/// bit 2 — includes some other (non-colon) character
|
||||||
|
/// bit 3 — includes a range, char/equivalence class, or collating element
|
||||||
|
/// The warning fires exactly when the state ends equal to `7` (bits 0–2 set,
|
||||||
|
/// bit 3 clear). On the way it advances `i` past the closing `]`.
|
||||||
|
fn bracket_warns(bytes: &[u8], i: &mut usize) -> bool {
|
||||||
|
fn fetch(bytes: &[u8], i: &mut usize) -> Option<u8> {
|
||||||
|
let b = bytes.get(*i).copied();
|
||||||
|
if b.is_some() {
|
||||||
|
*i += 1;
|
||||||
|
}
|
||||||
|
b
|
||||||
|
}
|
||||||
|
|
||||||
|
let Some(first) = fetch(bytes, i) else {
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
let mut c = first;
|
||||||
|
if c == b'^' {
|
||||||
|
match fetch(bytes, i) {
|
||||||
|
Some(x) => c = x,
|
||||||
|
None => return false,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let mut state: u8 = u8::from(c == b':');
|
||||||
|
|
||||||
|
'scan: loop {
|
||||||
|
state &= !2;
|
||||||
|
let mut c1: Option<u8> = None;
|
||||||
|
|
||||||
|
if c == b'[' {
|
||||||
|
let Some(nc1) = fetch(bytes, i) else {
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
// `[:`, `[.` and `[=` introduce a class / collating / equivalence
|
||||||
|
// element; consume it whole and mark bit 3.
|
||||||
|
if nc1 == b':' || nc1 == b'.' || nc1 == b'=' {
|
||||||
|
loop {
|
||||||
|
match fetch(bytes, i) {
|
||||||
|
None => break,
|
||||||
|
Some(cc) if cc == nc1 && bytes.get(*i).copied() == Some(b']') => break,
|
||||||
|
Some(_) => {}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if fetch(bytes, i).is_none() {
|
||||||
|
return false; // consumes the `]`
|
||||||
|
}
|
||||||
|
state |= 8;
|
||||||
|
match fetch(bytes, i) {
|
||||||
|
Some(b']') => break 'scan,
|
||||||
|
Some(x) => {
|
||||||
|
c = x;
|
||||||
|
continue 'scan;
|
||||||
|
}
|
||||||
|
None => return false,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Otherwise `[` is an ordinary character; `nc1` is the lookahead.
|
||||||
|
c1 = Some(nc1);
|
||||||
|
}
|
||||||
|
|
||||||
|
if c1.is_none() {
|
||||||
|
c1 = fetch(bytes, i);
|
||||||
|
}
|
||||||
|
|
||||||
|
if c1 == Some(b'-') {
|
||||||
|
let Some(mut c2) = fetch(bytes, i) else {
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
if c2 == b'[' && bytes.get(*i).copied() == Some(b'.') {
|
||||||
|
c2 = b']';
|
||||||
|
}
|
||||||
|
if c2 == b']' {
|
||||||
|
// `[x-]`: the hyphen is a literal; put the `]` back so the
|
||||||
|
// loop terminator sees it next.
|
||||||
|
*i -= 1;
|
||||||
|
} else {
|
||||||
|
state |= 8;
|
||||||
|
match fetch(bytes, i) {
|
||||||
|
Some(b']') => break 'scan,
|
||||||
|
Some(x) => {
|
||||||
|
c = x;
|
||||||
|
continue 'scan;
|
||||||
|
}
|
||||||
|
None => return false,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
state |= if c == b':' { 2 } else { 4 };
|
||||||
|
|
||||||
|
match c1 {
|
||||||
|
Some(b']') => break 'scan,
|
||||||
|
Some(x) => c = x,
|
||||||
|
None => return false,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
state == 7
|
||||||
|
}
|
||||||
|
|||||||
+3
-3
@@ -281,7 +281,7 @@ impl<'a> Searcher<'a> {
|
|||||||
return Ok(false);
|
return Ok(false);
|
||||||
}
|
}
|
||||||
|
|
||||||
if let Some(positions) = self.session_match_line(line) {
|
if let Some(positions) = self.session_match_line(line)? {
|
||||||
// TODO: GNU grep respects LANG. Here, I'm always checking for valid UTF-8.
|
// TODO: GNU grep respects LANG. Here, I'm always checking for valid UTF-8.
|
||||||
if !self.session_mark_binary_if(|| std::str::from_utf8(line).is_err()) {
|
if !self.session_mark_binary_if(|| std::str::from_utf8(line).is_err()) {
|
||||||
return Ok(false);
|
return Ok(false);
|
||||||
@@ -321,9 +321,9 @@ impl<'a> Searcher<'a> {
|
|||||||
self.config.binary_mode != BinaryMode::WithoutMatch
|
self.config.binary_mode != BinaryMode::WithoutMatch
|
||||||
}
|
}
|
||||||
|
|
||||||
fn session_match_line(&self, line: &[u8]) -> Option<Vec<(usize, usize)>> {
|
fn session_match_line(&self, line: &[u8]) -> io::Result<Option<Vec<(usize, usize)>>> {
|
||||||
if !self.session_can_match() {
|
if !self.session_can_match() {
|
||||||
None
|
Ok(None)
|
||||||
} else if self.session_needs_match_positions() {
|
} else if self.session_needs_match_positions() {
|
||||||
self.matcher.match_line(line)
|
self.matcher.match_line(line)
|
||||||
} else {
|
} else {
|
||||||
|
|||||||
@@ -126,6 +126,142 @@ fn ere_invalid_pattern_is_error() {
|
|||||||
.stderr_contains("invalid pattern");
|
.stderr_contains("invalid pattern");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn confusing_bracket_class_is_error() {
|
||||||
|
// GNU grep rejects the misspelled `[:name:]` form (meant to be
|
||||||
|
// `[[:name:]]`) with a dedicated diagnostic and exit code 2.
|
||||||
|
// No piped input: the pattern is rejected at compile time, before stdin is
|
||||||
|
// read, so feeding stdin would race with the child exiting (broken pipe).
|
||||||
|
for pattern in ["[:space:]", "[:digit:]", "[^:space:]", "x[:space:]y"] {
|
||||||
|
let (_s, mut c) = ucmd();
|
||||||
|
c.args(&[pattern])
|
||||||
|
.fails_with_code(2)
|
||||||
|
.stderr_is("grep: character class syntax is [[:space:]], not [:space:]\n");
|
||||||
|
}
|
||||||
|
|
||||||
|
// The same diagnostic applies in extended mode.
|
||||||
|
let (_s, mut c) = ucmd();
|
||||||
|
c.args(&["-E", "[:space:]"])
|
||||||
|
.fails_with_code(2)
|
||||||
|
.stderr_is("grep: character class syntax is [[:space:]], not [:space:]\n");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn lookalike_brackets_are_not_confusing() {
|
||||||
|
// Patterns that are NOT the confusing `[:name:]` form must compile
|
||||||
|
// normally (no diagnostic). A proper class, a colon set, a range, a
|
||||||
|
// trailing colon set, and `-F` literal text all stay valid.
|
||||||
|
for pattern in [
|
||||||
|
"[[:space:]]",
|
||||||
|
"[::]",
|
||||||
|
"[:space]",
|
||||||
|
"[:spac-e:]",
|
||||||
|
"[a:space:]",
|
||||||
|
] {
|
||||||
|
let (_s, mut c) = ucmd();
|
||||||
|
c.args(&[pattern])
|
||||||
|
.pipe_in("z\n")
|
||||||
|
.fails_with_code(1)
|
||||||
|
.no_output();
|
||||||
|
}
|
||||||
|
|
||||||
|
// `\[` does not open a bracket expression.
|
||||||
|
let (_s, mut c) = ucmd();
|
||||||
|
c.args(&["\\[:space:]"])
|
||||||
|
.pipe_in("z\n")
|
||||||
|
.fails_with_code(1)
|
||||||
|
.no_output();
|
||||||
|
|
||||||
|
// `-F` treats the text literally, so no diagnostic.
|
||||||
|
let (_s, mut c) = ucmd();
|
||||||
|
c.args(&["-F", "[:space:]"])
|
||||||
|
.pipe_in("x\n")
|
||||||
|
.fails_with_code(1)
|
||||||
|
.no_output();
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn reversed_range_uses_gnu_wording() {
|
||||||
|
// A range like `[b-a]` is an error; GNU prints the bare POSIX diagnostic
|
||||||
|
// "Invalid range end" (not oniguruma's phrasing) and exits 2.
|
||||||
|
// No piped input: the pattern is rejected before stdin is read.
|
||||||
|
for args in [&["[b-a]"][..], &["-E", "[b-a]"][..]] {
|
||||||
|
let (_s, mut c) = ucmd();
|
||||||
|
c.args(args)
|
||||||
|
.fails_with_code(2)
|
||||||
|
.stderr_is("grep: Invalid range end\n");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn pcre_backtracking_limit_does_not_abort() {
|
||||||
|
// A pathological PCRE pattern can exceed oniguruma's retry limit. GNU
|
||||||
|
// grep reports this and exits 2 (it must not crash); stdout stays empty.
|
||||||
|
let (_s, mut c) = ucmd();
|
||||||
|
c.args(&["-P", "((a+)*)+$"])
|
||||||
|
.pipe_in("aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaab\n")
|
||||||
|
.fails_with_code(2)
|
||||||
|
.stdout_is("")
|
||||||
|
.stderr_contains("backtracking limit");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn invalid_backreference_uses_gnu_wording() {
|
||||||
|
// A back-reference to a non-existent group is worded differently by GNU
|
||||||
|
// depending on the engine: PCRE (-P) vs gnulib regex (BRE/ERE).
|
||||||
|
// No piped input: these patterns are rejected before stdin is read.
|
||||||
|
let (_s, mut c) = ucmd();
|
||||||
|
c.args(&["-P", r"(.)\2"])
|
||||||
|
.fails_with_code(2)
|
||||||
|
.stderr_is("grep: reference to non-existent subpattern\n");
|
||||||
|
|
||||||
|
for args in [&["-E", r"(.)\2"][..], &[r"\(.\)\2"][..]] {
|
||||||
|
let (_s, mut c) = ucmd();
|
||||||
|
c.args(args)
|
||||||
|
.fails_with_code(2)
|
||||||
|
.stderr_is("grep: Invalid back reference\n");
|
||||||
|
}
|
||||||
|
|
||||||
|
// A valid back-reference with -Pw / -Px must still match.
|
||||||
|
for flag in ["-Pw", "-Px"] {
|
||||||
|
let (_s, mut c) = ucmd();
|
||||||
|
c.args(&[flag, r"(.)\1"])
|
||||||
|
.pipe_in("aa\n")
|
||||||
|
.succeeds()
|
||||||
|
.stdout_is("aa\n");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn grep_colors_mt_and_grep_color_deprecation() {
|
||||||
|
// GREP_COLORS `mt` sets the match color (both selected and context).
|
||||||
|
let (_s, mut c) = ucmd();
|
||||||
|
c.args(&["--color=always", "."])
|
||||||
|
.env("GREP_COLORS", "mt=36")
|
||||||
|
.pipe_in("x\n")
|
||||||
|
.succeeds()
|
||||||
|
.stdout_is("\u{1b}[36m\u{1b}[Kx\u{1b}[m\u{1b}[K\n");
|
||||||
|
|
||||||
|
// GREP_COLOR is deprecated: it still sets the match color, but emits a
|
||||||
|
// warning when color is actually produced.
|
||||||
|
let (_s, mut c) = ucmd();
|
||||||
|
c.args(&["--color=always", "."])
|
||||||
|
.env("GREP_COLOR", "36")
|
||||||
|
.pipe_in("x\n")
|
||||||
|
.succeeds()
|
||||||
|
.stdout_is("\u{1b}[36m\u{1b}[Kx\u{1b}[m\u{1b}[K\n")
|
||||||
|
.stderr_is("grep: warning: GREP_COLOR='36' is deprecated; use GREP_COLORS='mt=36'\n");
|
||||||
|
|
||||||
|
// No warning when color output is disabled.
|
||||||
|
let (_s, mut c) = ucmd();
|
||||||
|
c.args(&["--color=never", "."])
|
||||||
|
.env("GREP_COLOR", "36")
|
||||||
|
.pipe_in("x\n")
|
||||||
|
.succeeds()
|
||||||
|
.stdout_is("x\n")
|
||||||
|
.no_stderr();
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn fixed_string_is_literal() {
|
fn fixed_string_is_literal() {
|
||||||
// Metacharacters are not interpreted.
|
// Metacharacters are not interpreted.
|
||||||
|
|||||||
Reference in New Issue
Block a user