From 6cc245db01f5b1f32da336c6aa8826081650a1a8 Mon Sep 17 00:00:00 2001 From: Sylvestre Ledru Date: Wed, 10 Sep 2025 10:41:45 +0200 Subject: [PATCH] Fix the doc generation and remove uuhelp_parser --- Cargo.lock | 7 +- Cargo.toml | 5 +- src/bin/uudoc.rs | 66 +++++++--- src/uucore_procs/Cargo.toml | 2 - src/uucore_procs/src/lib.rs | 130 +------------------ src/uuhelp_parser/Cargo.toml | 9 -- src/uuhelp_parser/LICENSE | 1 - src/uuhelp_parser/src/lib.rs | 236 ----------------------------------- util/publish.sh | 3 +- util/update-version.sh | 2 - 10 files changed, 53 insertions(+), 408 deletions(-) delete mode 100644 src/uuhelp_parser/Cargo.toml delete mode 120000 src/uuhelp_parser/LICENSE delete mode 100644 src/uuhelp_parser/src/lib.rs diff --git a/Cargo.lock b/Cargo.lock index df01ab543..9e378ebb8 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -497,6 +497,7 @@ dependencies = [ "clap_mangen", "ctor", "filetime", + "fluent-syntax", "glob", "hex-literal", "libc", @@ -620,7 +621,6 @@ dependencies = [ "uu_whoami", "uu_yes", "uucore", - "uuhelp_parser", "uutests", "walkdir", "xattr", @@ -4130,13 +4130,8 @@ version = "0.2.2" dependencies = [ "proc-macro2", "quote", - "uuhelp_parser", ] -[[package]] -name = "uuhelp_parser" -version = "0.2.2" - [[package]] name = "uuid" version = "1.17.0" diff --git a/Cargo.toml b/Cargo.toml index 405368c0b..b76faebf9 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -33,7 +33,7 @@ expensive_tests = [] # "test_risky_names" == enable tests that create problematic file names (would make a network share inaccessible to Windows, breaks SVN on Mac OS, etc.) test_risky_names = [] # * only build `uudoc` when `--feature uudoc` is activated -uudoc = ["zip", "dep:uuhelp_parser"] +uudoc = ["zip", "dep:fluent-syntax"] ## features ## Optional feature for stdbuf # "feat_external_libstdbuf" == use an external libstdbuf.so for stdbuf instead of embedding it @@ -282,7 +282,6 @@ members = [ "src/uu/stdbuf/src/libstdbuf", "src/uucore", "src/uucore_procs", - "src/uuhelp_parser", "tests/benches/factor", "tests/uutests", # "fuzz", # TODO @@ -415,8 +414,8 @@ phf.workspace = true selinux = { workspace = true, optional = true } textwrap.workspace = true zip = { workspace = true, optional = true } +fluent-syntax = { workspace = true, optional = true } -uuhelp_parser = { optional = true, version = "0.2.2", path = "src/uuhelp_parser" } # * uutils uu_test = { optional = true, version = "0.2.2", package = "uu_test", path = "src/uu/test" } diff --git a/src/bin/uudoc.rs b/src/bin/uudoc.rs index 6a215a4ad..8212a8a0b 100644 --- a/src/bin/uudoc.rs +++ b/src/bin/uudoc.rs @@ -5,6 +5,8 @@ // spell-checker:ignore tldr uuhelp use clap::Command; +use fluent_syntax::ast::{Entry, Message, Pattern}; +use fluent_syntax::parser; use std::collections::HashMap; use std::ffi::OsString; use std::fs::File; @@ -140,7 +142,7 @@ fn main() -> io::Result<()> { } let p = format!("docs/src/utils/{name}.md"); - let markdown = File::open(format!("src/uu/{name}/{name}.md")) + let fluent = File::open(format!("src/uu/{name}/locales/en-US.ftl")) .and_then(|mut f: File| { let mut s = String::new(); f.read_to_string(&mut s)?; @@ -155,7 +157,7 @@ fn main() -> io::Result<()> { name, tldr_zip: &mut tldr_zip, utils_per_platform: &utils_per_platform, - markdown, + fluent, } .markdown()?; println!("Wrote to '{p}'"); @@ -173,7 +175,7 @@ struct MDWriter<'a, 'b> { name: &'a str, tldr_zip: &'b mut Option>, utils_per_platform: &'b HashMap<&'b str, Vec>, - markdown: Option, + fluent: Option, } impl MDWriter<'_, '_> { @@ -189,6 +191,33 @@ impl MDWriter<'_, '_> { self.examples() } + /// Extract value for a Fluent key from the .ftl content + fn extract_fluent_value(&self, key: &str) -> Option { + let content = self.fluent.as_ref()?; + let resource = parser::parse(content.clone()).ok()?; + + for entry in resource.body { + if let Entry::Message(Message { + id, + value: Some(Pattern { elements }), + .. + }) = entry + { + if id.name == key { + // Simple text extraction - just concatenate text elements + let mut result = String::new(); + for element in elements { + if let fluent_syntax::ast::PatternElement::TextElement { value } = element { + result.push_str(&value); + } + } + return Some(result); + } + } + } + None + } + /// # Errors /// Returns an error if the writer fails. fn additional(&mut self) -> io::Result<()> { @@ -237,10 +266,7 @@ impl MDWriter<'_, '_> { /// # Errors /// Returns an error if the writer fails. fn usage(&mut self) -> io::Result<()> { - if let Some(markdown) = &self.markdown { - let usage = uuhelp_parser::parse_usage(markdown); - let usage = usage.replace("{}", self.name); - + if let Some(usage) = self.extract_fluent_value(&format!("{}-usage", self.name)) { writeln!(self.w, "\n```")?; writeln!(self.w, "{usage}")?; writeln!(self.w, "```") @@ -252,8 +278,8 @@ impl MDWriter<'_, '_> { /// # Errors /// Returns an error if the writer fails. fn about(&mut self) -> io::Result<()> { - if let Some(markdown) = &self.markdown { - writeln!(self.w, "{}", uuhelp_parser::parse_about(markdown)) + if let Some(about) = self.extract_fluent_value(&format!("{}-about", self.name)) { + writeln!(self.w, "{about}") } else { Ok(()) } @@ -262,13 +288,11 @@ impl MDWriter<'_, '_> { /// # Errors /// Returns an error if the writer fails. fn after_help(&mut self) -> io::Result<()> { - if let Some(markdown) = &self.markdown { - if let Some(after_help) = uuhelp_parser::parse_section("after help", markdown) { - return writeln!(self.w, "\n\n{after_help}"); - } + if let Some(after_help) = self.extract_fluent_value(&format!("{}-after-help", self.name)) { + writeln!(self.w, "\n\n{after_help}") + } else { + Ok(()) } - - Ok(()) } /// # Errors @@ -368,13 +392,17 @@ impl MDWriter<'_, '_> { write!(self.w, "")?; } writeln!(self.w, "")?; + let help_text = arg.get_help().unwrap_or_default().to_string(); + // Try to resolve Fluent key if it looks like one, otherwise use as-is + let resolved_help = if help_text.starts_with(&format!("{}-help-", self.name)) { + self.extract_fluent_value(&help_text).unwrap_or(help_text) + } else { + help_text + }; writeln!( self.w, "
\n\n{}\n\n
", - arg.get_help() - .unwrap_or_default() - .to_string() - .replace('\n', "
") + resolved_help.replace('\n', "
") )?; } writeln!(self.w, "\n") diff --git a/src/uucore_procs/Cargo.toml b/src/uucore_procs/Cargo.toml index 2e9924688..956813592 100644 --- a/src/uucore_procs/Cargo.toml +++ b/src/uucore_procs/Cargo.toml @@ -1,4 +1,3 @@ -# spell-checker:ignore uuhelp [package] name = "uucore_procs" description = "uutils ~ 'uucore' proc-macros" @@ -17,4 +16,3 @@ proc-macro = true [dependencies] proc-macro2 = "1.0.81" quote = "1.0.36" -uuhelp_parser = { path = "../uuhelp_parser", version = "0.2.2" } diff --git a/src/uucore_procs/src/lib.rs b/src/uucore_procs/src/lib.rs index 51741dd9e..e60e2b822 100644 --- a/src/uucore_procs/src/lib.rs +++ b/src/uucore_procs/src/lib.rs @@ -3,14 +3,12 @@ // For the full copyright and license information, please view the LICENSE // file that was distributed with this source code. // -// spell-checker:ignore backticks uuhelp SIGSEGV +// spell-checker:ignore SIGSEGV //! A collection of procedural macros for uutils. #![deny(missing_docs)] -use std::{fs::File, io::Read, path::PathBuf}; - -use proc_macro::{Literal, TokenStream, TokenTree}; +use proc_macro::TokenStream; use quote::quote; //## rust proc-macro background info @@ -48,127 +46,3 @@ pub fn main(_args: TokenStream, stream: TokenStream) -> TokenStream { TokenStream::from(new) } - -// FIXME: This is currently a stub. We could do much more here and could -// even pull in a full markdown parser to get better results. -/// Render markdown into a format that's easier to read in the terminal. -/// -/// For now, all this function does is remove backticks. -/// Some ideas for future improvement: -/// - Render headings as bold -/// - Convert triple backticks to indented -/// - Printing tables in a nice format -fn render_markdown(s: &str) -> String { - s.replace('`', "") -} - -/// Get the about text from the help file. -/// -/// The about text is assumed to be the text between the first markdown -/// code block and the next header, if any. It may span multiple lines. -#[proc_macro] -pub fn help_about(input: TokenStream) -> TokenStream { - let input: Vec = input.into_iter().collect(); - let filename = get_argument(&input, 0, "filename"); - let text: String = uuhelp_parser::parse_about(&read_help(&filename)); - assert!( - !text.is_empty(), - "About text not found! Make sure the markdown format is correct" - ); - TokenTree::Literal(Literal::string(&text)).into() -} - -/// Get the usage from the help file. -/// -/// The usage is assumed to be surrounded by markdown code fences. It may span -/// multiple lines. The first word of each line is assumed to be the name of -/// the util and is replaced by "{}" so that the output of this function can be -/// used with `uucore::format_usage`. -#[proc_macro] -pub fn help_usage(input: TokenStream) -> TokenStream { - let input: Vec = input.into_iter().collect(); - let filename = get_argument(&input, 0, "filename"); - let text: String = uuhelp_parser::parse_usage(&read_help(&filename)); - assert!( - !text.is_empty(), - "Usage text not found! Make sure the markdown format is correct" - ); - TokenTree::Literal(Literal::string(&text)).into() -} - -/// Reads a section from a file of the util as a `str` literal. -/// -/// It reads from the file specified as the second argument, relative to the -/// crate root. The contents of this file are read verbatim, without parsing or -/// escaping. The name of the help file should match the name of the util. -/// I.e. numfmt should have a file called `numfmt.md`. By convention, the file -/// should start with a top-level section with the name of the util. The other -/// sections must start with 2 `#` characters. Capitalization of the sections -/// does not matter. Leading and trailing whitespace of each section will be -/// removed. -/// -/// Example: -/// ```md -/// # numfmt -/// ## About -/// Convert numbers from/to human-readable strings -/// -/// ## Long help -/// This text will be the long help -/// ``` -/// -/// ```rust,ignore -/// help_section!("about", "numfmt.md"); -/// ``` -#[proc_macro] -pub fn help_section(input: TokenStream) -> TokenStream { - let input: Vec = input.into_iter().collect(); - let section = get_argument(&input, 0, "section"); - let filename = get_argument(&input, 1, "filename"); - - if let Some(text) = uuhelp_parser::parse_section(§ion, &read_help(&filename)) { - let rendered = render_markdown(&text); - TokenTree::Literal(Literal::string(&rendered)).into() - } else { - panic!( - "The section '{section}' could not be found in the help file. Maybe it is spelled wrong?" - ) - } -} - -/// Get an argument from the input vector of `TokenTree`. -/// -/// Asserts that the argument is a string literal and returns the string value, -/// otherwise it panics with an error. -fn get_argument(input: &[TokenTree], index: usize, name: &str) -> String { - // Multiply by two to ignore the `','` in between the arguments - let string = match &input.get(index * 2) { - Some(TokenTree::Literal(lit)) => lit.to_string(), - Some(_) => panic!("Argument {index} should be a string literal."), - None => panic!("Missing argument at index {index} for {name}"), - }; - - string - .parse::() - .unwrap() - .strip_prefix('"') - .unwrap() - .strip_suffix('"') - .unwrap() - .to_string() -} - -/// Read the help file -fn read_help(filename: &str) -> String { - let mut content = String::new(); - - let mut path = PathBuf::from(std::env::var("CARGO_MANIFEST_DIR").unwrap()); - path.push(filename); - - File::open(path) - .unwrap() - .read_to_string(&mut content) - .unwrap(); - - content -} diff --git a/src/uuhelp_parser/Cargo.toml b/src/uuhelp_parser/Cargo.toml deleted file mode 100644 index 32d4768d6..000000000 --- a/src/uuhelp_parser/Cargo.toml +++ /dev/null @@ -1,9 +0,0 @@ -# spell-checker:ignore uuhelp -[package] -name = "uuhelp_parser" -description = "A collection of functions to parse the markdown code of help files" -repository = "https://github.com/uutils/coreutils/tree/main/src/uuhelp_parser" -edition.workspace = true -homepage.workspace = true -license.workspace = true -version.workspace = true diff --git a/src/uuhelp_parser/LICENSE b/src/uuhelp_parser/LICENSE deleted file mode 120000 index 30cff7403..000000000 --- a/src/uuhelp_parser/LICENSE +++ /dev/null @@ -1 +0,0 @@ -../../LICENSE \ No newline at end of file diff --git a/src/uuhelp_parser/src/lib.rs b/src/uuhelp_parser/src/lib.rs deleted file mode 100644 index 0e0907f8a..000000000 --- a/src/uuhelp_parser/src/lib.rs +++ /dev/null @@ -1,236 +0,0 @@ -// This file is part of the uutils coreutils package. -// -// For the full copyright and license information, please view the LICENSE -// file that was distributed with this source code. -#![deny(missing_docs)] - -//! A collection of functions to parse the markdown code of help files. -//! -//! The structure of the markdown code is assumed to be: -//! -//! # util name -//! -//! ```text -//! usage info -//! ``` -//! -//! About text -//! -//! ## Section 1 -//! -//! Some content -//! -//! ## Section 2 -//! -//! Some content - -const MARKDOWN_CODE_FENCES: &str = "```"; - -/// Parses the text between the first markdown code block and the next header, if any, -/// into an about string. -pub fn parse_about(content: &str) -> String { - content - .lines() - .skip_while(|l| !l.starts_with(MARKDOWN_CODE_FENCES)) - .skip(1) - .skip_while(|l| !l.starts_with(MARKDOWN_CODE_FENCES)) - .skip(1) - .take_while(|l| !l.starts_with('#')) - .collect::>() - .join("\n") - .trim() - .to_string() -} - -/// Parses the first markdown code block into a usage string -/// -/// The code fences are removed and the name of the util is replaced -/// with `{}` so that it can be replaced with the appropriate name -/// at runtime. -pub fn parse_usage(content: &str) -> String { - content - .lines() - .skip_while(|l| !l.starts_with(MARKDOWN_CODE_FENCES)) - .skip(1) - .take_while(|l| !l.starts_with(MARKDOWN_CODE_FENCES)) - .map(|l| { - // Replace the util name (assumed to be the first word) with "{}" - // to be replaced with the runtime value later. - if let Some((_util, args)) = l.split_once(' ') { - format!("{{}} {args}\n") - } else { - "{}\n".to_string() - } - }) - .collect::() - .trim() - .to_string() -} - -/// Get a single section from content -/// -/// The section must be a second level section (i.e. start with `##`). -pub fn parse_section(section: &str, content: &str) -> Option { - fn is_section_header(line: &str, section: &str) -> bool { - line.strip_prefix("##") - .is_some_and(|l| l.trim().to_lowercase() == section) - } - - let section = §ion.to_lowercase(); - - // We cannot distinguish between an empty or non-existing section below, - // so we do a quick test to check whether the section exists - if content.lines().all(|l| !is_section_header(l, section)) { - return None; - } - - // Prefix includes space to allow processing of section with level 3-6 headers - let section_header_prefix = "## "; - - Some( - content - .lines() - .skip_while(|&l| !is_section_header(l, section)) - .skip(1) - .take_while(|l| !l.starts_with(section_header_prefix)) - .collect::>() - .join("\n") - .trim() - .to_string(), - ) -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn test_parse_section() { - let input = "\ - # ls\n\ - ## some section\n\ - This is some section\n\ - \n\ - ## ANOTHER SECTION - This is the other section\n\ - with multiple lines\n"; - - assert_eq!( - parse_section("some section", input).unwrap(), - "This is some section" - ); - assert_eq!( - parse_section("SOME SECTION", input).unwrap(), - "This is some section" - ); - assert_eq!( - parse_section("another section", input).unwrap(), - "This is the other section\nwith multiple lines" - ); - } - - #[test] - fn test_parse_section_with_sub_headers() { - let input = "\ - # ls\n\ - ## after section\n\ - This is some section\n\ - \n\ - ### level 3 header\n\ - \n\ - Additional text under the section.\n\ - \n\ - #### level 4 header\n\ - \n\ - Yet another paragraph\n"; - - assert_eq!( - parse_section("after section", input).unwrap(), - "This is some section\n\n\ - ### level 3 header\n\n\ - Additional text under the section.\n\n\ - #### level 4 header\n\n\ - Yet another paragraph" - ); - } - - #[test] - fn test_parse_non_existing_section() { - let input = "\ - # ls\n\ - ## some section\n\ - This is some section\n\ - \n\ - ## ANOTHER SECTION - This is the other section\n\ - with multiple lines\n"; - - assert!(parse_section("non-existing section", input).is_none()); - } - - #[test] - fn test_parse_usage() { - let input = "\ - # ls\n\ - ```\n\ - ls -l\n\ - ```\n\ - ## some section\n\ - This is some section\n\ - \n\ - ## ANOTHER SECTION - This is the other section\n\ - with multiple lines\n"; - - assert_eq!(parse_usage(input), "{} -l"); - } - - #[test] - fn test_parse_multi_line_usage() { - let input = "\ - # ls\n\ - ```\n\ - ls -a\n\ - ls -b\n\ - ls -c\n\ - ```\n\ - ## some section\n\ - This is some section\n"; - - assert_eq!(parse_usage(input), "{} -a\n{} -b\n{} -c"); - } - - #[test] - fn test_parse_about() { - let input = "\ - # ls\n\ - ```\n\ - ls -l\n\ - ```\n\ - \n\ - This is the about section\n\ - \n\ - ## some section\n\ - This is some section\n"; - - assert_eq!(parse_about(input), "This is the about section"); - } - - #[test] - fn test_parse_multi_line_about() { - let input = "\ - # ls\n\ - ```\n\ - ls -l\n\ - ```\n\ - \n\ - about a\n\ - \n\ - about b\n\ - \n\ - ## some section\n\ - This is some section\n"; - - assert_eq!(parse_about(input), "about a\n\nabout b"); - } -} diff --git a/util/publish.sh b/util/publish.sh index 00cc9bd00..45c4113e3 100755 --- a/util/publish.sh +++ b/util/publish.sh @@ -1,5 +1,4 @@ #!/bin/sh -# spell-checker:ignore uuhelp ARG="" if test "$1" != "--do-it"; then ARG="--dry-run --allow-dirty" @@ -54,7 +53,7 @@ TOTAL_ORDER=${TOTAL_ORDER#ROOT} CRATE_VERSION=$(grep '^version =' Cargo.toml | head -n1 | cut -d '"' -f2) set -e -for dir in src/uuhelp_parser/ src/uucore_procs/ src/uucore/ src/uu/stdbuf/src/libstdbuf/ tests/uutests/ fuzz/uufuzz/; do +for dir in src/uucore_procs/ src/uucore/ src/uu/stdbuf/src/libstdbuf/ tests/uutests/ fuzz/uufuzz/; do ( cd "$dir" CRATE_NAME=$(grep '^name =' "Cargo.toml" | head -n1 | cut -d '"' -f2) diff --git a/util/update-version.sh b/util/update-version.sh index 45080a93f..3fb76f0a5 100755 --- a/util/update-version.sh +++ b/util/update-version.sh @@ -29,8 +29,6 @@ sed -i -e "s|version = \"$FROM\"|version = \"$TO\"|" $PROGS # Update uucore_procs sed -i -e "s|version = \"$FROM\"|version = \"$TO\"|" src/uucore_procs/Cargo.toml -# Update uuhelp_parser -sed -i -e "s|version = \"$FROM\"|version = \"$TO\"|" src/uuhelp_parser/Cargo.toml # Update the stdbuf stuff sed -i -e "s|libstdbuf = { version=\"$FROM\"|libstdbuf = { version=\"$TO\"|" src/uu/stdbuf/Cargo.toml