From b76a8f55237d5d10b6f6d529ae5262e5e10b19c8 Mon Sep 17 00:00:00 2001 From: Sylvestre Ledru Date: Sat, 26 Sep 2026 19:21:54 +0200 Subject: [PATCH 1/2] docs: move extensions and incompatibilities to docs/src/extensions.md Start an mdBook under docs/, laid out as in coreutils, and move the list of extensions and incompatibilities out of the README into it. The README keeps a pointer. --- README.md | 69 ++------------------------------------- docs/.gitignore | 1 + docs/book.toml | 8 +++++ docs/src/SUMMARY.md | 5 +++ docs/src/extensions.md | 73 ++++++++++++++++++++++++++++++++++++++++++ docs/src/index.md | 10 ++++++ 6 files changed, 99 insertions(+), 67 deletions(-) create mode 100644 docs/.gitignore create mode 100644 docs/book.toml create mode 100644 docs/src/SUMMARY.md create mode 100644 docs/src/extensions.md create mode 100644 docs/src/index.md diff --git a/README.md b/README.md index ce60d17e..40ca9b83 100644 --- a/README.md +++ b/README.md @@ -88,73 +88,8 @@ cargo test ``` ## Extensions and incompatibilities -### Supported GNU extensions -* Command-line arguments can be specified in long (`--`) form. -* Spaces can precede a regular expression modifier. -* `I` can be used in as a synonym for the `i` (case insensitive) substitution - flag. -* `M` and `m` substitution flags allow multi-line matching. -* In addition to `\n`, other escape sequences (octal, hex, C) are supported - in the strings of the `y` command. - Under POSIX these yield undefined behavior. -* The `a`, `c`, and `i` commands do not require an initial backslash, - allow text to appear on the same line, and support escape sequences - in the specified text. -* The `a`, `i`, `=`, `l`, `q` and `r` commands support address range as an extension to POSIX. -* The substitution command replacement group `\0` is a synonym for &. -* An `F` command outputs the name of the file currently being processed. -* A `Q` command (optionally followed by an exit code) quits immediately. -* The `q` command can be optionally followed by an exit code. -* A `W` command writes to a file the pattern's first line. -* An `R` command reads one line at a time from a file. -* The `l` command can be optionally followed by the output width. -* The `--follow-symlinks` option for in-place editing. -* The `--sandbox` option that limits potentially destructive commands. -* Address 0 can be used to specify an address range that is already - active on line 1 and can finish with the specified regular expression. -* Address steps can be specified in the form of start~step and start,~step - ranges. -* Address 0 can be used in the `r` command to prepend a file. - -### Supported BSD and GNU extensions -* The second address in a range can be specified as a relative address with +N. -* In-place editing of file with the `-i` flag. - -### New extensions -* Unicode characters can be specified in regular expression pattern, replacement - and transliteration sequences using `\uXXXX` or `\UXXXXXXXX` sequences. - -### Incompatible extensions -The `-U` or `--uutil-extensions` option enables useful extensions or bug fixes -that aren't compatible with GNU sed or POSIX. - -* The `l` command lists Unicode characters using the `\uXXXX` and `\UXXXXXXXX` - escapes rather than as octal UTF-8 byte sequences. - -### Incompatibilities -* Similarly to GNU _sed_, input is processed as raw bytes or as valid UTF-8 - (this includes 7-bit ASCII) based on the locale as specified by the - `LC_ALL`, `LC_CTYPE`, and `LANG` environment variables, - with the default being byte processing. - However, in contrast with GNU _sed_, other locales (e.g. ISO-8859-1) - are not supported. If the input is in another code page or encoding - and requires locale-specific processing (e.g. ignore/map case, - character classes), consider converting it through UTF-8 to ensure - the correct handling of locale-specific regular expressions. - This _sed_ program can also handle arbitrary byte sequences - if no part of the input requires treating it as a Rust String. -* Back-references aren't supported when input is processed as bytes - (`LC_ALL=C`). -* The command will report an error and fail if duplicate labels are found - in the script. - This matches the BSD behavior. The GNU version accepts duplicate labels. -* The last line (`$`) address is interpreted as the last non-empty line of - the last file. If files specified in subsequent arguments until the last - one are empty, then the last line condition will never be triggered. - This behavior is consistent with the - [original implementation](https://github.com/dspinellis/unix-history-repo/blob/Research-V7/usr/src/cmd/sed/sed1.c#L665). -* Labels are parsed for alphanumeric characters. The BSD version parses them - until the end of the line, preventing ; to be used as a separator. +The GNU, BSD and new extensions _sed_ supports, and where it differs from GNU +_sed_, are listed in [docs/src/extensions.md](docs/src/extensions.md). ## GNU test suite compatibility diff --git a/docs/.gitignore b/docs/.gitignore new file mode 100644 index 00000000..7585238e --- /dev/null +++ b/docs/.gitignore @@ -0,0 +1 @@ +book diff --git a/docs/book.toml b/docs/book.toml new file mode 100644 index 00000000..1e203748 --- /dev/null +++ b/docs/book.toml @@ -0,0 +1,8 @@ +[book] +language = "en" +multilingual = false +src = "src" +title = "uutils sed Documentation" + +[output.html] +git-repository-url = "https://github.com/uutils/sed/tree/main/docs/src" diff --git a/docs/src/SUMMARY.md b/docs/src/SUMMARY.md new file mode 100644 index 00000000..21ccb4b9 --- /dev/null +++ b/docs/src/SUMMARY.md @@ -0,0 +1,5 @@ +# Summary + +[Introduction](index.md) + +* [Extensions and incompatibilities](extensions.md) diff --git a/docs/src/extensions.md b/docs/src/extensions.md new file mode 100644 index 00000000..c81b4b53 --- /dev/null +++ b/docs/src/extensions.md @@ -0,0 +1,73 @@ +# Extensions and incompatibilities + +The main goal of the project is compatibility with GNU _sed_, but _sed_ also +supports features that GNU _sed_ does not, and differs from it in a few places. +Below is a list of these extensions and incompatibilities. + +## Supported GNU extensions +* Command-line arguments can be specified in long (`--`) form. +* Spaces can precede a regular expression modifier. +* `I` can be used in as a synonym for the `i` (case insensitive) substitution + flag. +* `M` and `m` substitution flags allow multi-line matching. +* In addition to `\n`, other escape sequences (octal, hex, C) are supported + in the strings of the `y` command. + Under POSIX these yield undefined behavior. +* The `a`, `c`, and `i` commands do not require an initial backslash, + allow text to appear on the same line, and support escape sequences + in the specified text. +* The `a`, `i`, `=`, `l`, `q` and `r` commands support address range as an extension to POSIX. +* The substitution command replacement group `\0` is a synonym for &. +* An `F` command outputs the name of the file currently being processed. +* A `Q` command (optionally followed by an exit code) quits immediately. +* The `q` command can be optionally followed by an exit code. +* A `W` command writes to a file the pattern's first line. +* An `R` command reads one line at a time from a file. +* The `l` command can be optionally followed by the output width. +* The `--follow-symlinks` option for in-place editing. +* The `--sandbox` option that limits potentially destructive commands. +* Address 0 can be used to specify an address range that is already + active on line 1 and can finish with the specified regular expression. +* Address steps can be specified in the form of start~step and start,~step + ranges. +* Address 0 can be used in the `r` command to prepend a file. + +## Supported BSD and GNU extensions +* The second address in a range can be specified as a relative address with +N. +* In-place editing of file with the `-i` flag. + +## New extensions +* Unicode characters can be specified in regular expression pattern, replacement + and transliteration sequences using `\uXXXX` or `\UXXXXXXXX` sequences. + +## Incompatible extensions +The `-U` or `--uutil-extensions` option enables useful extensions or bug fixes +that aren't compatible with GNU sed or POSIX. + +* The `l` command lists Unicode characters using the `\uXXXX` and `\UXXXXXXXX` + escapes rather than as octal UTF-8 byte sequences. + +## Incompatibilities +* Similarly to GNU _sed_, input is processed as raw bytes or as valid UTF-8 + (this includes 7-bit ASCII) based on the locale as specified by the + `LC_ALL`, `LC_CTYPE`, and `LANG` environment variables, + with the default being byte processing. + However, in contrast with GNU _sed_, other locales (e.g. ISO-8859-1) + are not supported. If the input is in another code page or encoding + and requires locale-specific processing (e.g. ignore/map case, + character classes), consider converting it through UTF-8 to ensure + the correct handling of locale-specific regular expressions. + This _sed_ program can also handle arbitrary byte sequences + if no part of the input requires treating it as a Rust String. +* Back-references aren't supported when input is processed as bytes + (`LC_ALL=C`). +* The command will report an error and fail if duplicate labels are found + in the script. + This matches the BSD behavior. The GNU version accepts duplicate labels. +* The last line (`$`) address is interpreted as the last non-empty line of + the last file. If files specified in subsequent arguments until the last + one are empty, then the last line condition will never be triggered. + This behavior is consistent with the + [original implementation](https://github.com/dspinellis/unix-history-repo/blob/Research-V7/usr/src/cmd/sed/sed1.c#L665). +* Labels are parsed for alphanumeric characters. The BSD version parses them + until the end of the line, preventing ; to be used as a separator. diff --git a/docs/src/index.md b/docs/src/index.md new file mode 100644 index 00000000..12622f2c --- /dev/null +++ b/docs/src/index.md @@ -0,0 +1,10 @@ +# uutils sed + +_sed_ is a Rust reimplementation of the +[sed utility](https://pubs.opengroup.org/onlinepubs/9799919799/utilities/sed.html) +with some [GNU sed](https://www.gnu.org/software/sed/manual/sed.html), +[FreeBSD sed](https://man.freebsd.org/cgi/man.cgi?sed(1)), +and other extensions. + +It is part of the [uutils](https://uutils.github.io/) project. The source code +is on [GitHub](https://github.com/uutils/sed). From 46b0c6884c5f88a3d0ec70929968fb7aac8f9d32 Mon Sep 17 00:00:00 2001 From: Sylvestre Ledru Date: Sun, 2 Aug 2026 09:12:57 +0200 Subject: [PATCH 2/2] sed: underline the offending script character at a terminal sed already knows exactly where a script error is -- every message carries input:line:column -- but a one-line message has nowhere to show it. Quote the offending script line back and underline the character at fault, gated on uucore::diagnostics::enabled() so that UUTILS_DIAG behaves as in every other utility, and drawn with ariadne, the renderer uucore uses. uucore's own Snapshot renders argument lists, so it would label the excerpt sed:1:N and could name neither a -f script nor the line within it. The excerpt is drawn only when stderr is a terminal, so a pipe, a script or the test suite still sees the single line it parses: the ~21 exact stderr assertions in the suite needed no change, and UUTILS_DIAG=always is how the new tests ask for a report down a pipe. Everything flows through the two existing constructors, so no call site changes: compilation_error covers the ~66 sites in compiler.rs and delimited_parser.rs, and location_error covers semantic errors. Semantic errors are raised once the whole script is compiled, long after the line was consumed -- the line provider streams and keeps no history -- so ScriptLocation now carries the line text along with the position. It is only captured when diagnostics are on, otherwise every compiled command would pay for a copy nobody reads, and even then the commands compiled from one line share a single copy of it -- a generated one-line script of 20,000 commands would otherwise hold 20,000 copies of itself. Only the offending line is still in hand, so it is padded with the newlines that came before it; that way the gutter shows the line number the message quotes rather than always 1. Errors that run off the end of the line report the column just past the last character, so the line gets a trailing space for the caret to sit on and the drawn column matches the one the message names. Anything that cannot be drawn -- a script that is not valid UTF-8, an empty line, a location with no line recorded, an error hit once the whole script has been read (whose message names line 0) -- leaves the message alone rather than guessing. --- Cargo.lock | 12 +++ Cargo.toml | 4 +- docs/src/extensions.md | 77 +++++++++++++++ src/sed/error_handling.rs | 160 ++++++++++++++++++++++++++++++-- src/sed/script_char_provider.rs | 20 ++++ tests/by-util/test_sed.rs | 156 +++++++++++++++++++++++++++++++ 6 files changed, 418 insertions(+), 11 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index ed343578..d81b381e 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -85,6 +85,16 @@ dependencies = [ "num-traits", ] +[[package]] +name = "ariadne" +version = "0.6.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8454c8a44ce2cb9cc7e7fae67fc6128465b343b92c6631e94beca3c8d1524ea5" +dependencies = [ + "unicode-width", + "yansi", +] + [[package]] name = "assert_fs" version = "1.1.4" @@ -1105,6 +1115,7 @@ dependencies = [ name = "sed" version = "0.2.0" dependencies = [ + "ariadne", "assert_fs", "chrono", "clap", @@ -1453,6 +1464,7 @@ version = "0.12.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f24a5910ebbf2c0baf486a9125705f8e026a8f7741a3ea8cf5eca3c9257c9bdf" dependencies = [ + "ariadne", "clap", "dns-lookup", "fluent", diff --git a/Cargo.toml b/Cargo.toml index 67c255c3..49fc6172 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -40,6 +40,7 @@ feat_common_core = ["sed"] sed = [] [workspace.dependencies] +ariadne = "0.6.0" assert_fs = "1.1.3" chrono = { version = "0.4.37", default-features = false, features = [ "clock", @@ -62,12 +63,13 @@ sha2 = { version = "0.11.0", default-features = false, features = ["alloc"] } tempfile = "3.10.1" terminal_size = "0.4.2" textwrap = { version = "0.16.1", features = ["terminal_size"] } -uucore = { version = "0.12.0", features = ["libc"] } +uucore = { version = "0.12.0", features = ["diagnostics", "libc"] } rustix = "1.1.4" xattr = "1.3.1" [dependencies] +ariadne = { workspace = true } clap = { workspace = true } clap_complete = { workspace = true } clap_mangen = { workspace = true } diff --git a/docs/src/extensions.md b/docs/src/extensions.md index c81b4b53..9d68c971 100644 --- a/docs/src/extensions.md +++ b/docs/src/extensions.md @@ -39,6 +39,83 @@ Below is a list of these extensions and incompatibilities. ## New extensions * Unicode characters can be specified in regular expression pattern, replacement and transliteration sequences using `\uXXXX` or `\UXXXXXXXX` sequences. +* Script errors are reported with the offending line quoted back and the + character at fault underlined, whenever standard error is a terminal. See + [Script error diagnostics](#script-error-diagnostics). + +## Script error diagnostics +When standard error is a terminal, a script error is followed by the script +line it was found on, with the character at fault underlined. The usual +one-line message always comes first and is unchanged, and when standard error +is a pipe or a file it is the whole of the output, so nothing that parses +_sed_'s output is affected. + +The `UUTILS_DIAG` environment variable overrides the terminal check, as it does +for the other uutils: `never` always gives the single-line message, and +`always` gives the report even when redirected. Colors follow `NO_COLOR`. + +Positions are `input:line:column`, where input is the script file name or +`