mirror of
https://github.com/sharkdp/bat
synced 2026-08-11 20:11:43 +00:00
Compare commits
48 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| b671e53c2c | |||
| de2e651bf5 | |||
| 061fb216cb | |||
| af59a32183 | |||
| c63838efeb | |||
| cecbb80093 | |||
| 3ca4a01be6 | |||
| 8112350da6 | |||
| a5deaf63ff | |||
| 2fb2415731 | |||
| a58f23724a | |||
| 531d54b110 | |||
| db69aab106 | |||
| 2ba8db9c14 | |||
| d344844486 | |||
| aeb0e4c457 | |||
| 1cf12d49bd | |||
| 7cb212646c | |||
| 6573ca9307 | |||
| f04dc80bc7 | |||
| e6ac6ce08d | |||
| 949f18e23a | |||
| 969b33dad1 | |||
| 73dc3258be | |||
| df38225653 | |||
| c2cc79daa7 | |||
| 0230cba42d | |||
| 227e0fa00f | |||
| 384f2f5d13 | |||
| 060ccb1385 | |||
| feabb25209 | |||
| 839f2300b7 | |||
| ea0742babc | |||
| bfa6e59978 | |||
| 9710a231d8 | |||
| a9b67b2d99 | |||
| b22be63fec | |||
| 3a7a750969 | |||
| be5343c742 | |||
| c371044276 | |||
| d449ce5b7e | |||
| 218afc30ac | |||
| 78951393e2 | |||
| 0a52e4321f | |||
| d72163160e | |||
| 971f9679da | |||
| 57d436861a | |||
| 530e83a455 |
@@ -3,11 +3,14 @@
|
||||
|
||||
## Other
|
||||
|
||||
- Update Cargo dependencies to resolve current RustSec advisories, see #3861 (@TyceHerrman)
|
||||
- Add instructions for removing fish help abbreviations to README, see #3655 (@claw-explorer). Closes #3536
|
||||
- Add .NET slnx extension, see #3682 (@ltrzesniewski)
|
||||
|
||||
## Features
|
||||
|
||||
- Add `-b` / `--number-nonblank` flag to only number non-blank lines, for `cat -b` compatibility. Closes #3856, see #3857 (@MeGaurav4)
|
||||
- Add a `--sanitize=<auto|always|never>` flag for safe display of untrusted input. It implies `--strip-ansi` at the same value and additionally substitutes terminal-active control bytes (cursor moves, charset switches, beep, etc.) and Unicode bidi / zero-width formatting characters with the Unicode replacement character (U+FFFD). Mitigates Trojan-Source-style spoofing (CVE-2021-42574). See #3729 (@curious-rabbit)
|
||||
- Map justfile, Justfile, .justfile, and *.justfile to Makefile syntax highlighting, see #3623 (@zachvalenta)
|
||||
- Preserve `--diff` change markers and snip separators when `--plain` is set. Closes #3630, see #3643 (@mvanhorn)
|
||||
- Added support for `hidden_file_extensions` from `.sublime-syntax` files, see #3613 (@Matei02355)
|
||||
@@ -22,6 +25,10 @@
|
||||
- Syntax highlighting for Python files using uv as script runner in shebang #3689 (@janlarres)
|
||||
|
||||
## Bugfixes
|
||||
- Avoid repeated scans of long lines in Log syntax highlighting, see #3876 (@Matei02355)
|
||||
- Fix `--list-languages` respecting `--paging=never`, see #3828 (@cyphercodes)
|
||||
- Fix `--sanitize` passing through the bidi control characters U+200E, U+200F and U+061C, see #3862 (@lenamonj)
|
||||
- `--strip-ansi`: also strip 8-bit C1 introducers (U+0090, U+0098, U+009B, U+009D, U+009E, U+009F) and DCS/SOS/PM/APC sequence bodies, which previously passed through. See #3729 (@curious-rabbit)
|
||||
- Fix `--ignored-suffix` not falling back to first-line/shebang detection when the ignored suffix is also a registered extension (e.g. `--ignored-suffix .txt` on a shebang script), see #2745 and #3816 (@adnrivera)
|
||||
- Fix `capacity overflow` panic when printing a snip separator at `--terminal-width=1` with multiple line ranges. Closes #3803, see #3804 (@leeewee)
|
||||
- Pass `--no-paging` to `bat` invocations inside the bash / zsh / fish / PowerShell shell completion scripts so that shell-level pager wiring (e.g. `LESSOPEN='|-bat -f -pp %s'`) cannot inject ANSI escape sequences into the completion candidates. Closes #3760 (@mvanhorn)
|
||||
|
||||
Generated
+186
-200
File diff suppressed because it is too large
Load Diff
+2
-2
@@ -70,13 +70,13 @@ regex = { version = "1.12.3", optional = true }
|
||||
walkdir = { version = "2.5", optional = true }
|
||||
bytesize = { version = "2.3.1" }
|
||||
encoding_rs = "0.8.35"
|
||||
execute = { version = "0.2.15", optional = true }
|
||||
execute = { version = "0.3.0", optional = true }
|
||||
terminal-colorsaurus = "1.0"
|
||||
unicode-segmentation = "1.13.2"
|
||||
itertools = "0.14.0"
|
||||
|
||||
[dependencies.gix]
|
||||
version = "0.85"
|
||||
version = "0.86"
|
||||
optional = true
|
||||
default-features = false
|
||||
features = ["sha1", "blob-diff"]
|
||||
|
||||
@@ -843,6 +843,10 @@ If your version of `bat` supports the `--strip-ansi=auto` option, it can be used
|
||||
before syntax highlighting. Alternatively, you may disable both syntax highlighting and wrapping by
|
||||
passing the `--color=never --wrap=never` options to `bat`.
|
||||
|
||||
For untrusted input, the `--sanitize=auto|always|never` option additionally replaces terminal-active
|
||||
control bytes and Unicode bidi / zero-width formatting characters with the Unicode replacement
|
||||
character. It implies `--strip-ansi` at the same value.
|
||||
|
||||
> [!NOTE]
|
||||
> The `auto` option of `--strip-ansi` avoids removing escape sequences when the syntax is plain text.
|
||||
|
||||
@@ -898,7 +902,7 @@ cargo install --path . --locked --force
|
||||
```
|
||||
|
||||
If you want to build an application that uses `bat`'s pretty-printing
|
||||
features as a library, check out the [the API documentation](https://docs.rs/bat/).
|
||||
features as a library, check out the [API documentation](https://docs.rs/bat/).
|
||||
Note that you have to use either `regex-onig` or `regex-fancy` as a feature
|
||||
when you depend on `bat` as a library.
|
||||
|
||||
|
||||
Vendored
+6
@@ -100,6 +100,11 @@ Register-ArgumentCompleter -Native -CommandName '{{PROJECT_EXECUTABLE}}' -Script
|
||||
ForEach-Object {[System.Management.Automation.CompletionResult]::new($_, $_, [CompletionResultType]::ParameterValue, $_)}
|
||||
break
|
||||
}
|
||||
'*;--sanitize' {
|
||||
$ArrayWhen |
|
||||
ForEach-Object {[System.Management.Automation.CompletionResult]::new($_, $_, [CompletionResultType]::ParameterValue, $_)}
|
||||
break
|
||||
}
|
||||
'*;--strip-ansi' {
|
||||
$ArrayWhen |
|
||||
ForEach-Object {[System.Management.Automation.CompletionResult]::new($_, $_, [CompletionResultType]::ParameterValue, $_)}
|
||||
@@ -158,6 +163,7 @@ Register-ArgumentCompleter -Native -CommandName '{{PROJECT_EXECUTABLE}}' -Script
|
||||
[CompletionResult]::new('--ignored-suffix' , 'ignored-suffix' , [CompletionResultType]::ParameterName, 'Ignore extension. For example: ''bat --ignored-suffix ".dev" my_file.json.dev'' will use JSON syntax, and ignore ''.dev''')
|
||||
[CompletionResult]::new('--squeeze-blank' , 'squeeze-blank' , [CompletionResultType]::ParameterName, 'Squeeze consecutive empty lines into a single empty line.')
|
||||
[CompletionResult]::new('--squeeze-limit' , 'squeeze-limit' , [CompletionResultType]::ParameterName, 'Set the maximum number of consecutive empty lines to be printed.')
|
||||
[CompletionResult]::new('--sanitize' , 'sanitize' , [CompletionResultType]::ParameterName, 'Specify when to sanitize untrusted input for safe display. Implies --strip-ansi and also replaces terminal-active and bidi / zero-width bytes. (auto, always, *never*).')
|
||||
[CompletionResult]::new('--strip-ansi' , 'strip-ansi' , [CompletionResultType]::ParameterName, 'Specify when to strip ANSI escape sequences from the input. The automatic mode will remove escape sequences unless the syntax highlighting language is plain text. (auto, always, *never*).')
|
||||
# [CompletionResult]::new('-p' , 'p' , [CompletionResultType]::ParameterName, 'Show plain style (alias for ''--style=plain'').')
|
||||
[CompletionResult]::new('--plain' , 'plain' , [CompletionResultType]::ParameterName, 'Show plain style (alias for ''--style=plain'').')
|
||||
|
||||
Vendored
+5
@@ -132,6 +132,10 @@ _bat() {
|
||||
COMPREPLY=($(compgen -W "auto never always" -- "$cur"))
|
||||
return 0
|
||||
;;
|
||||
--sanitize)
|
||||
COMPREPLY=($(compgen -W "auto never always" -- "$cur"))
|
||||
return 0
|
||||
;;
|
||||
--completion)
|
||||
COMPREPLY=($(compgen -W "bash fish zsh ps1" -- "$cur"))
|
||||
return 0
|
||||
@@ -221,6 +225,7 @@ _bat() {
|
||||
--list-themes
|
||||
--squeeze-blank
|
||||
--squeeze-limit
|
||||
--sanitize
|
||||
--strip-ansi
|
||||
--style
|
||||
--line-range
|
||||
|
||||
Vendored
+1
@@ -221,6 +221,7 @@ complete -c $bat -s s -l squeeze-blank -d "Squeeze consecutive empty lines into
|
||||
complete -c $bat -l squeeze-limit -x -d "Set the maximum number of consecutive empty lines to be printed" -n __bat_no_excl_args
|
||||
|
||||
complete -c $bat -l strip-ansi -x -a "auto never always" -d "Specify when to strip ANSI escape sequences from the input" -n __bat_no_excl_args
|
||||
complete -c $bat -l sanitize -x -a "auto never always" -d "Specify when to sanitize untrusted input for safe display" -n __bat_no_excl_args
|
||||
|
||||
complete -c $bat -s p -l plain -d "Disable decorations" -n __bat_no_excl_args
|
||||
|
||||
|
||||
Vendored
+1
@@ -54,6 +54,7 @@ _{{PROJECT_EXECUTABLE}}_main() {
|
||||
--squeeze-blank'[squeeze consecutive empty lines into a single empty line]'
|
||||
--squeeze-limit='[set the maximum number of consecutive empty lines]:limit:'
|
||||
--strip-ansi='[specify when to strip ANSI escape sequences]:when:(auto never always)'
|
||||
--sanitize='[specify when to sanitize untrusted input for safe display]:when:(auto never always)'
|
||||
--style='[comma-separated list of style elements to display]: : _values "style [default]"
|
||||
default auto full plain changes header header-filename header-filesize grid rule numbers snip'
|
||||
\*{-r+,--line-range=}'[only print the specified line range]:start\:end'
|
||||
|
||||
Vendored
+1
-1
Submodule assets/syntaxes/02_Extra/LESS updated: 836b47ec61...d076a4dd41
+23
-4
@@ -38,22 +38,41 @@ contexts:
|
||||
scope: markup.underline.link.scheme.log
|
||||
push: url-host
|
||||
log_level_lines:
|
||||
- match: (?=.*{{error}})
|
||||
# Prevent retrying each lookahead at every position.
|
||||
- match: ^(?=.*{{error}})
|
||||
push:
|
||||
- error_line_meta
|
||||
- main_pop_at_eol
|
||||
- match: (?=.*{{warning}})
|
||||
- match: ^(?=.*{{warning}})
|
||||
push:
|
||||
- warning_line_meta
|
||||
- main_pop_at_eol
|
||||
- match: (?=.*{{info}})
|
||||
- match: ^(?=.*{{info}})
|
||||
push:
|
||||
- info_line_meta
|
||||
- main_pop_at_eol
|
||||
- match: (?=.*{{debug}})
|
||||
- match: ^(?=.*{{debug}})
|
||||
push:
|
||||
- debug_line_meta
|
||||
- main_pop_at_eol
|
||||
# For syntaxes that embed Log after consuming a prefix. Call this context once.
|
||||
log_level_lines_at_current_position:
|
||||
- match: (?=.*{{error}})
|
||||
set:
|
||||
- error_line_meta
|
||||
- main_pop_at_eol
|
||||
- match: (?=.*{{warning}})
|
||||
set:
|
||||
- warning_line_meta
|
||||
- main_pop_at_eol
|
||||
- match: (?=.*{{info}})
|
||||
set:
|
||||
- info_line_meta
|
||||
- main_pop_at_eol
|
||||
- match: (?=.*{{debug}})
|
||||
set:
|
||||
- debug_line_meta
|
||||
- main_pop_at_eol
|
||||
log_levels:
|
||||
- match: '{{error}}'
|
||||
scope: markup.error.log
|
||||
|
||||
+11
-1
@@ -50,6 +50,17 @@ contexts:
|
||||
- match: (?=\])
|
||||
pop: true
|
||||
text:
|
||||
- include: text_special
|
||||
- match: ''
|
||||
set: classify_log_text
|
||||
classify_log_text:
|
||||
- include: scope:text.log#log_level_lines_at_current_position
|
||||
- match: ''
|
||||
set: text_body
|
||||
text_body:
|
||||
- include: text_special
|
||||
- include: scope:text.log
|
||||
text_special:
|
||||
- match: $
|
||||
pop: true
|
||||
- match: '<\w+>'
|
||||
@@ -62,4 +73,3 @@ contexts:
|
||||
escape: \)$
|
||||
escape_captures:
|
||||
0: punctuation.section.block.end.syslog
|
||||
- include: scope:text.log
|
||||
|
||||
@@ -83,6 +83,16 @@ Options:
|
||||
-n, --number
|
||||
Only show line numbers, no other decorations. This is an alias for '--style=numbers'
|
||||
|
||||
-b, --number-nonblank
|
||||
Only show line numbers for non-blank lines, no other decorations. This is an alias for
|
||||
'--style=numbers'. Non-blank lines are lines that contain any character before the line
|
||||
ending, including spaces and tabs. When used together with --number (-n),
|
||||
--number-nonblank (-b) takes precedence.
|
||||
|
||||
Example:
|
||||
printf 'alpha\n\nbeta\n' | bat -b
|
||||
numbers 'alpha' and 'beta', but skips the empty line.
|
||||
|
||||
--color <when>
|
||||
Specify when to use colored output. The automatic mode only enables colors if an
|
||||
interactive terminal is detected - colors are automatically disabled if the output goes to
|
||||
@@ -164,6 +174,14 @@ Options:
|
||||
escape sequences unless the syntax highlighting language is plain text. Possible values:
|
||||
auto, always, *never*.
|
||||
|
||||
--sanitize <when>
|
||||
Specify when to sanitize input bytes for safe terminal display. Implies --strip-ansi to
|
||||
the same value, and additionally substitutes terminal-active control bytes (cursor moves,
|
||||
charset switches, beep, etc.) and Unicode bidi / zero-width formatting characters with the
|
||||
Unicode replacement character (U+FFFD). Tab, LF, FF, and CRLF pass through. Useful for
|
||||
displaying untrusted file content (e.g. file-manager preview panes). Possible values:
|
||||
auto, always, *never*.
|
||||
|
||||
--style <components>
|
||||
Configure which elements (line numbers, file headers, grid borders, Git modifications, ..)
|
||||
to display in addition to the file contents. The argument is a comma-separated list of
|
||||
|
||||
@@ -33,6 +33,8 @@ Options:
|
||||
Truncate all lines longer than screen width. Alias for '--wrap=never'.
|
||||
-n, --number
|
||||
Show line numbers (alias for '--style=numbers').
|
||||
-b, --number-nonblank
|
||||
Show line numbers for non-blank lines only (alias for '--style=numbers').
|
||||
--color <when>
|
||||
When to use colors (*auto*, never, always).
|
||||
--italic-text <when>
|
||||
|
||||
+2
-2
@@ -363,11 +363,11 @@ impl HighlightingAssets {
|
||||
reader: &mut InputReader,
|
||||
) -> Result<Option<SyntaxReferenceInSet<'_>>> {
|
||||
let syntax_set = self.get_syntax_set()?;
|
||||
Ok(String::from_utf8(reader.first_line.clone())
|
||||
Ok(std::str::from_utf8(&reader.first_line)
|
||||
.ok()
|
||||
.and_then(|l| {
|
||||
// Strip UTF-8 BOM if present
|
||||
let line = l.strip_prefix('\u{feff}').unwrap_or(&l);
|
||||
let line = l.strip_prefix('\u{feff}').unwrap_or(l);
|
||||
syntax_set.find_syntax_by_first_line(line)
|
||||
})
|
||||
.map(|syntax| SyntaxReferenceInSet { syntax, syntax_set }))
|
||||
|
||||
+72
-9
@@ -38,6 +38,15 @@ pub fn env_no_color() -> bool {
|
||||
env::var_os("NO_COLOR").is_some_and(|x| !x.is_empty())
|
||||
}
|
||||
|
||||
fn parse_strip_ansi_value(raw: Option<&str>, flag_name: &str) -> StripAnsiMode {
|
||||
match raw {
|
||||
Some("never") | None => StripAnsiMode::Never,
|
||||
Some("always") => StripAnsiMode::Always,
|
||||
Some("auto") => StripAnsiMode::Auto,
|
||||
_ => unreachable!("other values for {flag_name} are not allowed"),
|
||||
}
|
||||
}
|
||||
|
||||
enum HelpType {
|
||||
Short,
|
||||
Long,
|
||||
@@ -50,6 +59,10 @@ pub struct App {
|
||||
/// (not from config file or environment variables).
|
||||
/// This is used to honor the flag when piping output, similar to `cat -n`.
|
||||
number_from_cli: bool,
|
||||
/// True if -b / --number-nonblank was passed on the command line
|
||||
/// (not from config file or environment variables).
|
||||
/// This is used to honor the flag when piping output, similar to `cat -b`.
|
||||
number_nonblank_from_cli: bool,
|
||||
}
|
||||
|
||||
impl App {
|
||||
@@ -90,6 +103,29 @@ impl App {
|
||||
false
|
||||
});
|
||||
|
||||
// Check if the -b / --number-nonblank option was passed on the command line
|
||||
// (before merging with config file and environment variables).
|
||||
// This is needed to honor the -b flag when piping output, similar to `cat -b`.
|
||||
// The same combined-flag logic applies as for -n above.
|
||||
let number_nonblank_from_cli = wild::args_os().any(|arg| {
|
||||
let arg_str = arg.to_string_lossy();
|
||||
if arg_str == "-b" || arg_str == "--number-nonblank" {
|
||||
return true;
|
||||
}
|
||||
// Handle combined short flags
|
||||
if arg_str.starts_with('-') && !arg_str.starts_with("--") && arg_str.len() > 2 {
|
||||
let chars: Vec<char> = arg_str.chars().skip(1).collect();
|
||||
let b_pos = chars.iter().position(|&c| c == 'b');
|
||||
let p_pos = chars.iter().position(|&c| c == 'p');
|
||||
if let Some(b) = b_pos {
|
||||
if p_pos.is_none() || b > p_pos.unwrap() {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
false
|
||||
});
|
||||
|
||||
let matches = Self::matches(interactive_output)?;
|
||||
|
||||
if matches.get_flag("help") {
|
||||
@@ -130,6 +166,7 @@ impl App {
|
||||
matches,
|
||||
interactive_output,
|
||||
number_from_cli,
|
||||
number_nonblank_from_cli,
|
||||
})
|
||||
}
|
||||
|
||||
@@ -445,7 +482,8 @@ impl App {
|
||||
.map(|s| s.as_str())
|
||||
== Some("always")
|
||||
|| self.matches.get_flag("force-colorization")
|
||||
|| self.number_from_cli),
|
||||
|| self.number_from_cli
|
||||
|| self.number_nonblank_from_cli),
|
||||
tab_width: self
|
||||
.matches
|
||||
.get_one::<String>("tabs")
|
||||
@@ -458,18 +496,36 @@ impl App {
|
||||
4
|
||||
},
|
||||
),
|
||||
strip_ansi: match self
|
||||
.matches
|
||||
strip_ansi: {
|
||||
let sanitize = parse_strip_ansi_value(
|
||||
self.matches
|
||||
.get_one::<String>("sanitize")
|
||||
.map(|s| s.as_str()),
|
||||
"--sanitize",
|
||||
);
|
||||
let strip_ansi = parse_strip_ansi_value(
|
||||
self.matches
|
||||
.get_one::<String>("strip-ansi")
|
||||
.map(|s| s.as_str())
|
||||
{
|
||||
Some("never") => StripAnsiMode::Never,
|
||||
Some("always") => StripAnsiMode::Always,
|
||||
Some("auto") => StripAnsiMode::Auto,
|
||||
_ => unreachable!("other values for --strip-ansi are not allowed"),
|
||||
.map(|s| s.as_str()),
|
||||
"--strip-ansi",
|
||||
);
|
||||
// --sanitize implies --strip-ansi to the same value.
|
||||
if sanitize != StripAnsiMode::Never {
|
||||
sanitize
|
||||
} else {
|
||||
strip_ansi
|
||||
}
|
||||
},
|
||||
sanitize: parse_strip_ansi_value(
|
||||
self.matches
|
||||
.get_one::<String>("sanitize")
|
||||
.map(|s| s.as_str()),
|
||||
"--sanitize",
|
||||
),
|
||||
quiet_empty: self.matches.get_flag("quiet-empty"),
|
||||
unbuffered: self.matches.get_flag("unbuffered"),
|
||||
number_nonblank: self.matches.get_flag("number-nonblank")
|
||||
|| self.number_nonblank_from_cli,
|
||||
theme: theme(self.theme_options()).to_string(),
|
||||
visible_lines: match self.matches.try_contains_id("diff").unwrap_or_default()
|
||||
&& self.matches.get_flag("diff")
|
||||
@@ -588,6 +644,13 @@ impl App {
|
||||
])));
|
||||
}
|
||||
|
||||
// Only line numbers for non-blank lines if `--number-nonblank`.
|
||||
if self.matches.get_flag("number-nonblank") || self.number_nonblank_from_cli {
|
||||
return Some(StyleComponents(HashSet::from([
|
||||
StyleComponent::LineNumbers,
|
||||
])));
|
||||
}
|
||||
|
||||
// Plain if `--plain` is specified at least once.
|
||||
if self.matches.get_count("plain") > 0 {
|
||||
let mut components = HashSet::from([StyleComponent::Plain]);
|
||||
|
||||
@@ -97,6 +97,7 @@ pub fn build_app(interactive_output: bool) -> Command {
|
||||
Arg::new("plain")
|
||||
.overrides_with("plain")
|
||||
.overrides_with("number")
|
||||
.overrides_with("number-nonblank")
|
||||
.short('p')
|
||||
.long("plain")
|
||||
.action(ArgAction::Count)
|
||||
@@ -277,6 +278,23 @@ pub fn build_app(interactive_output: bool) -> Command {
|
||||
'--style=numbers'",
|
||||
),
|
||||
)
|
||||
.arg(
|
||||
Arg::new("number-nonblank")
|
||||
.long("number-nonblank")
|
||||
.overrides_with("number-nonblank")
|
||||
.short('b')
|
||||
.action(ArgAction::SetTrue)
|
||||
.help("Show line numbers for non-blank lines only (alias for '--style=numbers').")
|
||||
.long_help(
|
||||
"Only show line numbers for non-blank lines, no other decorations. This is an \
|
||||
alias for '--style=numbers'. Non-blank lines are lines that contain any \
|
||||
character before the line ending, including spaces and tabs. When used \
|
||||
together with --number (-n), --number-nonblank (-b) takes precedence.\n\n\
|
||||
Example:\n \
|
||||
printf 'alpha\\n\\nbeta\\n' | bat -b\n \
|
||||
numbers 'alpha' and 'beta', but skips the empty line.",
|
||||
),
|
||||
)
|
||||
.arg(
|
||||
Arg::new("color")
|
||||
.long("color")
|
||||
@@ -480,6 +498,24 @@ pub fn build_app(interactive_output: bool) -> Command {
|
||||
language is plain text. Possible values: auto, always, *never*.")
|
||||
.hide_short_help(true)
|
||||
)
|
||||
.arg(
|
||||
Arg::new("sanitize")
|
||||
.long("sanitize")
|
||||
.overrides_with("sanitize")
|
||||
.value_name("when")
|
||||
.value_parser(["auto", "always", "never"])
|
||||
.default_value("never")
|
||||
.hide_default_value(true)
|
||||
.help("Sanitize untrusted input for safe display (auto, always, *never*)")
|
||||
.long_help("Specify when to sanitize input bytes for safe terminal display. \
|
||||
Implies --strip-ansi to the same value, and additionally substitutes \
|
||||
terminal-active control bytes (cursor moves, charset switches, beep, etc.) \
|
||||
and Unicode bidi / zero-width formatting characters with the Unicode \
|
||||
replacement character (U+FFFD). Tab, LF, FF, and CRLF pass through. Useful \
|
||||
for displaying untrusted file content (e.g. file-manager preview panes). \
|
||||
Possible values: auto, always, *never*.")
|
||||
.hide_short_help(true)
|
||||
)
|
||||
.arg(
|
||||
Arg::new("style")
|
||||
.long("style")
|
||||
|
||||
+3
-5
@@ -427,11 +427,9 @@ fn run() -> Result<bool> {
|
||||
if app.matches.get_flag("list-languages") {
|
||||
let languages: String = get_languages(&config, cache_dir)?;
|
||||
let inputs: Vec<Input> = vec![Input::from_reader(Box::new(languages.as_bytes()))];
|
||||
let plain_config = Config {
|
||||
style_components: StyleComponents::new(StyleComponent::Plain.components(false)),
|
||||
paging_mode: PagingMode::QuitIfOneScreen,
|
||||
..Default::default()
|
||||
};
|
||||
let mut plain_config = config.clone();
|
||||
plain_config.style_components =
|
||||
StyleComponents::new(StyleComponent::Plain.components(false));
|
||||
run_controller(inputs, &plain_config, cache_dir)
|
||||
} else if app.matches.get_flag("list-themes") {
|
||||
list_themes(&config, config_dir, cache_dir, app.theme_options())?;
|
||||
|
||||
@@ -111,11 +111,18 @@ pub struct Config<'a> {
|
||||
// Whether or not to strip ANSI escape codes from the input
|
||||
pub strip_ansi: StripAnsiMode,
|
||||
|
||||
// Substitute terminal-active and spoofing-relevant bytes; implies strip_ansi.
|
||||
pub sanitize: StripAnsiMode,
|
||||
|
||||
/// Whether or not to produce no output when input is empty
|
||||
pub quiet_empty: bool,
|
||||
|
||||
/// Whether or not to use unbuffered input reading for streaming use cases
|
||||
pub unbuffered: bool,
|
||||
|
||||
/// Only number non-blank lines (like `cat -b`). Has no effect if `style_components` doesn't
|
||||
/// include `LineNumbers`.
|
||||
pub number_nonblank: bool,
|
||||
}
|
||||
|
||||
#[cfg(all(feature = "minimal-application", feature = "paging"))]
|
||||
|
||||
+12
-1
@@ -276,10 +276,21 @@ impl Controller<'_> {
|
||||
}
|
||||
if !reached_eof {
|
||||
if reader.read_line(&mut current_line_buffer)? {
|
||||
// Fill the buffer
|
||||
// Fill the buffer. In number-nonblank mode, don't advance the
|
||||
// line counter for empty lines (content-free lines that contain
|
||||
// only the line terminator).
|
||||
if self.config.number_nonblank
|
||||
&& current_line_buffer
|
||||
.iter()
|
||||
.all(|&b| b == b'\r' || b == b'\n')
|
||||
{
|
||||
buffered_lines
|
||||
.push_back((mem::take(&mut current_line_buffer), current_line_number));
|
||||
} else {
|
||||
buffered_lines
|
||||
.push_back((mem::take(&mut current_line_buffer), current_line_number));
|
||||
current_line_number += 1;
|
||||
}
|
||||
} else {
|
||||
// No more data to read
|
||||
reached_eof = true;
|
||||
|
||||
@@ -45,6 +45,14 @@ impl Decoration for LineNumberDecoration {
|
||||
continuation: bool,
|
||||
_printer: &InteractivePrinter,
|
||||
) -> DecorationText {
|
||||
if line_number == 0 {
|
||||
// Blank line in number-nonblank mode: show empty space instead of a number.
|
||||
return DecorationText {
|
||||
text: self.color.paint(" ".repeat(self.width())).to_string(),
|
||||
width: self.width(),
|
||||
};
|
||||
}
|
||||
|
||||
if continuation {
|
||||
if line_number >= self.cached_wrap_invalid_at {
|
||||
let new_width = self.cached_wrap.width + 1;
|
||||
|
||||
+204
-2
@@ -139,16 +139,111 @@ pub fn replace_nonprintable(
|
||||
/// Strips ANSI escape sequences from the input.
|
||||
pub fn strip_ansi(line: &str) -> String {
|
||||
let mut buffer = String::with_capacity(line.len());
|
||||
|
||||
for seq in EscapeSequenceOffsetsIterator::new(line) {
|
||||
if let EscapeSequenceOffsets::Text { .. } = seq {
|
||||
buffer.push_str(&line[seq.index_of_start()..seq.index_past_end()]);
|
||||
}
|
||||
}
|
||||
|
||||
buffer
|
||||
}
|
||||
|
||||
/// Strips ANSI escape sequences and substitutes terminal-active control bytes
|
||||
/// and visual-spoofing Unicode codepoints (bidi, zero-width) with U+FFFD.
|
||||
pub fn sanitize(line: &str) -> String {
|
||||
let stripped = strip_ansi(line);
|
||||
let mut buffer = String::with_capacity(stripped.len());
|
||||
let bytes = stripped.as_bytes();
|
||||
let mut start = 0;
|
||||
let mut i = 0;
|
||||
// Skip directly to the next trigger byte instead of testing each one.
|
||||
while let Some(off) = bytes[i..].iter().position(|&b| is_sanitize_trigger(b)) {
|
||||
i += off;
|
||||
let len = sanitize_at(bytes, i, &stripped, &mut buffer, &mut start);
|
||||
i += len;
|
||||
}
|
||||
buffer.push_str(&stripped[start..]);
|
||||
buffer
|
||||
}
|
||||
|
||||
#[inline]
|
||||
fn is_sanitize_trigger(b: u8) -> bool {
|
||||
// C0 controls minus \t \n \f; DEL; UTF-8 leads with dangerous codepoints.
|
||||
matches!(b, 0x00..=0x08 | 0x0B | 0x0D..=0x1F | 0x7F | 0xC2 | 0xD8 | 0xE2 | 0xEF)
|
||||
}
|
||||
|
||||
/// Substitutes the byte/sequence at `bytes[i]` (or passes it through on
|
||||
/// false-alarm trigger), flushing the prefix from `start`. Returns bytes consumed.
|
||||
fn sanitize_at(
|
||||
bytes: &[u8],
|
||||
i: usize,
|
||||
full: &str,
|
||||
buffer: &mut String,
|
||||
start: &mut usize,
|
||||
) -> usize {
|
||||
buffer.push_str(&full[*start..i]);
|
||||
let consumed = match bytes[i] {
|
||||
b'\r' if bytes.get(i + 1) == Some(&b'\n') => {
|
||||
buffer.push_str("\r\n");
|
||||
2
|
||||
}
|
||||
// 0xC2 leads U+0080..U+00FF; filter the C1 range.
|
||||
0xC2 if matches!(bytes.get(i + 1), Some(0x80..=0x9F)) => {
|
||||
buffer.push('\u{FFFD}');
|
||||
2
|
||||
}
|
||||
0xE2 if is_dangerous_e2(bytes, i) => {
|
||||
buffer.push('\u{FFFD}');
|
||||
3
|
||||
}
|
||||
// 0xEF 0xBB 0xBF = U+FEFF (BOM / zero-width no-break space).
|
||||
0xEF if bytes.get(i + 1) == Some(&0xBB) && bytes.get(i + 2) == Some(&0xBF) => {
|
||||
buffer.push('\u{FFFD}');
|
||||
3
|
||||
}
|
||||
// 0xD8 0x9C = U+061C (Arabic letter mark, a bidi control). The rest of
|
||||
// the 0xD8 block is ordinary Arabic text.
|
||||
0xD8 if bytes.get(i + 1) == Some(&0x9C) => {
|
||||
buffer.push('\u{FFFD}');
|
||||
2
|
||||
}
|
||||
// False-alarm trigger: pass the full UTF-8 sequence through.
|
||||
lead @ (0xC2 | 0xD8 | 0xE2 | 0xEF) => {
|
||||
let n = utf8_len_from_lead(lead);
|
||||
buffer.push_str(&full[i..i + n]);
|
||||
n
|
||||
}
|
||||
_ => {
|
||||
buffer.push('\u{FFFD}');
|
||||
1
|
||||
}
|
||||
};
|
||||
*start = i + consumed;
|
||||
consumed
|
||||
}
|
||||
|
||||
#[inline]
|
||||
fn is_dangerous_e2(bytes: &[u8], i: usize) -> bool {
|
||||
// U+200B..D (zero-width), U+200E..F (LRM/RLM), U+202A..E (bidi embedding
|
||||
// and override), U+2066..9 (bidi isolates).
|
||||
matches!(
|
||||
(bytes.get(i + 1), bytes.get(i + 2)),
|
||||
(Some(0x80), Some(0x8B..=0x8F | 0xAA..=0xAE)) | (Some(0x81), Some(0xA6..=0xA9))
|
||||
)
|
||||
}
|
||||
|
||||
#[inline]
|
||||
fn utf8_len_from_lead(lead: u8) -> usize {
|
||||
if lead < 0x80 {
|
||||
1
|
||||
} else if lead < 0xE0 {
|
||||
2
|
||||
} else if lead < 0xF0 {
|
||||
3
|
||||
} else {
|
||||
4
|
||||
}
|
||||
}
|
||||
|
||||
/// Escape C0, DEL, and C1 control characters so a string from an untrusted
|
||||
/// filename or path can be safely written to the terminal.
|
||||
pub fn sanitize_for_terminal(input: &str) -> String {
|
||||
@@ -270,6 +365,113 @@ fn test_strip_ansi() {
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_strip_ansi_8bit_c1_introducers() {
|
||||
assert_eq!(strip_ansi("a\u{9B}31mRED\u{9B}0mb"), "aREDb");
|
||||
assert_eq!(strip_ansi("a\x1bP1;0|payload\x1b\\b"), "ab");
|
||||
assert_eq!(strip_ansi("a\u{90}body\u{9C}b"), "ab");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_strip_ansi_single_char_esc() {
|
||||
// strip_ansi must consume both bytes of single-byte ESC sequences (RIS, DECSC, keypad, VT52).
|
||||
assert_eq!(strip_ansi("a\x1bcb"), "ab");
|
||||
assert_eq!(strip_ansi("a\x1b7b\x1b8c"), "abc");
|
||||
assert_eq!(strip_ansi("a\x1b=b\x1b>c"), "abc");
|
||||
assert_eq!(strip_ansi("a\x1bZb"), "ab");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_strip_ansi_preserves_control_bytes() {
|
||||
// strip_ansi removes only ANSI escape sequences; control bytes pass through.
|
||||
assert_eq!(strip_ansi("safe\rEVIL"), "safe\rEVIL");
|
||||
assert_eq!(strip_ansi("a\x08b\x07c\x0E\x0Fd"), "a\x08b\x07c\x0E\x0Fd");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_sanitize_substitutes_dangerous_bytes() {
|
||||
let r = '\u{FFFD}';
|
||||
assert_eq!(sanitize("safe\rEVIL"), format!("safe{r}EVIL"));
|
||||
assert_eq!(sanitize("a\x08b"), format!("a{r}b"));
|
||||
assert_eq!(sanitize("a\x07b"), format!("a{r}b"));
|
||||
assert_eq!(sanitize("a\x0Bb"), format!("a{r}b"));
|
||||
assert_eq!(sanitize("a\x0Eb\x0Fc"), format!("a{r}b{r}c"));
|
||||
assert_eq!(sanitize("a\u{8D}b"), format!("a{r}b"));
|
||||
assert_eq!(sanitize("a\u{85}b"), format!("a{r}b"));
|
||||
assert_eq!(sanitize("trailing\r"), format!("trailing{r}"));
|
||||
assert_eq!(sanitize("a\x7Fb"), format!("a{r}b"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_sanitize_substitutes_bidi_and_zero_width() {
|
||||
let r = '\u{FFFD}';
|
||||
// Trojan-Source bidi formatting: U+202A..U+202E
|
||||
assert_eq!(sanitize("a\u{202A}b"), format!("a{r}b"));
|
||||
assert_eq!(sanitize("a\u{202E}b"), format!("a{r}b"));
|
||||
// Bidi isolates: U+2066..U+2069
|
||||
assert_eq!(sanitize("a\u{2066}b"), format!("a{r}b"));
|
||||
assert_eq!(sanitize("a\u{2069}b"), format!("a{r}b"));
|
||||
// Zero-width: U+200B..U+200D
|
||||
assert_eq!(sanitize("a\u{200B}b"), format!("a{r}b"));
|
||||
assert_eq!(sanitize("a\u{200D}b"), format!("a{r}b"));
|
||||
// BOM in middle of file: U+FEFF
|
||||
assert_eq!(sanitize("a\u{FEFF}b"), format!("a{r}b"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_sanitize_substitutes_every_bidi_control() {
|
||||
// Unicode Bidi_Control is exactly these 12 codepoints. Covering only some
|
||||
// of them leaves a reordering attack available through the rest.
|
||||
let r = '\u{FFFD}';
|
||||
for c in [
|
||||
'\u{061C}', '\u{200E}', '\u{200F}', '\u{202A}', '\u{202B}', '\u{202C}', '\u{202D}',
|
||||
'\u{202E}', '\u{2066}', '\u{2067}', '\u{2068}', '\u{2069}',
|
||||
] {
|
||||
assert_eq!(
|
||||
sanitize(&format!("a{c}b")),
|
||||
format!("a{r}b"),
|
||||
"U+{:04X} was not substituted",
|
||||
c as u32
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_sanitize_preserves_arabic_sharing_the_alm_lead_byte() {
|
||||
// U+061C is reached via lead byte 0xD8, which also leads ordinary Arabic.
|
||||
assert_eq!(
|
||||
sanitize("\u{0600}\u{061F}\u{06FF}"),
|
||||
"\u{0600}\u{061F}\u{06FF}"
|
||||
);
|
||||
assert_eq!(sanitize("مرحبا بالعالم"), "مرحبا بالعالم");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_sanitize_preserves_legitimate_bytes() {
|
||||
assert_eq!(sanitize("crlf\r\nline\r\n"), "crlf\r\nline\r\n");
|
||||
assert_eq!(sanitize("a\tb\nc"), "a\tb\nc");
|
||||
assert_eq!(sanitize("plain ascii"), "plain ascii");
|
||||
assert_eq!(sanitize("üñíçödé"), "üñíçödé");
|
||||
// FF (U+000C) passes through; section separator in C source / Emacs Lisp.
|
||||
assert_eq!(sanitize("section1\x0Csection2"), "section1\x0Csection2");
|
||||
// Common Unicode that shares a UTF-8 lead byte with dangerous codepoints
|
||||
// must pass through unchanged.
|
||||
assert_eq!(sanitize("snowman ☃ moon ☾"), "snowman ☃ moon ☾");
|
||||
assert_eq!(sanitize("emoji 🎉 ☃"), "emoji 🎉 ☃");
|
||||
assert_eq!(sanitize("0xC2 lead: ÿ ñ ç"), "0xC2 lead: ÿ ñ ç");
|
||||
assert_eq!(sanitize("CJK 漢字 emoji 🦀"), "CJK 漢字 emoji 🦀");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_sanitize_strips_ansi() {
|
||||
let r = '\u{FFFD}';
|
||||
// sanitize is a strict superset of strip_ansi.
|
||||
assert_eq!(sanitize("a\x1B[31mb\x1B[0mc"), "abc");
|
||||
assert_eq!(sanitize("a\u{9B}31mb\u{9B}0mc"), "abc");
|
||||
// ANSI then dangerous byte: ANSI gone, byte substituted.
|
||||
assert_eq!(sanitize("\x1B[31mhello\rEVIL"), format!("hello{r}EVIL"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_strip_overstrike() {
|
||||
// Bold: X\x08X (same char repeated)
|
||||
|
||||
@@ -192,6 +192,15 @@ impl<'a> PrettyPrinter<'a> {
|
||||
self
|
||||
}
|
||||
|
||||
/// Whether to sanitize untrusted input for safe display (default: never)
|
||||
///
|
||||
/// Strips ANSI escape sequences and additionally substitutes terminal-active
|
||||
/// control bytes and bidi / zero-width codepoints with U+FFFD.
|
||||
pub fn sanitize(&mut self, mode: StripAnsiMode) -> &mut Self {
|
||||
self.config.sanitize = mode;
|
||||
self
|
||||
}
|
||||
|
||||
/// Text wrapping mode (default: do not wrap)
|
||||
pub fn wrapping_mode(&mut self, mode: WrappingMode) -> &mut Self {
|
||||
self.config.wrapping_mode = mode;
|
||||
|
||||
+27
-5
@@ -30,7 +30,8 @@ use crate::input::OpenedInput;
|
||||
use crate::line_range::{MaxBufferedLineNumber, RangeCheckResult};
|
||||
use crate::output::OutputHandle;
|
||||
use crate::preprocessor::{
|
||||
expand_tabs, replace_nonprintable, sanitize_for_terminal, strip_ansi, strip_overstrike,
|
||||
expand_tabs, replace_nonprintable, sanitize, sanitize_for_terminal, strip_ansi,
|
||||
strip_overstrike,
|
||||
};
|
||||
use crate::style::StyleComponent;
|
||||
use crate::terminal::{as_terminal_escaped, to_ansi_color};
|
||||
@@ -210,6 +211,7 @@ pub(crate) struct InteractivePrinter<'a> {
|
||||
background_color_highlight: Option<Color>,
|
||||
consecutive_empty_lines: usize,
|
||||
strip_ansi: bool,
|
||||
sanitize: bool,
|
||||
strip_overstrike: bool,
|
||||
}
|
||||
|
||||
@@ -273,7 +275,9 @@ impl<'a> InteractivePrinter<'a> {
|
||||
|
||||
let needs_to_match_syntax = (!is_printing_binary
|
||||
|| matches!(config.binary, BinaryBehavior::AsText))
|
||||
&& (config.colored_output || config.strip_ansi == StripAnsiMode::Auto);
|
||||
&& (config.colored_output
|
||||
|| config.strip_ansi == StripAnsiMode::Auto
|
||||
|| config.sanitize == StripAnsiMode::Auto);
|
||||
|
||||
let (is_plain_text, strip_overstrike, highlighter_from_set) = if needs_to_match_syntax {
|
||||
// Determine the type of syntax for highlighting
|
||||
@@ -319,6 +323,14 @@ impl<'a> InteractivePrinter<'a> {
|
||||
_ => false,
|
||||
};
|
||||
|
||||
let sanitize = match config.sanitize {
|
||||
_ if config.show_nonprintable => false,
|
||||
StripAnsiMode::Always => true,
|
||||
StripAnsiMode::Auto if is_plain_text => false,
|
||||
StripAnsiMode::Auto => true,
|
||||
_ => false,
|
||||
};
|
||||
|
||||
Ok(InteractivePrinter {
|
||||
panel_width,
|
||||
colors,
|
||||
@@ -332,6 +344,7 @@ impl<'a> InteractivePrinter<'a> {
|
||||
background_color_highlight,
|
||||
consecutive_empty_lines: 0,
|
||||
strip_ansi,
|
||||
sanitize,
|
||||
strip_overstrike,
|
||||
})
|
||||
}
|
||||
@@ -675,8 +688,10 @@ impl Printer for InteractivePrinter<'_> {
|
||||
}
|
||||
}
|
||||
|
||||
// If ANSI escape sequences are supposed to be stripped, do it before syntax highlighting.
|
||||
if self.strip_ansi {
|
||||
// Sanitize is the strict superset; otherwise strip-ansi alone.
|
||||
if self.sanitize {
|
||||
line = sanitize(&line).into()
|
||||
} else if self.strip_ansi {
|
||||
line = strip_ansi(&line).into()
|
||||
}
|
||||
|
||||
@@ -723,10 +738,17 @@ impl Printer for InteractivePrinter<'_> {
|
||||
|
||||
// Line decorations.
|
||||
if self.panel_width > 0 {
|
||||
let display_line_number =
|
||||
if self.config.number_nonblank && line.trim_end_matches(['\r', '\n']).is_empty() {
|
||||
0
|
||||
} else {
|
||||
line_number
|
||||
};
|
||||
|
||||
let decorations = self
|
||||
.decorations
|
||||
.iter()
|
||||
.map(|d| d.generate(line_number, false, self));
|
||||
.map(|d| d.generate(display_line_number, false, self));
|
||||
|
||||
for deco in decorations {
|
||||
write!(handle, "{} ", deco.text)?;
|
||||
|
||||
+202
-12
@@ -386,12 +386,23 @@ impl<'a> EscapeSequenceOffsetsIterator<'a> {
|
||||
}
|
||||
|
||||
fn next_text(&mut self) -> Option<EscapeSequenceOffsets> {
|
||||
self.chars_take_while(|c| c != '\x1B')
|
||||
self.chars_take_while(|c| !is_sequence_introducer(c))
|
||||
.map(|(start, end)| EscapeSequenceOffsets::Text { start, end })
|
||||
}
|
||||
|
||||
fn next_sequence(&mut self) -> Option<EscapeSequenceOffsets> {
|
||||
let (start_sequence, c) = self.chars.next().expect("to not be finished");
|
||||
|
||||
// Handle 8-bit C1 introducers as their 7-bit `ESC <x>` equivalents.
|
||||
match c {
|
||||
'\u{9B}' => return self.next_csi_body(start_sequence),
|
||||
'\u{9D}' => return self.next_osc_body(start_sequence),
|
||||
'\u{90}' | '\u{98}' | '\u{9E}' | '\u{9F}' => {
|
||||
return self.next_string_terminated_body(start_sequence)
|
||||
}
|
||||
_ => {}
|
||||
}
|
||||
|
||||
match self.chars.peek() {
|
||||
None => Some(EscapeSequenceOffsets::Unknown {
|
||||
start: start_sequence,
|
||||
@@ -400,12 +411,24 @@ impl<'a> EscapeSequenceOffsetsIterator<'a> {
|
||||
|
||||
Some((_, ']')) => self.next_osc(start_sequence),
|
||||
Some((_, '[')) => self.next_csi(start_sequence),
|
||||
Some((i, c)) => match c {
|
||||
// 7-bit DCS/SOS/PM/APC: `ESC P/X/^/_` introduces a string body.
|
||||
Some((_, 'P' | 'X' | '^' | '_')) => {
|
||||
self.chars.next();
|
||||
self.next_string_terminated_body(start_sequence)
|
||||
}
|
||||
Some(&(i, c)) => match c {
|
||||
'\x20'..='\x2F' => self.next_nf(start_sequence),
|
||||
c => Some(EscapeSequenceOffsets::Unknown {
|
||||
c => {
|
||||
// Single-byte ESC sequence (RIS, DECSC/DECRC, keypad, VT52 etc.).
|
||||
let end = match self.chars.next() {
|
||||
Some((j, fc)) => j + fc.len_utf8(),
|
||||
None => i + c.len_utf8(),
|
||||
};
|
||||
Some(EscapeSequenceOffsets::Unknown {
|
||||
start: start_sequence,
|
||||
end: i + c.len_utf8(),
|
||||
}),
|
||||
end,
|
||||
})
|
||||
}
|
||||
},
|
||||
}
|
||||
}
|
||||
@@ -413,12 +436,27 @@ impl<'a> EscapeSequenceOffsetsIterator<'a> {
|
||||
fn next_osc(&mut self, start_sequence: usize) -> Option<EscapeSequenceOffsets> {
|
||||
let (osc_open_index, osc_open_char) = self.chars.next().expect("to not be finished");
|
||||
debug_assert_eq!(osc_open_char, ']');
|
||||
let start_command = osc_open_index + osc_open_char.len_utf8();
|
||||
Some(self.read_osc_body(start_sequence, start_command))
|
||||
}
|
||||
|
||||
/// OSC body parser entered after the 8-bit introducer U+009D was consumed.
|
||||
fn next_osc_body(&mut self, start_sequence: usize) -> Option<EscapeSequenceOffsets> {
|
||||
let start_command = start_sequence + '\u{9D}'.len_utf8();
|
||||
Some(self.read_osc_body(start_sequence, start_command))
|
||||
}
|
||||
|
||||
fn read_osc_body(
|
||||
&mut self,
|
||||
start_sequence: usize,
|
||||
start_command: usize,
|
||||
) -> EscapeSequenceOffsets {
|
||||
let mut start_terminator: usize;
|
||||
let mut end_sequence: usize;
|
||||
|
||||
loop {
|
||||
match self.chars_take_while(|c| !matches!(c, '\x07' | '\x1B')) {
|
||||
// ST is BEL, ESC `\\`, or U+009C.
|
||||
match self.chars_take_while(|c| !matches!(c, '\x07' | '\x1B' | '\u{9C}')) {
|
||||
None => {
|
||||
start_terminator = self.text.len();
|
||||
end_sequence = start_terminator;
|
||||
@@ -437,6 +475,11 @@ impl<'a> EscapeSequenceOffsetsIterator<'a> {
|
||||
break;
|
||||
}
|
||||
|
||||
Some((ti, '\u{9C}')) => {
|
||||
end_sequence = ti + '\u{9C}'.len_utf8();
|
||||
break;
|
||||
}
|
||||
|
||||
Some((ti, '\x1B')) => {
|
||||
match self.chars.next() {
|
||||
Some((i, '\\')) => {
|
||||
@@ -466,20 +509,82 @@ impl<'a> EscapeSequenceOffsetsIterator<'a> {
|
||||
}
|
||||
}
|
||||
|
||||
Some(EscapeSequenceOffsets::OSC {
|
||||
EscapeSequenceOffsets::OSC {
|
||||
start_sequence,
|
||||
start_command: osc_open_index + osc_open_char.len_utf8(),
|
||||
start_command,
|
||||
start_terminator,
|
||||
end: end_sequence,
|
||||
}
|
||||
}
|
||||
|
||||
/// DCS/SOS/PM/APC body parser. Emitted as `Unknown` so the body is stripped.
|
||||
fn next_string_terminated_body(
|
||||
&mut self,
|
||||
start_sequence: usize,
|
||||
) -> Option<EscapeSequenceOffsets> {
|
||||
let mut end_sequence: usize;
|
||||
|
||||
loop {
|
||||
match self.chars_take_while(|c| !matches!(c, '\x07' | '\x1B' | '\u{9C}')) {
|
||||
None => {
|
||||
end_sequence = self.text.len();
|
||||
break;
|
||||
}
|
||||
Some((_, end)) => {
|
||||
end_sequence = end;
|
||||
}
|
||||
}
|
||||
|
||||
match self.chars.next() {
|
||||
Some((ti, '\x07')) => {
|
||||
end_sequence = ti + '\x07'.len_utf8();
|
||||
break;
|
||||
}
|
||||
Some((ti, '\u{9C}')) => {
|
||||
end_sequence = ti + '\u{9C}'.len_utf8();
|
||||
break;
|
||||
}
|
||||
Some((ti, '\x1B')) => match self.chars.next() {
|
||||
Some((i, '\\')) => {
|
||||
end_sequence = i + '\\'.len_utf8();
|
||||
break;
|
||||
}
|
||||
None => {
|
||||
end_sequence = ti + '\x1B'.len_utf8();
|
||||
break;
|
||||
}
|
||||
_ => {}
|
||||
},
|
||||
None => break,
|
||||
Some((_, tc)) => {
|
||||
panic!("this should not be reached: char {tc:?}")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Some(EscapeSequenceOffsets::Unknown {
|
||||
start: start_sequence,
|
||||
end: end_sequence,
|
||||
})
|
||||
}
|
||||
|
||||
fn next_csi(&mut self, start_sequence: usize) -> Option<EscapeSequenceOffsets> {
|
||||
let (csi_open_index, csi_open_char) = self.chars.next().expect("to not be finished");
|
||||
debug_assert_eq!(csi_open_char, '[');
|
||||
Some(self.read_csi_body(start_sequence, csi_open_index + csi_open_char.len_utf8()))
|
||||
}
|
||||
|
||||
let start_parameters: usize = csi_open_index + csi_open_char.len_utf8();
|
||||
/// CSI body parser entered after the 8-bit introducer U+009B was consumed.
|
||||
fn next_csi_body(&mut self, start_sequence: usize) -> Option<EscapeSequenceOffsets> {
|
||||
let start_parameters = start_sequence + '\u{9B}'.len_utf8();
|
||||
Some(self.read_csi_body(start_sequence, start_parameters))
|
||||
}
|
||||
|
||||
fn read_csi_body(
|
||||
&mut self,
|
||||
start_sequence: usize,
|
||||
start_parameters: usize,
|
||||
) -> EscapeSequenceOffsets {
|
||||
// Keep iterating while within the range of `0x30-0x3F`.
|
||||
let mut start_intermediates: usize = start_parameters;
|
||||
if let Some((_, end)) = self.chars_take_while(|c| matches!(c, '\x30'..='\x3F')) {
|
||||
@@ -498,13 +603,13 @@ impl<'a> EscapeSequenceOffsetsIterator<'a> {
|
||||
Some((i, c)) => i + c.len_utf8(),
|
||||
};
|
||||
|
||||
Some(EscapeSequenceOffsets::CSI {
|
||||
EscapeSequenceOffsets::CSI {
|
||||
start_sequence,
|
||||
start_parameters,
|
||||
start_intermediates,
|
||||
start_final_byte,
|
||||
end: end_of_sequence,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
fn next_nf(&mut self, start_sequence: usize) -> Option<EscapeSequenceOffsets> {
|
||||
@@ -543,13 +648,25 @@ impl Iterator for EscapeSequenceOffsetsIterator<'_> {
|
||||
type Item = EscapeSequenceOffsets;
|
||||
fn next(&mut self) -> Option<Self::Item> {
|
||||
match self.chars.peek() {
|
||||
Some((_, '\x1B')) => self.next_sequence(),
|
||||
Some((_, c)) if is_sequence_introducer(*c) => self.next_sequence(),
|
||||
Some((_, _)) => self.next_text(),
|
||||
None => None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// True for ESC and the 8-bit C1 sequence introducers (DCS/SOS/CSI/OSC/PM/APC).
|
||||
#[inline]
|
||||
fn is_sequence_introducer(c: char) -> bool {
|
||||
if (c as u32) < 0x80 {
|
||||
return c == '\x1B';
|
||||
}
|
||||
matches!(
|
||||
c,
|
||||
'\u{90}' | '\u{98}' | '\u{9B}' | '\u{9D}' | '\u{9E}' | '\u{9F}'
|
||||
)
|
||||
}
|
||||
|
||||
/// An iterator over ANSI/VT escape sequences within a string.
|
||||
///
|
||||
/// ## Example
|
||||
@@ -717,6 +834,79 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_escape_sequence_offsets_iterator_parses_8bit_csi() {
|
||||
let mut iter = EscapeSequenceOffsetsIterator::new("\u{9B}31m");
|
||||
assert_eq!(
|
||||
iter.next(),
|
||||
Some(EscapeSequenceOffsets::CSI {
|
||||
start_sequence: 0,
|
||||
start_parameters: 2,
|
||||
start_intermediates: 4,
|
||||
start_final_byte: 4,
|
||||
end: 5,
|
||||
})
|
||||
);
|
||||
assert_eq!(iter.next(), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_escape_sequence_offsets_iterator_parses_8bit_osc_with_8bit_st() {
|
||||
let mut iter = EscapeSequenceOffsetsIterator::new("\u{9D}0;title\u{9C}");
|
||||
assert_eq!(
|
||||
iter.next(),
|
||||
Some(EscapeSequenceOffsets::OSC {
|
||||
start_sequence: 0,
|
||||
start_command: 2,
|
||||
start_terminator: 9,
|
||||
end: 11,
|
||||
})
|
||||
);
|
||||
assert_eq!(iter.next(), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_escape_sequence_offsets_iterator_parses_7bit_dcs_consumes_body() {
|
||||
let mut iter = EscapeSequenceOffsetsIterator::new("\x1BP1;0;|payload\x1B\\rest");
|
||||
assert_eq!(
|
||||
iter.next(),
|
||||
Some(EscapeSequenceOffsets::Unknown { start: 0, end: 16 })
|
||||
);
|
||||
assert_eq!(
|
||||
iter.next(),
|
||||
Some(EscapeSequenceOffsets::Text { start: 16, end: 20 })
|
||||
);
|
||||
assert_eq!(iter.next(), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_escape_sequence_offsets_iterator_8bit_dcs_consumes_body() {
|
||||
let mut iter = EscapeSequenceOffsetsIterator::new("\u{90}body\u{9C}rest");
|
||||
assert_eq!(
|
||||
iter.next(),
|
||||
Some(EscapeSequenceOffsets::Unknown { start: 0, end: 8 })
|
||||
);
|
||||
assert_eq!(
|
||||
iter.next(),
|
||||
Some(EscapeSequenceOffsets::Text { start: 8, end: 12 })
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_escape_sequence_offsets_iterator_truncated_dcs_consumes_to_eof() {
|
||||
// Unterminated DCS must still be consumed, not emitted as `Text`.
|
||||
let input = "\x1BPno-terminator";
|
||||
let mut iter = EscapeSequenceOffsetsIterator::new(input);
|
||||
assert_eq!(
|
||||
iter.next(),
|
||||
Some(EscapeSequenceOffsets::Unknown {
|
||||
start: 0,
|
||||
end: input.len(),
|
||||
})
|
||||
);
|
||||
assert_eq!(iter.next(), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_escape_sequence_offsets_iterator_parses_csi() {
|
||||
let mut iter = EscapeSequenceOffsetsIterator::new("\x1B[m");
|
||||
|
||||
+361
-1
@@ -213,6 +213,69 @@ fn numbers_honored_from_cli_when_preceeded_by_plain_in_loop_through_mode() {
|
||||
.stdout(" 1 line 1\n 2 line 2\n 3 line 3\n 4 line 4\n 5 line 5\n 6 line 6\n 7 line 7\n 8 line 8\n 9 line 9\n 10 line 10\n");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn number_nonblank_style() {
|
||||
bat()
|
||||
.arg("empty_lines.txt")
|
||||
.arg("-b")
|
||||
.arg("--decorations=always")
|
||||
.assert()
|
||||
.success()
|
||||
.stdout(" 1 line 1\n \n \n \n 2 line 5\n \n \n \n \n \n \n \n \n \n \n \n \n \n \n 3 line 20\n 4 line 21\n \n \n 5 line 24\n \n 6 line 26\n \n \n \n 7 line 30\n");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn number_nonblank_from_cli_in_loop_through_mode() {
|
||||
bat()
|
||||
.arg("empty_lines.txt")
|
||||
.arg("-b")
|
||||
.assert()
|
||||
.success()
|
||||
.stdout(" 1 line 1\n \n \n \n 2 line 5\n \n \n \n \n \n \n \n \n \n \n \n \n \n \n 3 line 20\n 4 line 21\n \n \n 5 line 24\n \n 6 line 26\n \n \n \n 7 line 30\n");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn number_nonblank_takes_precedence_over_number() {
|
||||
// -bn should behave like -b
|
||||
bat()
|
||||
.arg("empty_lines.txt")
|
||||
.arg("-bn")
|
||||
.arg("--decorations=always")
|
||||
.assert()
|
||||
.success()
|
||||
.stdout(" 1 line 1\n \n \n \n 2 line 5\n \n \n \n \n \n \n \n \n \n \n \n \n \n \n 3 line 20\n 4 line 21\n \n \n 5 line 24\n \n 6 line 26\n \n \n \n 7 line 30\n");
|
||||
|
||||
// -nb should also behave like -b
|
||||
bat()
|
||||
.arg("empty_lines.txt")
|
||||
.arg("-nb")
|
||||
.arg("--decorations=always")
|
||||
.assert()
|
||||
.success()
|
||||
.stdout(" 1 line 1\n \n \n \n 2 line 5\n \n \n \n \n \n \n \n \n \n \n \n \n \n \n 3 line 20\n 4 line 21\n \n \n 5 line 24\n \n 6 line 26\n \n \n \n 7 line 30\n");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn number_nonblank_ignored_when_followed_by_plain() {
|
||||
bat()
|
||||
.arg("empty_lines.txt")
|
||||
.arg("-bp")
|
||||
.arg("--decorations=auto")
|
||||
.assert()
|
||||
.success()
|
||||
.stdout("line 1\n\n\n\nline 5\n\n\n\n\n\n\n\n\n\n\n\n\n\n\nline 20\nline 21\n\n\nline 24\n\nline 26\n\n\n\nline 30\n");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn piped_output_with_number_nonblank_flag() {
|
||||
bat()
|
||||
.arg("-b")
|
||||
.write_stdin("hello\n\nworld\n")
|
||||
.assert()
|
||||
.success()
|
||||
.stdout(" 1 hello\n \n 2 world\n");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn line_range_2_3() {
|
||||
bat()
|
||||
@@ -626,12 +689,18 @@ fn list_themes_to_piped_output() {
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[serial]
|
||||
fn list_languages() {
|
||||
mocked_pagers::with_mocked_versions_of_more_and_most_in_path(|| {
|
||||
bat()
|
||||
.env("PAGER", mocked_pagers::from("echo pager-output"))
|
||||
.arg("--list-languages")
|
||||
.arg("--paging=never")
|
||||
.assert()
|
||||
.success()
|
||||
.stdout(predicate::str::contains("Rust").normalize());
|
||||
.stdout(predicate::str::contains("Rust").normalize())
|
||||
.stdout(predicate::str::contains("pager-output").not());
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -2553,6 +2622,145 @@ fn no_first_line_fallback_when_mapping_to_invalid_syntax() {
|
||||
.stderr(predicate::str::contains("unknown syntax: 'InvalidSyntax'"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn stdin_detects_bash_from_first_line() {
|
||||
let content = "#!/bin/bash\necho hi\n";
|
||||
|
||||
let detected_output = bat()
|
||||
.arg("--color=always")
|
||||
.arg("--style=plain")
|
||||
.write_stdin(content)
|
||||
.assert()
|
||||
.success()
|
||||
.get_output()
|
||||
.stdout
|
||||
.clone();
|
||||
|
||||
let explicit_output = bat()
|
||||
.arg("--color=always")
|
||||
.arg("--style=plain")
|
||||
.arg("--language=bash")
|
||||
.write_stdin(content)
|
||||
.assert()
|
||||
.success()
|
||||
.get_output()
|
||||
.stdout
|
||||
.clone();
|
||||
|
||||
assert_eq!(
|
||||
from_utf8(&detected_output).expect("output is valid utf-8"),
|
||||
from_utf8(&explicit_output).expect("output is valid utf-8")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn stdin_detects_diff_from_first_line() {
|
||||
let content = "diff --git a/x b/y\n--- a/x\n+++ b/y\n@@ -1 +1 @@\n-old\n+new\n";
|
||||
|
||||
let detected_output = bat()
|
||||
.arg("--color=always")
|
||||
.arg("--style=plain")
|
||||
.write_stdin(content)
|
||||
.assert()
|
||||
.success()
|
||||
.get_output()
|
||||
.stdout
|
||||
.clone();
|
||||
|
||||
let explicit_output = bat()
|
||||
.arg("--color=always")
|
||||
.arg("--style=plain")
|
||||
.arg("--language=diff")
|
||||
.write_stdin(content)
|
||||
.assert()
|
||||
.success()
|
||||
.get_output()
|
||||
.stdout
|
||||
.clone();
|
||||
|
||||
assert_eq!(
|
||||
from_utf8(&detected_output).expect("output is valid utf-8"),
|
||||
from_utf8(&explicit_output).expect("output is valid utf-8")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn empty_stdin_does_not_detect_syntax() {
|
||||
bat()
|
||||
.arg("--color=always")
|
||||
.arg("--style=plain")
|
||||
.write_stdin("")
|
||||
.assert()
|
||||
.success()
|
||||
.stdout("")
|
||||
.stderr("");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn binary_stdin_does_not_detect_syntax_from_invalid_utf8_first_line() {
|
||||
let content = b"#!/bin/bash\xff\necho hi\n";
|
||||
|
||||
let detected_output = bat()
|
||||
.arg("--binary=as-text")
|
||||
.arg("--color=always")
|
||||
.arg("--style=plain")
|
||||
.write_stdin(content.as_slice())
|
||||
.assert()
|
||||
.success()
|
||||
.get_output()
|
||||
.stdout
|
||||
.clone();
|
||||
|
||||
let plain_text_output = bat()
|
||||
.arg("--binary=as-text")
|
||||
.arg("--color=always")
|
||||
.arg("--style=plain")
|
||||
.arg("--language=txt")
|
||||
.write_stdin(content.as_slice())
|
||||
.assert()
|
||||
.success()
|
||||
.get_output()
|
||||
.stdout
|
||||
.clone();
|
||||
|
||||
assert_eq!(
|
||||
from_utf8(&detected_output).expect("output is valid utf-8"),
|
||||
from_utf8(&plain_text_output).expect("output is valid utf-8")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn explicit_language_overrides_stdin_first_line_detection() {
|
||||
let content = "diff --git a/x b/y\n--- a/x\n+++ b/y\n@@ -1 +1 @@\n-old\n+new\n";
|
||||
|
||||
let detected_output = bat()
|
||||
.arg("--color=always")
|
||||
.arg("--style=plain")
|
||||
.write_stdin(content)
|
||||
.assert()
|
||||
.success()
|
||||
.get_output()
|
||||
.stdout
|
||||
.clone();
|
||||
|
||||
let explicit_output = bat()
|
||||
.arg("--color=always")
|
||||
.arg("--style=plain")
|
||||
.arg("--language=json")
|
||||
.arg("-")
|
||||
.write_stdin(content)
|
||||
.assert()
|
||||
.success()
|
||||
.get_output()
|
||||
.stdout
|
||||
.clone();
|
||||
|
||||
assert_ne!(
|
||||
from_utf8(&detected_output).expect("output is valid utf-8"),
|
||||
from_utf8(&explicit_output).expect("output is valid utf-8")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn fallback_syntax_is_used_when_no_syntax_is_detected() {
|
||||
let content = "# comment\nfoo=bar\n";
|
||||
@@ -3844,6 +4052,158 @@ fn strip_ansi_auto_does_not_strip_ansi_when_plain_text_by_option() {
|
||||
assert!(output.contains("\x1B[33mYellow"))
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sanitize_implies_strip_ansi() {
|
||||
bat()
|
||||
.arg("--style=plain")
|
||||
.arg("--decorations=always")
|
||||
.arg("--color=never")
|
||||
.arg("--sanitize=always")
|
||||
.write_stdin("\x1B[33mYellow\x1B[m")
|
||||
.assert()
|
||||
.success()
|
||||
.stdout("Yellow");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sanitize_strips_osc_clipboard_hijack() {
|
||||
// OSC 52 sets the system clipboard. A file containing this would silently
|
||||
// overwrite the user's clipboard if displayed unfiltered.
|
||||
bat()
|
||||
.arg("--style=plain")
|
||||
.arg("--decorations=always")
|
||||
.arg("--color=never")
|
||||
.arg("--sanitize=always")
|
||||
.write_stdin("safe\x1B]52;c;cm0=\x07payload")
|
||||
.assert()
|
||||
.success()
|
||||
.stdout("safepayload");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sanitize_strips_osc_8_hyperlink_spoof() {
|
||||
// OSC 8 hyperlinks let displayed text point to an arbitrary URL.
|
||||
bat()
|
||||
.arg("--style=plain")
|
||||
.arg("--decorations=always")
|
||||
.arg("--color=never")
|
||||
.arg("--sanitize=always")
|
||||
.write_stdin("\x1B]8;;https://evil.example\x07click here\x1B]8;;\x07")
|
||||
.assert()
|
||||
.success()
|
||||
.stdout("click here");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sanitize_strips_window_title_injection() {
|
||||
// OSC 0/1/2 set the terminal window title.
|
||||
bat()
|
||||
.arg("--style=plain")
|
||||
.arg("--decorations=always")
|
||||
.arg("--color=never")
|
||||
.arg("--sanitize=always")
|
||||
.write_stdin("hello\x1B]0;evil-title\x07world")
|
||||
.assert()
|
||||
.success()
|
||||
.stdout("helloworld");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sanitize_strips_8bit_csi() {
|
||||
// 8-bit CSI introducer (U+009B) is the single-codepoint equivalent of ESC [.
|
||||
bat()
|
||||
.arg("--style=plain")
|
||||
.arg("--decorations=always")
|
||||
.arg("--color=never")
|
||||
.arg("--sanitize=always")
|
||||
.write_stdin("a\u{9B}31mRED\u{9B}0mb")
|
||||
.assert()
|
||||
.success()
|
||||
.stdout("aREDb");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sanitize_substitutes_bare_cr() {
|
||||
// Bare CR (not part of CRLF) is the line-overwrite forgery vector.
|
||||
bat()
|
||||
.arg("--style=plain")
|
||||
.arg("--decorations=always")
|
||||
.arg("--color=never")
|
||||
.arg("--sanitize=always")
|
||||
.write_stdin("safe\rEVIL")
|
||||
.assert()
|
||||
.success()
|
||||
.stdout("safe\u{FFFD}EVIL");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sanitize_preserves_crlf() {
|
||||
bat()
|
||||
.arg("--style=plain")
|
||||
.arg("--decorations=always")
|
||||
.arg("--color=never")
|
||||
.arg("--sanitize=always")
|
||||
.write_stdin("line1\r\nline2\r\n")
|
||||
.assert()
|
||||
.success()
|
||||
.stdout("line1\r\nline2\r\n");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sanitize_substitutes_bidi_controls() {
|
||||
// Trojan-Source attack (CVE-2021-42574): U+202E (RLO) reorders display.
|
||||
bat()
|
||||
.arg("--style=plain")
|
||||
.arg("--decorations=always")
|
||||
.arg("--color=never")
|
||||
.arg("--sanitize=always")
|
||||
.write_stdin("admin\u{202E}check")
|
||||
.assert()
|
||||
.success()
|
||||
.stdout("admin\u{FFFD}check");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sanitize_substitutes_zero_width() {
|
||||
// Zero-width chars allow invisible content / identifier confusion.
|
||||
bat()
|
||||
.arg("--style=plain")
|
||||
.arg("--decorations=always")
|
||||
.arg("--color=never")
|
||||
.arg("--sanitize=always")
|
||||
.write_stdin("ad\u{200B}min")
|
||||
.assert()
|
||||
.success()
|
||||
.stdout("ad\u{FFFD}min");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sanitize_preserves_form_feed_in_source() {
|
||||
// FF (U+000C) is used as a section separator in C source and Emacs Lisp.
|
||||
bat()
|
||||
.arg("--style=plain")
|
||||
.arg("--decorations=always")
|
||||
.arg("--color=never")
|
||||
.arg("--sanitize=always")
|
||||
.write_stdin("section1\x0Csection2")
|
||||
.assert()
|
||||
.success()
|
||||
.stdout("section1\x0Csection2");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sanitize_preserves_unicode_text() {
|
||||
bat()
|
||||
.arg("--style=plain")
|
||||
.arg("--decorations=always")
|
||||
.arg("--color=never")
|
||||
.arg("--sanitize=always")
|
||||
.write_stdin("snowman ☃ CJK 漢字 emoji 🦀")
|
||||
.assert()
|
||||
.success()
|
||||
.stdout("snowman ☃ CJK 漢字 emoji 🦀");
|
||||
}
|
||||
|
||||
// Tests that style components can be removed with `-component`.
|
||||
#[test]
|
||||
fn style_components_can_be_removed() {
|
||||
|
||||
Reference in New Issue
Block a user