From e273b6df2b51da1f4d386bc52c72162e9886edca Mon Sep 17 00:00:00 2001 From: darkraider01 Date: Tue, 6 Oct 2026 19:45:24 +0530 Subject: [PATCH 01/10] wc: record current UTF-8 character counting --- src/uu/wc/BENCHMARKING.md | 6 +++ src/uu/wc/benches/wc_bench.rs | 11 +++++ tests/by-util/test_wc.rs | 84 +++++++++++++++++++++++++++++++++-- 3 files changed, 98 insertions(+), 3 deletions(-) diff --git a/src/uu/wc/BENCHMARKING.md b/src/uu/wc/BENCHMARKING.md index ff0c61a3f44..62f852f937e 100644 --- a/src/uu/wc/BENCHMARKING.md +++ b/src/uu/wc/BENCHMARKING.md @@ -80,6 +80,12 @@ candidate as it's fairly large. Use [`hyperfine`](https://github.com/sharkdp/hyperfine) to compare the performance. For example, `hyperfine 'wc somefile' 'uuwc somefile'`. +Run the UTF-8 character-count benchmark with an explicit UTF-8 locale: + +```shell +LC_ALL=C.UTF-8 cargo bench -p uu_wc --bench wc_bench -- wc_chars_utf8 +``` + If you want to get fancy and exhaustive, generate a table: | | moby64.txt | odyssey256.txt | 25Mshortlines | /usr/bin/docker | diff --git a/src/uu/wc/benches/wc_bench.rs b/src/uu/wc/benches/wc_bench.rs index 6abcb5efd4a..226490de35c 100644 --- a/src/uu/wc/benches/wc_bench.rs +++ b/src/uu/wc/benches/wc_bench.rs @@ -78,6 +78,17 @@ fn wc_chars_large_line_count(bencher: Bencher, num_lines: usize) { .bench_values(|args| black_box(uumain(args))); } +#[divan::bench(args = [100_000])] +fn wc_chars_utf8(bencher: Bencher, num_lines: usize) { + let temp_dir = tempfile::tempdir().unwrap(); + let data = "hello ä € 💩\n".repeat(num_lines); + let file_path = create_test_file(data.as_bytes(), temp_dir.path()); + + bencher + .with_inputs(|| get_bench_args(&[&"-cm", &file_path]).into_iter()) + .bench_values(|args| black_box(uumain(args))); +} + /// Benchmark word counting on large line counts #[divan::bench(args = [100_000])] fn wc_words_large_line_count(bencher: Bencher, num_lines: usize) { diff --git a/tests/by-util/test_wc.rs b/tests/by-util/test_wc.rs index 16fd79293f5..62ed8becf9e 100644 --- a/tests/by-util/test_wc.rs +++ b/tests/by-util/test_wc.rs @@ -5,10 +5,8 @@ // spell-checker:ignore (flags) lwmcL clmwL ; (path) bogusfile emptyfile manyemptylines moby notrailingnewline onelongemptyline onelongword weirdchars ioerrdir -#[cfg(unix)] -use uutests::at_and_ucmd; -use uutests::new_ucmd; use uutests::util::vec_of_size; +use uutests::{at_and_ucmd, new_ucmd}; #[test] fn test_invalid_arg() { @@ -123,6 +121,86 @@ fn test_utf8_bytes_chars() { .stdout_is(" 442 513\n"); } +#[test] +fn test_utf8_fast_path_counts_sequence_leaders() { + let cases: &[(&[u8], usize)] = &[ + (b"", 0), + (b"\xc3\xa4\n", 2), + (b"\xe2\x82\xac\n", 2), + (b"\xf0\x9f\x92\xa9\n", 2), + (b"\x80\n", 1), + (b"\xff\n", 2), + (b"\xe2\x82", 1), + (b"\xc3\xa4\xff\xe2\x82\xac\n", 4), + (b"a\xc3", 2), + (b"\xc0\xaf\n", 2), + (b"\xed\xa0\x80\n", 2), + ]; + for &(input, chars) in cases { + let lines = bytecount::count(input, b'\n'); + for (flag, expected) in [ + ("-m", format!("{chars}\n")), + ("-cm", format!("{chars:7} {:7}\n", input.len())), + ("-ml", format!("{lines:7} {chars:7}\n")), + ("-cml", format!("{lines:7} {chars:7} {:7}\n", input.len())), + ] { + new_ucmd!() + .env("LC_ALL", "C.UTF-8") + .arg(flag) + .pipe_in(input) + .succeeds() + .stdout_is(expected); + } + } +} + +#[cfg(unix)] +#[test] +fn test_utf8_locale_without_encoding_suffix_counts_sequence_leaders() { + let locale = "en_IN"; + if !uutests::util::is_locale_available(locale) { + return; + } + for variable in ["LC_ALL", "LC_CTYPE", "LANG"] { + new_ucmd!() + .env("LC_ALL", "") + .env("LC_CTYPE", "") + .env("LANG", "C") + .env(variable, locale) + .arg("-cm") + .pipe_in(b"\xc3\xa4\xff\n") + .succeeds() + .stdout_is(" 3 4\n"); + } +} + +#[test] +fn test_utf8_fast_path_counts_sequence_leaders_across_read_buffers() { + let cases: &[(&[u8], usize)] = &[ + (b"\xc3\xa4\xe2\x82\xac\xf0\x9f\x92\xa9\n", 4), + (b"\xe2\x82a\xc3\xa4\n", 4), + (b"\xf0\x9f\x92", 1), + (b"\xed\xa0\x80\xc0\xaf\xff\x80\n", 4), + (b"\xf4\x90\x80\x80\n", 2), + ]; + for prefix in [64 * 1024 - 3, 64 * 1024 - 2, 64 * 1024 - 1] { + for &(suffix, chars) in cases { + let mut input = vec![b'a'; prefix]; + input.extend_from_slice(suffix); + let expected = format!("{:5} {:5} input\n", prefix + chars, input.len()); + for tunables in ["", "glibc.cpu.hwcaps=-AVX2,-SSE2,-ASIMD"] { + let (at, mut ucmd) = at_and_ucmd!(); + at.write_bytes("input", &input); + ucmd.env("LC_ALL", "C.UTF-8") + .env("GLIBC_TUNABLES", tunables) + .args(&["-cm", "input"]) + .succeeds() + .stdout_is(&expected); + } + } + } +} + #[test] fn test_utf8_bytes_lines() { new_ucmd!() From 19df47e0ee08dc3d54556656daad345c6fda731d Mon Sep 17 00:00:00 2001 From: darkraider01 Date: Tue, 6 Oct 2026 19:46:52 +0530 Subject: [PATCH 02/10] wc: count complete valid characters in UTF-8 locales --- Cargo.lock | 7 +++ Cargo.toml | 1 + src/uu/wc/BENCHMARKING.md | 8 ++- src/uu/wc/Cargo.toml | 1 + src/uu/wc/src/count_fast.rs | 107 +++++++++++++++++++++++++++++++++++- src/uu/wc/src/utf8/mod.rs | 7 ++- src/uu/wc/src/wc.rs | 2 +- tests/by-util/test_wc.rs | 28 +++++----- 8 files changed, 139 insertions(+), 22 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index 6b7171ecbf2..04624d8f4ac 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -2747,6 +2747,12 @@ version = "0.3.10" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "3a219298ac11a56ea9a6d2120044824d6f01aeb034955e7af7bc16858527deea" +[[package]] +name = "simdutf8" +version = "0.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e3a9fe34e3e7a50316060351f37187a3f546bce95496156754b601a5fa71b76e" + [[package]] name = "siphasher" version = "1.0.3" @@ -4370,6 +4376,7 @@ dependencies = [ "fluent", "libc", "rustix", + "simdutf8", "tempfile", "thiserror 2.0.21", "unicode-width 0.2.2", diff --git a/Cargo.toml b/Cargo.toml index fa0934d7c18..b50bb34322e 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -119,6 +119,7 @@ rustc-hash = "2.1.1" rustix = { version = "1.1.4", default-features = false } self_cell = "1.0.4" selinux = "0.6" +simdutf8 = "0.1.5" tempfile = "3.15.0" terminal_size = "0.4.0" textwrap = { version = "0.16.1", features = ["terminal_size"] } diff --git a/src/uu/wc/BENCHMARKING.md b/src/uu/wc/BENCHMARKING.md index 62f852f937e..357a8f46005 100644 --- a/src/uu/wc/BENCHMARKING.md +++ b/src/uu/wc/BENCHMARKING.md @@ -27,9 +27,11 @@ output of uutils `cat` (with `splice()` support) into it. If a file is given as ### Counting lines and UTF-8 characters -If the flags set are a subset of `-clm` then the input doesn't have to be decoded. The -input is read in chunks and the `bytecount` crate is used to count the newlines (`-l` flag) -and/or UTF-8 characters (`-m` flag). +If the flags set are a subset of `-clm`, the input is read in chunks and the +`bytecount` crate is used to count newlines (`-l`) and UTF-8 characters (`-m`). +In UTF-8 locales, character counting validates the input with `simdutf8` when SIMD +is enabled, skips malformed sequences, and carries partial characters between +reads. ASCII chunks are counted by length. It's useful to vary the line length in the input. GNU wc seems particularly bad at short lines. diff --git a/src/uu/wc/Cargo.toml b/src/uu/wc/Cargo.toml index be25c406c5a..4d0a588f9a0 100644 --- a/src/uu/wc/Cargo.toml +++ b/src/uu/wc/Cargo.toml @@ -20,6 +20,7 @@ doctest = false bytecount = { workspace = true, features = ["runtime-dispatch-simd"] } clap = { workspace = true } fluent = { workspace = true } +simdutf8 = { workspace = true } thiserror = { workspace = true } unicode-width = { workspace = true } uucore = { workspace = true, features = [ diff --git a/src/uu/wc/src/count_fast.rs b/src/uu/wc/src/count_fast.rs index 82200fe08d9..b6613ef567f 100644 --- a/src/uu/wc/src/count_fast.rs +++ b/src/uu/wc/src/count_fast.rs @@ -3,8 +3,9 @@ // For the full copyright and license information, please view the LICENSE // file that was distributed with this source code. -use crate::{wc_simd_allowed, word_count::WordCount}; +use crate::{utf8::Incomplete, wc_simd_allowed, word_count::WordCount}; use uucore::hardware::SimdPolicy; +use uucore::i18n::{UEncoding, get_ctype_encoding}; use super::WordCountable; @@ -21,6 +22,52 @@ const FILE_ATTRIBUTE_NORMAL: u32 = 128; const BUF_SIZE: usize = 64 * 1024; +fn is_utf8_locale() -> bool { + static UTF8: std::sync::LazyLock = std::sync::LazyLock::new(|| { + #[cfg(any( + target_os = "linux", + target_vendor = "apple", + target_os = "freebsd", + target_os = "dragonfly", + target_os = "openbsd", + target_os = "illumos", + target_os = "solaris", + target_os = "aix", + target_os = "hurd" + ))] + { + use std::ffi::CStr; + // SAFETY: The empty name selects LC_CTYPE from the environment. A null + // base creates an owned locale; no process-wide locale is changed. + let locale = + unsafe { libc::newlocale(libc::LC_CTYPE_MASK, c"".as_ptr(), std::ptr::null_mut()) }; + if !locale.is_null() { + // SAFETY: locale is live. The previous thread locale is restored + // before freeing it, and CODESET's C string is read while it is live. + let result = unsafe { + let previous = libc::uselocale(locale); + let result = if previous.is_null() { + None + } else { + let codeset = libc::nl_langinfo(libc::CODESET); + let utf8 = !codeset.is_null() + && matches!(CStr::from_ptr(codeset).to_bytes(), b"UTF-8" | b"UTF8"); + libc::uselocale(previous); + Some(utf8) + }; + libc::freelocale(locale); + result + }; + if let Some(utf8) = result { + return utf8; + } + } + } + get_ctype_encoding() == UEncoding::Utf8 + }); + *UTF8 +} + /// This is a Linux-specific function to count the number of bytes using the /// `splice` system call, which is faster than using `read`. /// @@ -177,6 +224,58 @@ impl Default for AlignedBuffer { } } +// incomplete holds only a potentially valid UTF-8 suffix from the preceding read. +fn count_utf8_chars(mut input: &[u8], incomplete: &mut Incomplete, simd_allowed: bool) -> usize { + let count_chars = |bytes: &[u8]| { + if simd_allowed { + bytecount::num_chars(bytes) + } else { + bytecount::naive_num_chars(bytes) + } + }; + let mut chars = 0; + if !incomplete.is_empty() { + let (consumed, result) = incomplete.try_complete_offsets(input); + input = &input[consumed..]; + match result { + Some(Ok(())) => chars += count_chars(incomplete.take_buffer()), + Some(Err(())) => { + incomplete.take_buffer(); + } + None => return 0, + } + } + if input.is_ascii() { + return chars + input.len(); + } + let mut validate_with_simd = simd_allowed; + while !input.is_empty() { + let result = if validate_with_simd { + simdutf8::compat::from_utf8(input) + .map_err(|error| (error.valid_up_to(), error.error_len())) + } else { + std::str::from_utf8(input).map_err(|error| (error.valid_up_to(), error.error_len())) + }; + match result { + Ok(_) => return chars + count_chars(input), + Err((valid, invalid)) => { + // Repeated SIMD setup is expensive when errors are close together. + validate_with_simd = false; + chars += count_chars(&input[..valid]); + input = &input[valid..]; + if let Some(invalid) = invalid { + input = &input[invalid..]; + } else { + // A trailing partial character is counted only after a later read completes it. + *incomplete = Incomplete::new(input); + break; + } + } + } + } + chars +} + /// Returns a [`WordCount`] that counts the number of bytes, lines, and/or the number of Unicode characters encoded in UTF-8 read via a Reader. /// /// This corresponds to the `-c`, `-l` and `-m` command line flags to wc. @@ -196,6 +295,8 @@ pub(crate) fn count_bytes_chars_and_lines_fast< let buf: &mut [u8] = &mut AlignedBuffer::default().data; let policy = SimdPolicy::detect(); let simd_allowed = wc_simd_allowed(policy); + let validate_utf8 = COUNT_CHARS && is_utf8_locale(); + let mut incomplete = Incomplete::empty(); loop { match handle.read(buf) { Ok(0) => return (total, None), @@ -204,7 +305,9 @@ pub(crate) fn count_bytes_chars_and_lines_fast< total.bytes += n; } if COUNT_CHARS { - total.chars += if simd_allowed { + total.chars += if validate_utf8 { + count_utf8_chars(&buf[..n], &mut incomplete, simd_allowed) + } else if simd_allowed { bytecount::num_chars(&buf[..n]) } else { bytecount::naive_num_chars(&buf[..n]) diff --git a/src/uu/wc/src/utf8/mod.rs b/src/uu/wc/src/utf8/mod.rs index 281cc907bc0..962791a92bc 100644 --- a/src/uu/wc/src/utf8/mod.rs +++ b/src/uu/wc/src/utf8/mod.rs @@ -49,16 +49,19 @@ impl Incomplete { } } - fn take_buffer(&mut self) -> &[u8] { + pub(super) fn take_buffer(&mut self) -> &[u8] { let len = self.buffer_len as usize; self.buffer_len = 0; &self.buffer[..len] } + /// Complete a potentially valid UTF-8 prefix saved from a previous chunk. + /// The caller must drain the buffer with `take_buffer` when the result is `Some`. + /// /// `(consumed_from_input, None)`: not enough input /// `(consumed_from_input, Some(Err(())))`: error bytes in buffer /// `(consumed_from_input, Some(Ok(())))`: UTF-8 string in buffer - fn try_complete_offsets(&mut self, input: &[u8]) -> (usize, Option>) { + pub(super) fn try_complete_offsets(&mut self, input: &[u8]) -> (usize, Option>) { let initial_buffer_len = self.buffer_len as usize; let copied_from_input; { diff --git a/src/uu/wc/src/wc.rs b/src/uu/wc/src/wc.rs index 60d586c873e..cf742bcc5c1 100644 --- a/src/uu/wc/src/wc.rs +++ b/src/uu/wc/src/wc.rs @@ -491,7 +491,7 @@ fn word_count_from_reader( ) } - // Fast paths that can be computed without Unicode decoding. + // Fast paths for byte, character, and line counts. // show_lines (false, false, true, false, false) => { count_bytes_chars_and_lines_fast::<_, false, false, true>(&mut reader) diff --git a/tests/by-util/test_wc.rs b/tests/by-util/test_wc.rs index 62ed8becf9e..35cfef1a4ad 100644 --- a/tests/by-util/test_wc.rs +++ b/tests/by-util/test_wc.rs @@ -122,19 +122,19 @@ fn test_utf8_bytes_chars() { } #[test] -fn test_utf8_fast_path_counts_sequence_leaders() { +fn test_utf8_malformed_sequences_do_not_count_as_characters() { let cases: &[(&[u8], usize)] = &[ (b"", 0), (b"\xc3\xa4\n", 2), (b"\xe2\x82\xac\n", 2), (b"\xf0\x9f\x92\xa9\n", 2), (b"\x80\n", 1), - (b"\xff\n", 2), - (b"\xe2\x82", 1), - (b"\xc3\xa4\xff\xe2\x82\xac\n", 4), - (b"a\xc3", 2), - (b"\xc0\xaf\n", 2), - (b"\xed\xa0\x80\n", 2), + (b"\xff\n", 1), + (b"\xe2\x82", 0), + (b"\xc3\xa4\xff\xe2\x82\xac\n", 3), + (b"a\xc3", 1), + (b"\xc0\xaf\n", 1), + (b"\xed\xa0\x80\n", 1), ]; for &(input, chars) in cases { let lines = bytecount::count(input, b'\n'); @@ -156,7 +156,7 @@ fn test_utf8_fast_path_counts_sequence_leaders() { #[cfg(unix)] #[test] -fn test_utf8_locale_without_encoding_suffix_counts_sequence_leaders() { +fn test_utf8_locale_without_encoding_suffix_validates_characters() { let locale = "en_IN"; if !uutests::util::is_locale_available(locale) { return; @@ -170,18 +170,18 @@ fn test_utf8_locale_without_encoding_suffix_counts_sequence_leaders() { .arg("-cm") .pipe_in(b"\xc3\xa4\xff\n") .succeeds() - .stdout_is(" 3 4\n"); + .stdout_is(" 2 4\n"); } } #[test] -fn test_utf8_fast_path_counts_sequence_leaders_across_read_buffers() { +fn test_utf8_sequences_across_read_buffer_boundaries() { let cases: &[(&[u8], usize)] = &[ (b"\xc3\xa4\xe2\x82\xac\xf0\x9f\x92\xa9\n", 4), - (b"\xe2\x82a\xc3\xa4\n", 4), - (b"\xf0\x9f\x92", 1), - (b"\xed\xa0\x80\xc0\xaf\xff\x80\n", 4), - (b"\xf4\x90\x80\x80\n", 2), + (b"\xe2\x82a\xc3\xa4\n", 3), + (b"\xf0\x9f\x92", 0), + (b"\xed\xa0\x80\xc0\xaf\xff\x80\n", 1), + (b"\xf4\x90\x80\x80\n", 1), ]; for prefix in [64 * 1024 - 3, 64 * 1024 - 2, 64 * 1024 - 1] { for &(suffix, chars) in cases { From 2422b7fc2d00fbb087f652d333d3e9114026d5dd Mon Sep 17 00:00:00 2001 From: darkraider01 Date: Tue, 6 Oct 2026 21:26:25 +0530 Subject: [PATCH 03/10] wc: update fuzz lockfile for UTF-8 validation --- fuzz/Cargo.lock | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/fuzz/Cargo.lock b/fuzz/Cargo.lock index 8ec03377f8d..30d25ac67af 100644 --- a/fuzz/Cargo.lock +++ b/fuzz/Cargo.lock @@ -1436,6 +1436,12 @@ version = "2.0.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba" +[[package]] +name = "simdutf8" +version = "0.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e3a9fe34e3e7a50316060351f37187a3f546bce95496156754b601a5fa71b76e" + [[package]] name = "similar" version = "3.2.0" @@ -1820,6 +1826,7 @@ dependencies = [ "fluent", "libc", "rustix", + "simdutf8", "thiserror", "unicode-width", "uucore", From 2c1591193856e69ce2c7a5918e9ae532cab53a4a Mon Sep 17 00:00:00 2001 From: darkraider01 Date: Tue, 6 Oct 2026 21:26:37 +0530 Subject: [PATCH 04/10] tests: preserve environment overrides in WASI --- tests/by-util/test_wc.rs | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/tests/by-util/test_wc.rs b/tests/by-util/test_wc.rs index 35cfef1a4ad..3645e69ffa8 100644 --- a/tests/by-util/test_wc.rs +++ b/tests/by-util/test_wc.rs @@ -156,6 +156,10 @@ fn test_utf8_malformed_sequences_do_not_count_as_characters() { #[cfg(unix)] #[test] +#[cfg_attr( + wasi_runner, + ignore = "WASI: locale names without encoding suffixes require native locale lookup" +)] fn test_utf8_locale_without_encoding_suffix_validates_characters() { let locale = "en_IN"; if !uutests::util::is_locale_available(locale) { From 0a327ecc7f232177a7ec4dc607db0c08c65ecd3b Mon Sep 17 00:00:00 2001 From: darkraider01 Date: Tue, 6 Oct 2026 20:48:14 +0530 Subject: [PATCH 05/10] wc: record current C locale character counting --- tests/by-util/test_wc.rs | 59 ++++++++++++++++++++++++++++++++++++++++ 1 file changed, 59 insertions(+) diff --git a/tests/by-util/test_wc.rs b/tests/by-util/test_wc.rs index 3645e69ffa8..6101715b17f 100644 --- a/tests/by-util/test_wc.rs +++ b/tests/by-util/test_wc.rs @@ -205,6 +205,65 @@ fn test_utf8_sequences_across_read_buffer_boundaries() { } } +#[test] +fn test_c_locale_multibyte_character_counts() { + // In the C locale, characters should count as bytes. + // Documenting current behavior where multibyte sequences count as 1 character. + let newline_cases: &[(&[u8], usize, usize, usize)] = &[ + (b"\xc3\xa4\n", 1, 1, 2), + (b"\xe2\x82\xac\n", 1, 1, 2), + (b"\xf0\x9f\x92\xa9\n", 1, 1, 2), + (b"hello \xc3\xa4\nworld\n", 2, 3, 14), + ]; + for &(input, lines, words, chars) in newline_cases { + let bytes = input.len(); + for (flag, expected) in [ + ("-m", format!("{chars}\n")), + ("-cm", format!("{chars:7} {bytes:7}\n")), + ("-ml", format!("{lines:7} {chars:7}\n")), + ("-cml", format!("{lines:7} {chars:7} {bytes:7}\n")), + ("-mw", format!("{words:7} {chars:7}\n")), + ("-cmw", format!("{words:7} {chars:7} {bytes:7}\n")), + ] { + new_ucmd!() + .env("LC_ALL", "C") + .arg(flag) + .pipe_in(input) + .succeeds() + .stdout_is(expected); + } + } + + let raw_cases: &[(&[u8], usize, usize)] = &[ + (b"\xc3\xa4", 1, 1), + (b"\xe2\x82\xac", 1, 1), + (b"\xf0\x9f\x92\xa9", 1, 1), + ]; + for &(input, words, chars) in raw_cases { + let bytes = input.len(); + for (flag, expected) in [ + ("-m", format!("{chars}\n")), + ("-cm", format!("{chars:7} {bytes:7}\n")), + ("-mw", format!("{words:7} {chars:7}\n")), + ("-cmw", format!("{words:7} {chars:7} {bytes:7}\n")), + ] { + new_ucmd!() + .env("LC_ALL", "C") + .arg(flag) + .pipe_in(input) + .succeeds() + .stdout_is(expected); + } + } + + let (at, mut ucmd) = at_and_ucmd!(); + at.write_bytes("input", b"\xc3\xa4\n"); + ucmd.env("LC_ALL", "C") + .args(&["-cm", "input"]) + .succeeds() + .stdout_is("2 3 input\n"); +} + #[test] fn test_utf8_bytes_lines() { new_ucmd!() From f9100fb12cea6107b292818ae505fef39bebfd61 Mon Sep 17 00:00:00 2001 From: darkraider01 Date: Tue, 6 Oct 2026 20:54:11 +0530 Subject: [PATCH 06/10] wc: count characters as bytes in single-byte locales --- src/uu/wc/src/count_fast.rs | 198 ++++++++++++++++++++++++++----- src/uu/wc/src/wc.rs | 38 +++++- tests/by-util/test_wc.rs | 227 +++++++++++++++++++++++++++++++++--- 3 files changed, 414 insertions(+), 49 deletions(-) diff --git a/src/uu/wc/src/count_fast.rs b/src/uu/wc/src/count_fast.rs index b6613ef567f..4486a0c7c8a 100644 --- a/src/uu/wc/src/count_fast.rs +++ b/src/uu/wc/src/count_fast.rs @@ -22,8 +22,146 @@ const FILE_ATTRIBUTE_NORMAL: u32 = 128; const BUF_SIZE: usize = 64 * 1024; -fn is_utf8_locale() -> bool { - static UTF8: std::sync::LazyLock = std::sync::LazyLock::new(|| { +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +struct LocaleEncodingInfo { + is_utf8: bool, + is_single_byte: bool, +} + +#[cfg(any( + target_os = "linux", + target_vendor = "apple", + target_os = "freebsd", + target_os = "dragonfly", + target_os = "openbsd", + target_os = "illumos", + target_os = "solaris", + target_os = "aix", + target_os = "hurd" +))] +fn is_known_multibyte_codeset(cs: &[u8]) -> bool { + let lower = cs.to_ascii_lowercase(); + let s = lower.as_slice(); + s.starts_with(b"utf") + || s.starts_with(b"gb18030") + || s.starts_with(b"gbk") + || s.starts_with(b"gb2312") + || s.starts_with(b"big5") + || s.starts_with(b"euc") + || s.starts_with(b"ujis") + || s.starts_with(b"sjis") + || s.starts_with(b"shift_jis") + || s.starts_with(b"shift-jis") + || s.starts_with(b"cp932") + || s.starts_with(b"cp949") + || s.starts_with(b"cp950") + || s.starts_with(b"cp936") +} + +#[cfg(any( + target_os = "linux", + target_vendor = "apple", + target_os = "freebsd", + target_os = "dragonfly", + target_os = "openbsd", + target_os = "illumos", + target_os = "solaris", + target_os = "aix", + target_os = "hurd" +))] +fn detect_native_locale_encoding() -> Option { + use std::ffi::CStr; + + // SAFETY: The empty name selects LC_CTYPE from the environment. A null + // base creates an owned locale; no process-wide locale is changed. + let locale = + unsafe { libc::newlocale(libc::LC_CTYPE_MASK, c"".as_ptr(), std::ptr::null_mut()) }; + if locale.is_null() { + return None; + } + + // SAFETY: locale is live. The previous thread locale is restored + // before freeing it, and locale-dependent queries are executed while it is live. + unsafe { + let previous = libc::uselocale(locale); + if previous.is_null() { + libc::freelocale(locale); + return None; + } + + let codeset = libc::nl_langinfo(libc::CODESET); + let cs = if codeset.is_null() { + None + } else { + Some(CStr::from_ptr(codeset).to_bytes()) + }; + + let is_utf8 = + cs.is_some_and(|b| b.eq_ignore_ascii_case(b"UTF-8") || b.eq_ignore_ascii_case(b"UTF8")); + + let mb_cur_max: Option = { + #[cfg(any(target_os = "linux", target_os = "android"))] + { + unsafe extern "C" { + fn __ctype_get_mb_cur_max() -> libc::size_t; + } + Some(__ctype_get_mb_cur_max() as usize) + } + #[cfg(target_vendor = "apple")] + { + unsafe extern "C" { + fn ___mb_cur_max() -> libc::size_t; + } + Some(___mb_cur_max() as usize) + } + #[cfg(any( + target_os = "freebsd", + target_os = "dragonfly", + target_os = "openbsd", + target_os = "netbsd" + ))] + { + unsafe extern "C" { + static __mb_cur_max: libc::c_int; + } + Some(__mb_cur_max as usize) + } + #[cfg(not(any( + target_os = "linux", + target_os = "android", + target_vendor = "apple", + target_os = "freebsd", + target_os = "dragonfly", + target_os = "openbsd", + target_os = "netbsd" + )))] + { + None + } + }; + + let is_single_byte = if let Some(max) = mb_cur_max { + max == 1 + } else if is_utf8 { + false + } else if let Some(b) = cs { + !is_known_multibyte_codeset(b) + } else { + false + }; + + libc::uselocale(previous); + libc::freelocale(locale); + + Some(LocaleEncodingInfo { + is_utf8, + is_single_byte, + }) + } +} + +fn locale_encoding_info() -> LocaleEncodingInfo { + static INFO: std::sync::LazyLock = std::sync::LazyLock::new(|| { #[cfg(any( target_os = "linux", target_vendor = "apple", @@ -35,37 +173,30 @@ fn is_utf8_locale() -> bool { target_os = "aix", target_os = "hurd" ))] - { - use std::ffi::CStr; - // SAFETY: The empty name selects LC_CTYPE from the environment. A null - // base creates an owned locale; no process-wide locale is changed. - let locale = - unsafe { libc::newlocale(libc::LC_CTYPE_MASK, c"".as_ptr(), std::ptr::null_mut()) }; - if !locale.is_null() { - // SAFETY: locale is live. The previous thread locale is restored - // before freeing it, and CODESET's C string is read while it is live. - let result = unsafe { - let previous = libc::uselocale(locale); - let result = if previous.is_null() { - None - } else { - let codeset = libc::nl_langinfo(libc::CODESET); - let utf8 = !codeset.is_null() - && matches!(CStr::from_ptr(codeset).to_bytes(), b"UTF-8" | b"UTF8"); - libc::uselocale(previous); - Some(utf8) - }; - libc::freelocale(locale); - result - }; - if let Some(utf8) = result { - return utf8; - } - } + if let Some(info) = detect_native_locale_encoding() { + return info; + } + + match get_ctype_encoding() { + UEncoding::Utf8 => LocaleEncodingInfo { + is_utf8: true, + is_single_byte: false, + }, + UEncoding::Ascii => LocaleEncodingInfo { + is_utf8: false, + is_single_byte: true, + }, } - get_ctype_encoding() == UEncoding::Utf8 }); - *UTF8 + *INFO +} + +pub(crate) fn is_utf8_locale() -> bool { + locale_encoding_info().is_utf8 +} + +pub(crate) fn is_single_byte_locale() -> bool { + locale_encoding_info().is_single_byte } /// This is a Linux-specific function to count the number of bytes using the @@ -276,7 +407,7 @@ fn count_utf8_chars(mut input: &[u8], incomplete: &mut Incomplete, simd_allowed: chars } -/// Returns a [`WordCount`] that counts the number of bytes, lines, and/or the number of Unicode characters encoded in UTF-8 read via a Reader. +/// Returns a [`WordCount`] with byte, line, and/or character counts from a Reader. /// /// This corresponds to the `-c`, `-l` and `-m` command line flags to wc. /// @@ -296,6 +427,7 @@ pub(crate) fn count_bytes_chars_and_lines_fast< let policy = SimdPolicy::detect(); let simd_allowed = wc_simd_allowed(policy); let validate_utf8 = COUNT_CHARS && is_utf8_locale(); + let count_chars_as_bytes = COUNT_CHARS && is_single_byte_locale(); let mut incomplete = Incomplete::empty(); loop { match handle.read(buf) { @@ -307,6 +439,8 @@ pub(crate) fn count_bytes_chars_and_lines_fast< if COUNT_CHARS { total.chars += if validate_utf8 { count_utf8_chars(&buf[..n], &mut incomplete, simd_allowed) + } else if count_chars_as_bytes { + n } else if simd_allowed { bytecount::num_chars(&buf[..n]) } else { diff --git a/src/uu/wc/src/wc.rs b/src/uu/wc/src/wc.rs index cf742bcc5c1..dc6a443dd7d 100644 --- a/src/uu/wc/src/wc.rs +++ b/src/uu/wc/src/wc.rs @@ -37,7 +37,7 @@ use uucore::{ }; use crate::{ - count_fast::{count_bytes_chars_and_lines_fast, count_bytes_fast}, + count_fast::{count_bytes_chars_and_lines_fast, count_bytes_fast, is_single_byte_locale}, countable::WordCountable, word_count::WordCount, }; @@ -498,7 +498,18 @@ fn word_count_from_reader( } // show_chars (false, true, false, false, false) => { - count_bytes_chars_and_lines_fast::<_, false, true, false>(&mut reader) + if *IS_SINGLE_BYTE_LOCALE { + let (bytes, error) = count_bytes_fast(&mut reader); + ( + WordCount { + chars: bytes, + ..WordCount::default() + }, + error, + ) + } else { + count_bytes_chars_and_lines_fast::<_, false, true, false>(&mut reader) + } } // show_chars, show_lines (false, true, true, false, false) => { @@ -510,7 +521,19 @@ fn word_count_from_reader( } // show_bytes, show_chars (true, true, false, false, false) => { - count_bytes_chars_and_lines_fast::<_, true, true, false>(&mut reader) + if *IS_SINGLE_BYTE_LOCALE { + let (bytes, error) = count_bytes_fast(&mut reader); + ( + WordCount { + bytes, + chars: bytes, + ..WordCount::default() + }, + error, + ) + } else { + count_bytes_chars_and_lines_fast::<_, true, true, false>(&mut reader) + } } // show_bytes, show_chars, show_lines (true, true, true, false, false) => { @@ -667,12 +690,19 @@ fn word_count_from_reader_specialized< } Err(e) => { if let Some(e) = handle_error(e, &mut total, &mut in_word) { + if SHOW_CHARS && *IS_SINGLE_BYTE_LOCALE { + total.chars = total.bytes; + } return (total, Some(e)); } } } } + if SHOW_CHARS && *IS_SINGLE_BYTE_LOCALE { + total.chars = total.bytes; + } + (total, None) } @@ -1032,5 +1062,7 @@ fn print_stats( writeln!(stdout) } +static IS_SINGLE_BYTE_LOCALE: LazyLock = LazyLock::new(is_single_byte_locale); + static IS_POSIXLY_CORRECT: LazyLock = LazyLock::new(|| env::var_os("POSIXLY_CORRECT").is_some()); diff --git a/tests/by-util/test_wc.rs b/tests/by-util/test_wc.rs index 6101715b17f..07eb1130184 100644 --- a/tests/by-util/test_wc.rs +++ b/tests/by-util/test_wc.rs @@ -61,6 +61,7 @@ fn test_stdin_explicit() { #[test] fn test_utf8() { new_ucmd!() + .env("LC_ALL", "C.UTF-8") .args(&["-lwmcL"]) .pipe_in_fixture("UTF_8_test.txt") .succeeds() @@ -70,6 +71,7 @@ fn test_utf8() { #[test] fn test_utf8_words() { new_ucmd!() + .env("LC_ALL", "C.UTF-8") .arg("-w") .pipe_in_fixture("UTF_8_weirdchars.txt") .succeeds() @@ -79,6 +81,7 @@ fn test_utf8_words() { #[test] fn test_utf8_line_length_words() { new_ucmd!() + .env("LC_ALL", "C.UTF-8") .arg("-Lw") .pipe_in_fixture("UTF_8_weirdchars.txt") .succeeds() @@ -88,6 +91,7 @@ fn test_utf8_line_length_words() { #[test] fn test_utf8_line_length_chars() { new_ucmd!() + .env("LC_ALL", "C.UTF-8") .arg("-Lm") .pipe_in_fixture("UTF_8_weirdchars.txt") .succeeds() @@ -97,6 +101,7 @@ fn test_utf8_line_length_chars() { #[test] fn test_utf8_line_length_chars_words() { new_ucmd!() + .env("LC_ALL", "C.UTF-8") .arg("-Lmw") .pipe_in_fixture("UTF_8_weirdchars.txt") .succeeds() @@ -106,6 +111,7 @@ fn test_utf8_line_length_chars_words() { #[test] fn test_utf8_chars() { new_ucmd!() + .env("LC_ALL", "C.UTF-8") .arg("-m") .pipe_in_fixture("UTF_8_weirdchars.txt") .succeeds() @@ -115,6 +121,7 @@ fn test_utf8_chars() { #[test] fn test_utf8_bytes_chars() { new_ucmd!() + .env("LC_ALL", "C.UTF-8") .arg("-cm") .pipe_in_fixture("UTF_8_weirdchars.txt") .succeeds() @@ -205,18 +212,65 @@ fn test_utf8_sequences_across_read_buffer_boundaries() { } } +fn locale_charmap(locale: &str) -> Option { + #[cfg(unix)] + { + std::process::Command::new("locale") + .env("LC_ALL", locale) + .arg("charmap") + .output() + .ok() + .filter(|o| o.status.success()) + .map(|o| String::from_utf8_lossy(&o.stdout).trim().to_uppercase()) + } + #[cfg(not(unix))] + { + let _ = locale; + None + } +} + +fn is_single_byte_charmap(charmap: &str) -> bool { + let lower = charmap.to_ascii_lowercase(); + !lower.starts_with("utf") + && !lower.starts_with("gb") + && !lower.starts_with("big5") + && !lower.starts_with("euc") + && !lower.starts_with("ujis") + && !lower.starts_with("sjis") + && !lower.starts_with("shift_jis") + && !lower.starts_with("shift-jis") + && !lower.starts_with("cp932") + && !lower.starts_with("cp949") + && !lower.starts_with("cp950") + && !lower.starts_with("cp936") +} + +fn is_locale_single_byte(locale: &str) -> bool { + if let Some(charmap) = locale_charmap(locale) { + is_single_byte_charmap(&charmap) + } else { + matches!(locale, "C" | "POSIX") + } +} + #[test] fn test_c_locale_multibyte_character_counts() { - // In the C locale, characters should count as bytes. - // Documenting current behavior where multibyte sequences count as 1 character. - let newline_cases: &[(&[u8], usize, usize, usize)] = &[ - (b"\xc3\xa4\n", 1, 1, 2), - (b"\xe2\x82\xac\n", 1, 1, 2), - (b"\xf0\x9f\x92\xa9\n", 1, 1, 2), - (b"hello \xc3\xa4\nworld\n", 2, 3, 14), + let c_is_single_byte = is_locale_single_byte("C"); + + let newline_cases: &[(&[u8], usize, usize, usize, usize)] = &[ + (b"\xc3\xa4\n", 1, 1, 3, 2), + (b"\xe2\x82\xac\n", 1, 1, 4, 2), + (b"\xf0\x9f\x92\xa9\n", 1, 1, 5, 2), + (b"hello \xc3\xa4\nworld\n", 2, 3, 15, 14), ]; - for &(input, lines, words, chars) in newline_cases { + for &(input, lines, words, chars_sb, chars_utf8) in newline_cases { let bytes = input.len(); + let chars = if c_is_single_byte { + chars_sb + } else { + chars_utf8 + }; for (flag, expected) in [ ("-m", format!("{chars}\n")), ("-cm", format!("{chars:7} {bytes:7}\n")), @@ -234,13 +288,18 @@ fn test_c_locale_multibyte_character_counts() { } } - let raw_cases: &[(&[u8], usize, usize)] = &[ - (b"\xc3\xa4", 1, 1), - (b"\xe2\x82\xac", 1, 1), - (b"\xf0\x9f\x92\xa9", 1, 1), + let raw_cases: &[(&[u8], usize, usize, usize)] = &[ + (b"\xc3\xa4", 1, 2, 1), + (b"\xe2\x82\xac", 1, 3, 1), + (b"\xf0\x9f\x92\xa9", 1, 4, 1), ]; - for &(input, words, chars) in raw_cases { + for &(input, words, chars_sb, chars_utf8) in raw_cases { let bytes = input.len(); + let chars = if c_is_single_byte { + chars_sb + } else { + chars_utf8 + }; for (flag, expected) in [ ("-m", format!("{chars}\n")), ("-cm", format!("{chars:7} {bytes:7}\n")), @@ -256,17 +315,149 @@ fn test_c_locale_multibyte_character_counts() { } } + let expected_chars = if c_is_single_byte { 3 } else { 2 }; let (at, mut ucmd) = at_and_ucmd!(); at.write_bytes("input", b"\xc3\xa4\n"); ucmd.env("LC_ALL", "C") .args(&["-cm", "input"]) .succeeds() - .stdout_is("2 3 input\n"); + .stdout_is(format!("{expected_chars} 3 input\n")); + + let (at, mut ucmd) = at_and_ucmd!(); + at.write_bytes("input", b"\xc3\xa4\n"); + ucmd.env("LC_ALL", "C") + .args(&["-m", "input"]) + .succeeds() + .stdout_is(format!("{expected_chars} input\n")); +} + +#[cfg(unix)] +#[test] +#[cfg_attr( + wasi_runner, + ignore = "WASI: native single-byte locales are unavailable" +)] +fn test_non_c_single_byte_locale_counts_every_byte() { + let candidate_locales = [ + "en_US.iso88591", + "fr_FR.iso88591", + "de_DE.iso88591", + "es_ES.iso88591", + "en_GB.iso88591", + "en_US.ISO-8859-1", + "fr_FR.ISO-8859-1", + ]; + let fr_locale = std::env::var("LOCALE_FR").ok(); + let single_byte_locale = fr_locale + .as_deref() + .into_iter() + .chain(candidate_locales) + .find(|&loc| { + locale_charmap(loc) + .as_deref() + .is_some_and(is_single_byte_charmap) + }); + + let Some(locale) = single_byte_locale else { + return; + }; + + let cases: &[(&[u8], usize, usize, usize)] = &[ + (b"\xc3\xa4\n", 1, 1, 3), + (b"\xe2\x82\xac\n", 1, 1, 4), + (b"\xf0\x9f\x92\xa9\n", 1, 1, 5), + (b"hello \xc3\xa4\nworld\n", 2, 3, 15), + ]; + for &(input, lines, words, chars) in cases { + let bytes = input.len(); + for (flag, expected) in [ + ("-m", format!("{chars}\n")), + ("-cm", format!("{chars:7} {bytes:7}\n")), + ("-ml", format!("{lines:7} {chars:7}\n")), + ("-cml", format!("{lines:7} {chars:7} {bytes:7}\n")), + ("-mw", format!("{words:7} {chars:7}\n")), + ("-cmw", format!("{words:7} {chars:7} {bytes:7}\n")), + ] { + new_ucmd!() + .env("LC_ALL", locale) + .arg(flag) + .pipe_in(input) + .succeeds() + .stdout_is(expected); + } + } +} + +#[test] +fn test_c_and_posix_locale_precedence_counts_every_byte() { + for locale in ["C", "POSIX"] { + let is_single_byte = is_locale_single_byte(locale); + let expected = if is_single_byte { + " 4 4\n" + } else { + " 3 4\n" + }; + for variable in ["LC_ALL", "LC_CTYPE", "LANG"] { + new_ucmd!() + .env("LC_ALL", "") + .env("LC_CTYPE", "") + .env("LANG", "C.UTF-8") + .env(variable, locale) + .arg("-cm") + .pipe_in(b"\xc3\xa4\xff\n") + .succeeds() + .stdout_is(expected); + } + } +} + +#[cfg(unix)] +#[test] +#[cfg_attr(wasi_runner, ignore = "WASI: native multibyte locales are unavailable")] +fn test_non_utf8_multibyte_locales_preserve_character_counts() { + for (locale, charmap, input) in [ + ("ja_JP.eucjp", "EUC-JP", b"\xe0\xa1"), + ("zh_TW.big5", "BIG5", b"\xa4\x40"), + ] { + let Ok(output) = std::process::Command::new("locale") + .env("LC_ALL", locale) + .arg("charmap") + .output() + else { + continue; + }; + if !output.status.success() || String::from_utf8_lossy(&output.stdout).trim() != charmap { + continue; + } + for tunables in ["", "glibc.cpu.hwcaps=-AVX2,-SSE2,-ASIMD"] { + for (flag, expected) in [ + ("-m", "1\n"), + ("-cm", " 1 2\n"), + ("-ml", " 0 1\n"), + ("-cml", " 0 1 2\n"), + ] { + new_ucmd!() + .env("LC_ALL", locale) + .env("GLIBC_TUNABLES", tunables) + .arg(flag) + .pipe_in(input) + .succeeds() + .stdout_is(expected); + } + } + let (at, mut ucmd) = at_and_ucmd!(); + at.write_bytes("input", input); + ucmd.env("LC_ALL", locale) + .args(&["-cm", "input"]) + .succeeds() + .stdout_is("1 2 input\n"); + } } #[test] fn test_utf8_bytes_lines() { new_ucmd!() + .env("LC_ALL", "C.UTF-8") .arg("-cl") .pipe_in_fixture("UTF_8_weirdchars.txt") .succeeds() @@ -276,6 +467,7 @@ fn test_utf8_bytes_lines() { #[test] fn test_utf8_bytes_chars_lines() { new_ucmd!() + .env("LC_ALL", "C.UTF-8") .arg("-cml") .pipe_in_fixture("UTF_8_weirdchars.txt") .succeeds() @@ -285,6 +477,7 @@ fn test_utf8_bytes_chars_lines() { #[test] fn test_utf8_chars_words() { new_ucmd!() + .env("LC_ALL", "C.UTF-8") .arg("-mw") .pipe_in_fixture("UTF_8_weirdchars.txt") .succeeds() @@ -294,6 +487,7 @@ fn test_utf8_chars_words() { #[test] fn test_utf8_line_length_lines() { new_ucmd!() + .env("LC_ALL", "C.UTF-8") .arg("-Ll") .pipe_in_fixture("UTF_8_weirdchars.txt") .succeeds() @@ -303,6 +497,7 @@ fn test_utf8_line_length_lines() { #[test] fn test_utf8_line_length_lines_words() { new_ucmd!() + .env("LC_ALL", "C.UTF-8") .arg("-Llw") .pipe_in_fixture("UTF_8_weirdchars.txt") .succeeds() @@ -312,6 +507,7 @@ fn test_utf8_line_length_lines_words() { #[test] fn test_utf8_lines_chars() { new_ucmd!() + .env("LC_ALL", "C.UTF-8") .arg("-ml") .pipe_in_fixture("UTF_8_weirdchars.txt") .succeeds() @@ -321,6 +517,7 @@ fn test_utf8_lines_chars() { #[test] fn test_utf8_lines_words_chars() { new_ucmd!() + .env("LC_ALL", "C.UTF-8") .arg("-mlw") .pipe_in_fixture("UTF_8_weirdchars.txt") .succeeds() @@ -330,6 +527,7 @@ fn test_utf8_lines_words_chars() { #[test] fn test_utf8_line_length_lines_chars() { new_ucmd!() + .env("LC_ALL", "C.UTF-8") .arg("-Llm") .pipe_in_fixture("UTF_8_weirdchars.txt") .succeeds() @@ -339,6 +537,7 @@ fn test_utf8_line_length_lines_chars() { #[test] fn test_utf8_all() { new_ucmd!() + .env("LC_ALL", "C.UTF-8") .arg("-lwmcL") .pipe_in_fixture("UTF_8_weirdchars.txt") .succeeds() From d8aaad6083414cee39158702e2cee4107317f9d4 Mon Sep 17 00:00:00 2001 From: darkraider01 Date: Wed, 7 Oct 2026 16:52:47 +0530 Subject: [PATCH 07/10] wc: correct single-byte locale detection across platforms --- src/uu/wc/src/count_fast.rs | 95 ++++++++++++++++++++++++++++--------- 1 file changed, 72 insertions(+), 23 deletions(-) diff --git a/src/uu/wc/src/count_fast.rs b/src/uu/wc/src/count_fast.rs index 4486a0c7c8a..fd6c63f390c 100644 --- a/src/uu/wc/src/count_fast.rs +++ b/src/uu/wc/src/count_fast.rs @@ -105,26 +105,25 @@ fn detect_native_locale_encoding() -> Option { unsafe extern "C" { fn __ctype_get_mb_cur_max() -> libc::size_t; } - Some(__ctype_get_mb_cur_max() as usize) + Some(__ctype_get_mb_cur_max()) } - #[cfg(target_vendor = "apple")] + #[cfg(any( + target_vendor = "apple", + target_os = "freebsd", + target_os = "dragonfly" + ))] { unsafe extern "C" { - fn ___mb_cur_max() -> libc::size_t; + fn ___mb_cur_max() -> libc::c_int; } - Some(___mb_cur_max() as usize) + usize::try_from(___mb_cur_max()).ok() } - #[cfg(any( - target_os = "freebsd", - target_os = "dragonfly", - target_os = "openbsd", - target_os = "netbsd" - ))] + #[cfg(target_os = "openbsd")] { unsafe extern "C" { - static __mb_cur_max: libc::c_int; + fn __mb_cur_max() -> libc::size_t; } - Some(__mb_cur_max as usize) + Some(__mb_cur_max()) } #[cfg(not(any( target_os = "linux", @@ -133,7 +132,6 @@ fn detect_native_locale_encoding() -> Option { target_os = "freebsd", target_os = "dragonfly", target_os = "openbsd", - target_os = "netbsd" )))] { None @@ -177,20 +175,25 @@ fn locale_encoding_info() -> LocaleEncodingInfo { return info; } - match get_ctype_encoding() { - UEncoding::Utf8 => LocaleEncodingInfo { - is_utf8: true, - is_single_byte: false, - }, - UEncoding::Ascii => LocaleEncodingInfo { - is_utf8: false, - is_single_byte: true, - }, - } + let name = ["LC_ALL", "LC_CTYPE", "LANG"] + .iter() + .find_map(|key| std::env::var(key).ok().filter(|value| !value.is_empty())); + fallback_locale_encoding(get_ctype_encoding(), name.as_deref()) }); *INFO } +fn fallback_locale_encoding(encoding: UEncoding, name: Option<&str>) -> LocaleEncodingInfo { + let is_utf8 = encoding == UEncoding::Utf8; + // Other non-UTF-8 encodings may still use more than one byte per character. + let is_single_byte = + !is_utf8 && name.map_or(!cfg!(windows), |name| matches!(name, "C" | "POSIX")); + LocaleEncodingInfo { + is_utf8, + is_single_byte, + } +} + pub(crate) fn is_utf8_locale() -> bool { locale_encoding_info().is_utf8 } @@ -460,3 +463,49 @@ pub(crate) fn count_bytes_chars_and_lines_fast< } } } + +#[cfg(test)] +mod tests { + use super::{LocaleEncodingInfo, UEncoding, fallback_locale_encoding}; + + #[test] + fn fallback_keeps_other_encodings_multibyte() { + for name in ["ja_JP.eucjp", "zh_TW.big5", "en_US.iso88591", "unknown"] { + assert_eq!( + fallback_locale_encoding(UEncoding::Ascii, Some(name)), + LocaleEncodingInfo { + is_utf8: false, + is_single_byte: false + }, + "{name}" + ); + } + } + + #[test] + fn fallback_counts_bytes_only_for_c_and_posix() { + for name in ["C", "POSIX"] { + assert_eq!( + fallback_locale_encoding(UEncoding::Ascii, Some(name)), + LocaleEncodingInfo { + is_utf8: false, + is_single_byte: true + } + ); + } + assert_eq!( + fallback_locale_encoding(UEncoding::Utf8, Some("C.UTF-8")), + LocaleEncodingInfo { + is_utf8: true, + is_single_byte: false + } + ); + assert_eq!( + fallback_locale_encoding(UEncoding::Ascii, None), + LocaleEncodingInfo { + is_utf8: false, + is_single_byte: !cfg!(windows) + } + ); + } +} From 57fb77b97b38355dee27f3cbebc5d5a1098d3889 Mon Sep 17 00:00:00 2001 From: darkraider01 Date: Wed, 7 Oct 2026 16:52:51 +0530 Subject: [PATCH 08/10] tests/wc: skip unsupported single-byte locales on musl --- tests/by-util/test_wc.rs | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/tests/by-util/test_wc.rs b/tests/by-util/test_wc.rs index 07eb1130184..823ba94f91e 100644 --- a/tests/by-util/test_wc.rs +++ b/tests/by-util/test_wc.rs @@ -331,7 +331,8 @@ fn test_c_locale_multibyte_character_counts() { .stdout_is(format!("{expected_chars} input\n")); } -#[cfg(unix)] +// musl uses UTF-8 for non-C locales, even if the host locale command reports otherwise. +#[cfg(all(unix, not(target_env = "musl")))] #[test] #[cfg_attr( wasi_runner, From 5e532740430a6692582b38617b7a17ddbcb2ee7c Mon Sep 17 00:00:00 2001 From: darkraider01 Date: Fri, 9 Oct 2026 02:31:46 +0530 Subject: [PATCH 09/10] tests/wc: reject locale lookup warnings --- tests/by-util/test_wc.rs | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/tests/by-util/test_wc.rs b/tests/by-util/test_wc.rs index 823ba94f91e..d35edc854cb 100644 --- a/tests/by-util/test_wc.rs +++ b/tests/by-util/test_wc.rs @@ -220,7 +220,7 @@ fn locale_charmap(locale: &str) -> Option { .arg("charmap") .output() .ok() - .filter(|o| o.status.success()) + .filter(|o| o.status.success() && o.stderr.is_empty()) .map(|o| String::from_utf8_lossy(&o.stdout).trim().to_uppercase()) } #[cfg(not(unix))] @@ -254,6 +254,12 @@ fn is_locale_single_byte(locale: &str) -> bool { } } +#[cfg(unix)] +#[test] +fn test_unavailable_locale_is_not_reported_as_single_byte() { + assert_eq!(locale_charmap("uutests_missing_LOCALE"), None); +} + #[test] fn test_c_locale_multibyte_character_counts() { let c_is_single_byte = is_locale_single_byte("C"); From aee391ff08b0a2c5f8f0d52b960f870349de6c55 Mon Sep 17 00:00:00 2001 From: darkraider01 Date: Fri, 9 Oct 2026 02:31:48 +0530 Subject: [PATCH 10/10] wc: leave the UTF-8 benchmark in its separate PR --- src/uu/wc/BENCHMARKING.md | 6 ------ src/uu/wc/benches/wc_bench.rs | 11 ----------- 2 files changed, 17 deletions(-) diff --git a/src/uu/wc/BENCHMARKING.md b/src/uu/wc/BENCHMARKING.md index 357a8f46005..7a127471f12 100644 --- a/src/uu/wc/BENCHMARKING.md +++ b/src/uu/wc/BENCHMARKING.md @@ -82,12 +82,6 @@ candidate as it's fairly large. Use [`hyperfine`](https://github.com/sharkdp/hyperfine) to compare the performance. For example, `hyperfine 'wc somefile' 'uuwc somefile'`. -Run the UTF-8 character-count benchmark with an explicit UTF-8 locale: - -```shell -LC_ALL=C.UTF-8 cargo bench -p uu_wc --bench wc_bench -- wc_chars_utf8 -``` - If you want to get fancy and exhaustive, generate a table: | | moby64.txt | odyssey256.txt | 25Mshortlines | /usr/bin/docker | diff --git a/src/uu/wc/benches/wc_bench.rs b/src/uu/wc/benches/wc_bench.rs index 226490de35c..6abcb5efd4a 100644 --- a/src/uu/wc/benches/wc_bench.rs +++ b/src/uu/wc/benches/wc_bench.rs @@ -78,17 +78,6 @@ fn wc_chars_large_line_count(bencher: Bencher, num_lines: usize) { .bench_values(|args| black_box(uumain(args))); } -#[divan::bench(args = [100_000])] -fn wc_chars_utf8(bencher: Bencher, num_lines: usize) { - let temp_dir = tempfile::tempdir().unwrap(); - let data = "hello ä € 💩\n".repeat(num_lines); - let file_path = create_test_file(data.as_bytes(), temp_dir.path()); - - bencher - .with_inputs(|| get_bench_args(&[&"-cm", &file_path]).into_iter()) - .bench_values(|args| black_box(uumain(args))); -} - /// Benchmark word counting on large line counts #[divan::bench(args = [100_000])] fn wc_words_large_line_count(bencher: Bencher, num_lines: usize) {