From 6594fc10cebb20c7c4193ab4af3a16850dbe6d05 Mon Sep 17 00:00:00 2001 From: metif12 Date: Fri, 2 Oct 2026 09:42:37 +0330 Subject: [PATCH 1/3] build: make all 76 utilities compile against current V main does not build with V 0.5.2, and `make testfmt` is red too, so every CI job has been failing. This gets both gates back to green without changing any intended behaviour. Compile fixes, one per cause: - `v.mathutil` is gone; `min`/`max` now live in `math`. Affects basenc, cut and tail. - `for ; cond; post {}` and `for i := n; i; i-- {}` are no longer accepted: the condition has to be `bool`. Affects cksum and expand. - `time.sleep` takes a `time.Duration`, not a number of seconds. sleep now converts explicitly and saturates instead of overflowing, so `sleep inf` keeps waiting rather than wrapping around to roughly zero. - `const` cannot hold the result of a function call, so printenv's NUL terminator became a function. - `setup_cp_command`/`setup_mv_command`/`setup_command` (head) were declared with `?`, but every failure path in them calls the @[noreturn] `common.exit_with_error_message`, so they never return an error and their callers had no `err` to look at. They now return plain tuples. - `C.statvfs` is already declared privately by vlib/builtin/cfns.c.v, so a local declaration resolved to that one and failed to compile. stat binds a distinct name to the libc call with #define. - `struct utmpx` was declared twice, in common/readutmp_nix.c.v and again in src/users/users.c.v with a different, shorter field list, which V now rejects as a redeclaration. Declared once, with the full layout. stat additionally needed more than a compile fix: - src/stat/stat_to_vlib.v (now src/stat/modes.v) imported v.scanner and v.pref to decode --printf escapes. Building it pulled the whole V compiler into the program: `v build` of src/stat never finished, and taking the machine down with it. The escapes are now decoded directly. - `` is a glibc internal header, so it is absent on musl based distributions (issue #145). The kernel uapi `` is used instead. - `statx()` wrote the kernel's 224 byte struct through a `voidptr` into a V struct that was 160 bytes, so every call corrupted memory past the end of the allocation; stat segfaulted on every invocation. The C struct is now declared and used for real. `struct statvfs` has the same problem and its V mirror now matches the ABI including the trailing spare. - `$embed_file` no longer resolves inside a function body, so fstypes.txt is embedded at module scope. wc needed a fix of its own: - `read_chunk` returned `?FileChunk`, but its caller inspected `err`, which does not exist for an option. os.File.read also reports end of file as an `os.Eof` error, and reading a zero-length chunk indexed `buffer[-1]`, so `wc` on an empty file would have panicked. It now returns `!FileChunk` and distinguishes exhaustion from a real read error. - `\r` was matched both as "ignore" and as whitespace, which V rejects as a duplicate case. - `byte(bool)` is no longer a valid conversion; the single-count check counts the selected options instead. Finally, `v fmt -w .` over the 16 files that `v fmt -verify` rejected. With this, `make` builds all 76 utilities and `make testfmt` passes. --- common/readutmp_nix.c.v | 42 +++++----- common/testing/testrig.v | 3 +- src/base64/base64.v | 2 +- src/basenc/basenc.v | 4 +- src/cksum/cksum.v | 3 +- src/cp/helper.v | 6 +- src/cut/cut.v | 16 ++-- src/expand/expand.v | 2 +- src/expr/expr_test.v | 2 +- src/fmt/fmt.v | 6 +- src/head/head.v | 4 +- src/mv/helper.v | 6 +- src/nproc/nproc.v | 2 +- src/numfmt/options.v | 14 ++-- src/printenv/printenv.v | 8 +- src/printf/printf.v | 8 +- src/shred/config.v | 5 +- src/shuf/shuf.v | 4 +- src/sleep/sleep.v | 23 +++++- src/sort/options.v | 26 +++---- src/stat/{stat_to_vlib.v => modes.v} | 111 +++++++++++++++++++++++---- src/stat/stat.c.v | 92 +++++++++++++++++++--- src/stat/stat.v | 34 ++++---- src/tail/parse_args.v | 14 +++- src/tail/tail.v | 10 +-- src/truncate/truncate.v | 18 ++++- src/unlink/unlink.c.v | 1 + src/users/users.c.v | 18 ++--- src/users/users.v | 6 +- src/wc/wc.v | 24 +++--- 30 files changed, 358 insertions(+), 156 deletions(-) rename src/stat/{stat_to_vlib.v => modes.v} (51%) diff --git a/common/readutmp_nix.c.v b/common/readutmp_nix.c.v index 61a2ccd4..0b5b59f8 100644 --- a/common/readutmp_nix.c.v +++ b/common/readutmp_nix.c.v @@ -4,30 +4,30 @@ module common #include #include -// TODO: conflicts with `timeval` in -// -/* -pub struct C.timeval { - tv_sec i64 // Seconds. - tv_usec i64 // Microseconds. -}*/ - -// -pub struct C.timeval { -pub: - tv_sec u64 // Seconds. - tv_usec u64 // Microseconds. +// `struct utmpx` was previously declared twice, once here and once in +// src/users/users.c.v with a different (shorter) field list, which V rejects as +// a redeclaration. Declare it once, here, with the glibc layout. +// +// The fixed size fields are UT_LINESIZE/UT_NAMESIZE/UT_HOSTSIZE (32/32/256) and +// the element size has to match what getutxent() writes, because read_utmp() +// copies entries into a []C.utmpx. +pub struct C.exit_status { + et_code u16 + et_term u16 } pub struct C.utmpx { - ut_type i16 // Type of login. - ut_pid int // Process ID of login process. - ut_line [32]char // Devicename. - ut_id [4]char // Inittab ID. - ut_user [32]char // Username. - ut_host [256]char // Hostname for remote login. - ut_tv C.timeval // TODO: Declare sub struct correctly - ut_addr_v6 [4]int // Internet address of remote host. + ut_type i16 // Type of login. + ut_pid int // Process ID of login process. + ut_line [32]u8 // Devicename. + ut_id [4]u8 // Inittab ID. + ut_user [32]u8 // Username. + ut_host [256]u8 // Hostname for remote login. + ut_exit C.exit_status // Exit status of the process. + ut_session_id i32 // Session ID. + ut_tv C.timeval // Time entry. + ut_addr_v6 [4]i32 // Internet address of remote host. + __unused [20]u8 // Reserved. } // sets the name of the utmp-format file for the other utmp functions to access. diff --git a/common/testing/testrig.v b/common/testing/testrig.v index c87cceda..5206fa5c 100644 --- a/common/testing/testrig.v +++ b/common/testing/testrig.v @@ -84,8 +84,7 @@ pub fn (rig TestRig) assert_platform_util() { } else { // uptime was moved to procps-ng and may not be available in coreutils - assert - ver == 'uptime (GNU coreut' || ver == 'uptime (coreutils)' || ver == 'uptime from procps' + assert ver == 'uptime (GNU coreut' || ver == 'uptime (coreutils)' || ver == 'uptime from procps' } } } diff --git a/src/base64/base64.v b/src/base64/base64.v index 456c9a87..cd1cda29 100644 --- a/src/base64/base64.v +++ b/src/base64/base64.v @@ -159,7 +159,7 @@ fn main() { exit(1) } if wraping_opt < 0 { - eprintln('${application_name}: invalid wrap size: \'${wraping_opt}\'') + eprintln("${application_name}: invalid wrap size: '${wraping_opt}'") exit(1) } diff --git a/src/basenc/basenc.v b/src/basenc/basenc.v index b9175f70..f04a2e99 100644 --- a/src/basenc/basenc.v +++ b/src/basenc/basenc.v @@ -1,5 +1,5 @@ import os -import v.mathutil +import math import encoding.base64 import encoding.base32 @@ -51,7 +51,7 @@ fn print_encoded(encoded string, options Options) { return } for start := 0; start < encoded.len; start += options.wrap { - end := mathutil.min(start + options.wrap, encoded.len) + end := math.min(start + options.wrap, encoded.len) // safe to use string slicing because all chars // are in printable ascii println(encoded[start..end]) diff --git a/src/cksum/cksum.v b/src/cksum/cksum.v index 146aef5f..f4e4bad4 100644 --- a/src/cksum/cksum.v +++ b/src/cksum/cksum.v @@ -70,8 +70,9 @@ fn calc_sums(args Args) { } mut len_counter := total_length - for ; len_counter; len_counter >>= 8 { + for len_counter > 0 { crc = (crc << 8) ^ crctab[0][((crc >> 24) ^ len_counter) & 0xFF] + len_counter >>= 8 } crc = ~crc & 0xffff_ffff diff --git a/src/cp/helper.v b/src/cp/helper.v index 9c36db01..b259355d 100644 --- a/src/cp/helper.v +++ b/src/cp/helper.v @@ -69,7 +69,7 @@ fn success_exit(messages ...string) { exit(0) } -fn setup_cp_command(args []string) ?(CpCommand, []string, string) { +fn setup_cp_command(args []string) (CpCommand, []string, string) { mut fp := common.flag_parser(args) fp.application('cp') fp.limit_free_args_to_at_least(1) or { common.exit_with_error_message(name, err.msg()) } @@ -143,9 +143,7 @@ fn setup_cp_command(args []string) ?(CpCommand, []string, string) { } pub fn run_cp(args []string) { - cp, sources, dest := setup_cp_command(args) or { - common.exit_with_error_message(name, err.msg()) - } + cp, sources, dest := setup_cp_command(args) if sources.len > 1 && !os.is_dir(dest) { common.exit_with_error_message(name, target_not_dir(dest)) } diff --git a/src/cut/cut.v b/src/cut/cut.v index c3764386..03084132 100644 --- a/src/cut/cut.v +++ b/src/cut/cut.v @@ -3,8 +3,8 @@ import arrays import common import flag import io +import math import os -import v.mathutil const app_name = 'cut' const space_comma = ' ,' @@ -124,13 +124,13 @@ fn combine_ranges_and_zero_index(ranges []Range, max int, complement bool) []Ran outer: for range in ranges.sorted(a.start < b.start) { start := range.start - 1 - end := if range.end == -1 { max } else { mathutil.min(max, range.end) } + end := if range.end == -1 { max } else { math.min(max, range.end) } for mut combined_range in combined_ranges { if range_overlaps_range(start, end, combined_range.start, combined_range.end) { combined_range = Range{ - start: mathutil.min(start, combined_range.start) - end: mathutil.max(end, combined_range.end) + start: math.min(start, combined_range.start) + end: math.max(end, combined_range.end) } continue outer } @@ -189,7 +189,7 @@ fn get_args(args []string) Args { delimiter := fp.string('delimter', `d`, '', 'use instead of TAB for field delimter') fields := fp.string('fields', `f`, '', 'select only fields; also print any line${wrap}' + - 'that contains no delimiter character, unless the${wrap}-s option is specified') + 'that contains no delimiter character, unless the${wrap}-s option is specified') fp.bool('', `n`, false, '(ignored)') complement := fp.bool('complement', ` `, false, 'complement the set of selected bytes, characters${wrap}or fields') @@ -207,9 +207,9 @@ fn get_args(args []string) Args { 'range, or many ranges separated by commas. Selected input is written${eol}' + 'in the same order that it is read, and is written exactly once.${eol}${eol}' + 'Each range is one of:${eol}${eol}' + - ' N N\'th byte, character or field, counted from 1${eol}' + - ' N- from N\'th byte, character or field, to end of line${eol}' + - ' N-M from N\'th to M\'th (included) byte, character or field${eol}' + " -M from first to M'th (included) byte, character or field") + " N N'th byte, character or field, counted from 1${eol}" + + " N- from N'th byte, character or field, to end of line${eol}" + + " N-M from N'th to M'th (included) byte, character or field${eol}" + " -M from first to M'th (included) byte, character or field") fp.footer(common.coreutils_footer()) file_args := fp.finalize() or { exit_error(err.msg()) } diff --git a/src/expand/expand.v b/src/expand/expand.v index 0e54e215..8771e1c0 100644 --- a/src/expand/expand.v +++ b/src/expand/expand.v @@ -8,7 +8,7 @@ const nl = '\n' fn process_line(line string, initial bool, tabs int) { mut sp := '' - for i := tabs; i; i-- { + for _ in 0 .. tabs { sp += ' ' } diff --git a/src/expr/expr_test.v b/src/expr/expr_test.v index 5dd9c71b..7d15be0e 100644 --- a/src/expr/expr_test.v +++ b/src/expr/expr_test.v @@ -161,7 +161,7 @@ const mb_tests = [ // fail: multi; broken UTF-8 is treated in different way // "substr \u03B1bc\xB4ef 3 3", 'match abcdef ab', - 'match abcdef "\\(ab\\)"' + 'match abcdef "\\(ab\\)"', // fail: single; this is a good news, regex in vlib supports UTF-8 // "match \u03B1bc\u03B4ef .bc", // fail: both; broken regex of vlib, see #10885 diff --git a/src/fmt/fmt.v b/src/fmt/fmt.v index 1226fdb9..441ff77c 100644 --- a/src/fmt/fmt.v +++ b/src/fmt/fmt.v @@ -241,11 +241,11 @@ fn process_args(args []string) App { crown_marg := fp.bool('crown-margin', `c`, false, 'preserve the indentation of the first two lines within' + - '${pad}a paragraph, and align the left margin of each' + - '${pad}subsequent line with that of the second line') + '${pad}a paragraph, and align the left margin of each' + + '${pad}subsequent line with that of the second line') prefix_str := fp.string('prefix', `p`, '', 'reformat only lines beginning with STRING, reattaching ' + - '${pad}the prefix to reformatted lines') + '${pad}the prefix to reformatted lines') split_only := fp.bool('split-only', `s`, false, 'split long lines, but do not refill') tagged_par := fp.bool('tagged-paragraph', `t`, false, 'indentation of first line different from second') diff --git a/src/head/head.v b/src/head/head.v index 5c02bca4..73fe15ee 100644 --- a/src/head/head.v +++ b/src/head/head.v @@ -294,7 +294,7 @@ fn wrap_long_command_description(description string, max_cols int) string { return buf.str() } -fn setup_command(args []string) ?(HeadCommand, []InputFile) { +fn setup_command(args []string) (HeadCommand, []InputFile) { mut fp := common.flag_parser(args) fp.application(name) fp.usage_example('[OPTION]... [FILE]...') @@ -330,7 +330,7 @@ fn setup_command(args []string) ?(HeadCommand, []InputFile) { } fn run_head(args []string) { - head, mut files := setup_command(args) or { common.exit_with_error_message(name, err.msg()) } + head, mut files := setup_command(args) head.run(mut files) } diff --git a/src/mv/helper.v b/src/mv/helper.v index 4bccee39..61513532 100644 --- a/src/mv/helper.v +++ b/src/mv/helper.v @@ -48,9 +48,7 @@ fn int_yes(prompt string) bool { } pub fn run_mv(args []string) { - mv, sources, dest := setup_mv_command(args) or { - common.exit_with_error_message(name, err.msg()) - } + mv, sources, dest := setup_mv_command(args) if sources.len > 1 && !os.is_dir(dest) { common.exit_with_error_message(name, target_not_dir(dest)) } @@ -59,7 +57,7 @@ pub fn run_mv(args []string) { } } -fn setup_mv_command(args []string) ?(MvCommand, []string, string) { +fn setup_mv_command(args []string) (MvCommand, []string, string) { mut fp := common.flag_parser(args) fp.application('mv') fp.limit_free_args_to_at_least(1) or { common.exit_with_error_message(name, err.msg()) } diff --git a/src/nproc/nproc.v b/src/nproc/nproc.v index 533c78b8..4638dd74 100644 --- a/src/nproc/nproc.v +++ b/src/nproc/nproc.v @@ -25,7 +25,7 @@ fn main() { // Check if ignored CPUs is > 0 if ignored_cpus < 0 { - println('nproc: invalid number: \'${ignored_cpus}\'') + println("nproc: invalid number: '${ignored_cpus}'") exit(1) } diff --git a/src/numfmt/options.v b/src/numfmt/options.v index 4ff0d7bd..c6268fd6 100644 --- a/src/numfmt/options.v +++ b/src/numfmt/options.v @@ -45,21 +45,21 @@ fn get_options() Options { 'use locale-defined grouping of digits, e.g. 1,000,000') header := fp.int('header', 0, 1, 'print (without converting) the first N header lines; \n${flag.space}' + - 'defaults to 1 if not specified') + 'defaults to 1 if not specified') invalid := fp.string('invalid', 0, 'abort', 'failure mode for invalid numbers: MODE can be:\n${flag.space}' + - 'abort (default), fail, warn, ignore') + 'abort (default), fail, warn, ignore') padding := fp.int('padding', 0, 0, 'pad the output to N characters; positive N will\n${flag.space}' + - 'right-align; negative N will left-align; padding is\n${flag.space}' + - 'ignored if the output is wider than N; the default is to\n${flag.space}' + - 'automatically pad if a whitespace is found') + 'right-align; negative N will left-align; padding is\n${flag.space}' + + 'ignored if the output is wider than N; the default is to\n${flag.space}' + + 'automatically pad if a whitespace is found') round := fp.string('round', 0, 'from-zero', 'use METHOD for rounding when scaling; METHOD can be:\n${flag.space}' + - 'up, down, from-zero (default), towards-zero, nearest') + 'up, down, from-zero (default), towards-zero, nearest') suffix := fp.string('suffix', 0, '', 'add SUFFIX to output numbers, and accept optional SUFFIX\n${flag.space}' + - 'in input numbers') + 'in input numbers') to := fp.string('to', 0, 'none', 'auto-scale output numbers to UNIT\n${flag.space}' + 'none, si, iec, iec-i') to_unit := fp.int('to-unit', 0, 1, 'the output unit size (instead of the default 1)') diff --git a/src/printenv/printenv.v b/src/printenv/printenv.v index 05f180ce..44367c52 100644 --- a/src/printenv/printenv.v +++ b/src/printenv/printenv.v @@ -1,7 +1,9 @@ import os import common -const zero_byte = byte(0).ascii_str() +fn zero_byte() string { + return u8(0).ascii_str() +} // Exit status: // 0 if all variables specified were found @@ -34,7 +36,7 @@ fn main() { mut s := '${k}=${v}' if opt_nul_terminate { print(s) - print(zero_byte) + print(zero_byte()) } else { println(s) } @@ -49,7 +51,7 @@ fn main() { } if opt_nul_terminate { print(v) - print(zero_byte) + print(zero_byte()) } else { println(v) } diff --git a/src/printf/printf.v b/src/printf/printf.v index 58c486fc..c76ffab7 100644 --- a/src/printf/printf.v +++ b/src/printf/printf.v @@ -242,7 +242,10 @@ fn apply_posix_escape(s string) string { } } } - return if has_unprintable { '\$\'${upout}\'' } else { s.replace_each([ + return if has_unprintable { + "\$'${upout}'" + } else { + s.replace_each([ '|', '\\|', '&', @@ -269,7 +272,8 @@ fn apply_posix_escape(s string) string { '\\"', ' ', '\\ ', - ]) } + ]) + } } // A code below is modded version of vlib/strconv/vprintf.v whose parameter to be an array of string diff --git a/src/shred/config.v b/src/shred/config.v index b1a8ef4c..c9776eb0 100644 --- a/src/shred/config.v +++ b/src/shred/config.v @@ -6,7 +6,7 @@ import os @[version: '0.1'] struct Config { force bool @[short: f; xdoc: 'change permissions to allow writing if necessary'] - iterations int = 3 @[short: n; xdoc: 'overwrite N times instead of the default (3)'] + iterations int = 3 @[short: n; xdoc: 'overwrite N times instead of the default (3)'] random_source string @[xdoc: 'get random bytes from '] size string @[short: s; xdoc: 'shred this many bytes (suffixes like K, M, G accepted)\n'] rm bool @[only: u; xdoc: 'deallocate and remove file after overwriting'] @@ -26,8 +26,7 @@ fn get_args() (Config, []string) { description: 'Usage: shred [OPTION]... FILE...\n' + 'Overwrite the specified FILE(s) repeatedly, in order to make it harder\n' + 'for even very expensive hardware probing to recover the data.' - footer: - '\nDelete FILE(s) if --remove (-u) is specified. The default is not to remove\n' + 'the files because it is common to operate on device files like /dev/hda,\n' + 'and those files usually should not be removed.\n\n' + 'The --remove parameter indicates how to remove a directory entry:\n' + " 'unlink' => use a standard unlink call.\n" + " 'wipe' => also first obfuscate bytes in the name.\n" + " 'wipesync' => also sync each obfuscated byte to the device.\n" + "The default mode is 'wipesync', but note it can be expensive.\n\n" + 'CAUTION: shred assumes the file system and hardware overwrite data in place.\n' + 'Although this is common, many platforms operate otherwise. Also, backups\n' + 'and mirrors may contain unremovable copies that will let a shredded file\n' + 'be recovered later.\n' + common.coreutils_footer() + footer: '\nDelete FILE(s) if --remove (-u) is specified. The default is not to remove\n' + 'the files because it is common to operate on device files like /dev/hda,\n' + 'and those files usually should not be removed.\n\n' + 'The --remove parameter indicates how to remove a directory entry:\n' + " 'unlink' => use a standard unlink call.\n" + " 'wipe' => also first obfuscate bytes in the name.\n" + " 'wipesync' => also sync each obfuscated byte to the device.\n" + "The default mode is 'wipesync', but note it can be expensive.\n\n" + 'CAUTION: shred assumes the file system and hardware overwrite data in place.\n' + 'Although this is common, many platforms operate otherwise. Also, backups\n' + 'and mirrors may contain unremovable copies that will let a shredded file\n' + 'be recovered later.\n' + common.coreutils_footer() ) or { panic(err) } println(doc) exit(0) diff --git a/src/shuf/shuf.v b/src/shuf/shuf.v index 51ec4137..5fce1649 100644 --- a/src/shuf/shuf.v +++ b/src/shuf/shuf.v @@ -88,7 +88,7 @@ fn shuffle_lines(lines []string, settings Settings) []string { exit(1) } mut bytes := io.read_all(io.ReadAllConfig{ reader: file, read_to_end_of_stream: true }) or { - eprintln('${app_name}: ${settings.random_source}: Can\'t read file') + eprintln("${app_name}: ${settings.random_source}: Can't read file") exit(1) } mut seed := u32(0) @@ -159,7 +159,7 @@ fn register_lines_by_file(lines []string, fname string, zero_terminated bool) [] if zero_terminated { mut bytes := io.read_all(io.ReadAllConfig{ reader: file, read_to_end_of_stream: true }) or { - eprintln('${app_name}: ${fname}: Can\'t read file') + eprintln("${app_name}: ${fname}: Can't read file") exit(1) } diff --git a/src/sleep/sleep.v b/src/sleep/sleep.v index c65d9cb0..bd6ef7a9 100644 --- a/src/sleep/sleep.v +++ b/src/sleep/sleep.v @@ -6,6 +6,11 @@ import common const cmd_ns = 'sleep' +// max_duration is the largest sleep time.Duration we are willing to hand to +// time.sleep(), and max_sleep_seconds is that same value expressed in seconds. +const max_duration = time.Duration(max_i64) +const max_sleep_seconds = 9223372036.854776 + // // str="-1.8e+308", ret = -inf, endptr => NULL // str="1.8e+308", ret = +inf, endptr => NULL @@ -51,6 +56,22 @@ fn invalid_time_interval_argument(s string) string { return "${cmd_ns}: invalid time interval '${s}'" } +// seconds_to_duration converts a possibly infinite number of seconds +// into a time.Duration. A Duration is an integer nanosecond count, so +// saturate instead of overflowing: an unbounded sleep must keep waiting +// rather than wrap around to (near) zero. +fn seconds_to_duration(seconds f64) time.Duration { + nanoseconds := seconds * 1e9 + // `!(x < y)` also catches NaN and +inf, both of which must saturate + if !(nanoseconds < max_sleep_seconds) { + return max_duration + } + if nanoseconds <= 0.0 { + return 0 + } + return time.Duration(nanoseconds) +} + fn main() { mut fp := common.flag_parser(os.args) fp.application(cmd_ns) @@ -103,7 +124,7 @@ fn main() { // if seconds = +inf, it would not sleep // but original `sleep` would sleep t := time.ticks() - time.sleep(seconds * time.second) + time.sleep(seconds_to_duration(seconds)) $if trace_sleep_ticks ? { println(time.ticks() - t) } diff --git a/src/sort/options.v b/src/sort/options.v index 8819b2ec..52317bf9 100644 --- a/src/sort/options.v +++ b/src/sort/options.v @@ -40,11 +40,11 @@ fn get_options() Options { 'consider only printable characters') numeric := fp.bool('numeric-sort', `n`, false, 'Restrict the sort key to an initial numeric\n${flag.space}' + - 'string, consisting of optional characters,\n${flag.space}' + - 'optional character, and zero or\n${flag.space}' + - 'more digits, which shall be sorted by arithmetic\n${flag.space}' + - 'value. An empty digit string shall be treated as\n${flag.space}' + - 'zero. Leading zeros shall not affect ordering.') + 'string, consisting of optional characters,\n${flag.space}' + + 'optional character, and zero or\n${flag.space}' + + 'more digits, which shall be sorted by arithmetic\n${flag.space}' + + 'value. An empty digit string shall be treated as\n${flag.space}' + + 'zero. Leading zeros shall not affect ordering.') reverse := fp.bool('reverse', `r`, false, 'reverse the result of comparisons\n\nOther options:') check_diagnose := fp.bool('', `c`, false, 'check for sorted input; do not sort') @@ -78,14 +78,14 @@ fn get_options() Options { numeric: numeric reverse: reverse // other options - check_diagnose: check_diagnose - check_quiet: check_quiet - sort_keys: sort_keys - field_separator: field_separator - merge: merge - output_file: output_file - unique: unique - files: scan_files_arg(files) + check_diagnose: check_diagnose + check_quiet: check_quiet + sort_keys: sort_keys + field_separator: field_separator + merge: merge + output_file: output_file + unique: unique + files: scan_files_arg(files) } } diff --git a/src/stat/stat_to_vlib.v b/src/stat/modes.v similarity index 51% rename from src/stat/stat_to_vlib.v rename to src/stat/modes.v index 42454c1b..d1f98f99 100644 --- a/src/stat/stat_to_vlib.v +++ b/src/stat/modes.v @@ -1,6 +1,14 @@ import os -import v.scanner -import v.pref +import strings + +// This file used to be called `stat_to_vlib.v`: it was a staging copy of the +// file mode helpers on their way into V's os module. The types have since +// landed in vlib/os, so it is no longer a copy of anything. +// +// It also used to import v.scanner/v.pref and decode --printf escapes with the +// V language scanner. Pulling the compiler into a coreutils binary made `v +// build` of this directory take minutes (it effectively hung), so the escape +// handling below is a plain hand-written decoder instead. pub enum FileType { unknown @@ -33,13 +41,6 @@ pub: special bool // setuid for owner, setgid for group, sticky for others } -pub enum FilePermissionGroup { - unknown - owner - group - world -} - @[inline] fn if_str(cond bool, str_if_true string, str_if_false string) string { return if cond { str_if_true } else { str_if_false } @@ -107,13 +108,93 @@ fn filetype_to_string(typ FileType) string { } } -// raw_to_printf_string converts a raw string into a printf-able string -// example: r"\t[\x76]" => "\t[\x76]" = " [v]" +// raw_to_printf_string decodes the backslash escapes accepted by GNU stat's +// --printf. Example: r"\t[\x76]" becomes " [v]". +// +// Recognised: \a \b \e \f \n \r \t \v \\ \" plus \NNN (1-3 octal digits) and +// \xHH (1-2 hex digits). Anything else is passed through as the escaped +// character, with a warning, which is what GNU does. fn raw_to_printf_string(raw_string string) string { - mut sc := scanner.new_scanner("'${raw_string}'", .skip_comments, &pref.Preferences{}) - return sc.scan().lit.replace(r'\a', '\a').replace(r'\b', '\b').replace(r'\e', '\e').replace(r'\f', - '\f').replace(r'\n', '\n').replace(r'\r', '\r').replace(r'\t', '\t').replace(r'\v', '\v').replace(r'\\', - '\\').replace(r"\'", "'").replace(r'\"', '"').replace(r'\?', '\?') + mut sb := strings.new_builder(raw_string.len) + mut i := 0 + for i < raw_string.len { + c := raw_string[i] + if c != `\\` { + sb.write_byte(c) + i++ + continue + } + i++ + if i == raw_string.len { + eprintln('stat: warning: backslash at end of format') + sb.write_byte(`\\`) + break + } + e := raw_string[i] + i++ + match e { + `a` { sb.write_byte(7) } + `b` { sb.write_byte(8) } + `e` { sb.write_byte(27) } + `f` { sb.write_byte(12) } + `n` { sb.write_byte(10) } + `r` { sb.write_byte(13) } + `t` { sb.write_byte(9) } + `v` { sb.write_byte(11) } + `\\` { sb.write_byte(`\\`) } + `"` { sb.write_byte(`"`) } + `0`...`7` { + // \NNN: up to three octal digits + mut value := int(e - `0`) + mut digits := 1 + for digits < 3 && i < raw_string.len && is_octal_digit(raw_string[i]) { + value = value * 8 + int(raw_string[i] - `0`) + i++ + digits++ + } + sb.write_byte(u8(value)) + } + `x` { + // \xHH: up to two hex digits + mut value := 0 + mut digits := 0 + for digits < 2 && i < raw_string.len { + d := hex_digit_value(raw_string[i]) or { break } + value = value * 16 + d + i++ + digits++ + } + if digits == 0 { + eprintln("stat: warning: unrecognized escape '\\x'") + sb.write_byte(`x`) + } else { + sb.write_byte(u8(value)) + } + } + else { + eprintln("stat: warning: unrecognized escape '\\${e}'") + sb.write_byte(e) + } + } + } + return sb.str() +} + +fn is_octal_digit(c u8) bool { + return c >= `0` && c <= `7` +} + +fn hex_digit_value(c u8) !int { + if c >= `0` && c <= `9` { + return int(c - `0`) + } + if c >= `a` && c <= `f` { + return int(c - `a`) + 10 + } + if c >= `A` && c <= `F` { + return int(c - `A`) + 10 + } + return error('not a hex digit') } fn filemode_to_string(mode u16) string { diff --git a/src/stat/stat.c.v b/src/stat/stat.c.v index 0192010b..dfee2a94 100644 --- a/src/stat/stat.c.v +++ b/src/stat/stat.c.v @@ -4,11 +4,55 @@ import os #include #include #include -#include +// The kernel uapi definitions of statx()/STATX_* live here. Do not use +// for that: it is a glibc internal header, so it is missing on +// musl based distributions (see issue #145). +#include + +// vlib/builtin/cfns.c.v already declares a private `fn C.statvfs`, so a +// `fn C.statvfs` here resolves to that one and fails to compile. Give the libc +// call a distinct V-level name instead. +#define vcu_statvfs statvfs + +// Mirror the kernel's `struct statx` exactly. Passing a V struct through a +// voidptr and hoping for the best truncates the kernel's 224 byte write to the +// size of the V view, which corrupts memory. +struct C.statx_timestamp { + tv_sec i64 + tv_nsec u32 + reserved i32 +} // Ref: https://www.man7.org/linux/man-pages/man2/statx.2.html -fn C.statx(int, &char, int, u32, voidptr) int -fn C.statvfs(&char, voidptr) int +struct C.statx { + stx_mask u32 + stx_blksize u32 + stx_attributes u64 + stx_nlink u32 + stx_uid u32 + stx_gid u32 + stx_mode u16 + __spare0 [1]u16 + stx_ino u64 + stx_size u64 + stx_blocks u64 + stx_attributes_mask u64 + stx_atime C.statx_timestamp + stx_btime C.statx_timestamp + stx_ctime C.statx_timestamp + stx_mtime C.statx_timestamp + stx_rdev_major u32 + stx_rdev_minor u32 + stx_dev_major u32 + stx_dev_minor u32 + stx_mnt_id u64 + stx_dio_mem_align u32 + stx_dio_offset_align u32 + __spare3 [12]u32 +} + +fn C.statx(int, &char, int, u32, &C.statx) int +fn C.vcu_statvfs(&char, voidptr) int fn C.readlink(pathname &char, buf &char, bufsiz usize) int const c_at_statx_sync_as_stat = 0x0000 // C.AT_STATX_SYNC_AS_STAT from fcntl.h @@ -18,8 +62,7 @@ const c_at_symlink_nofollow = 0x0100 // C.AT_SYMLINK_NOFOLLOW from fcntl.h const c_chmod_bits = C.S_ISUID | C.S_ISGID | C.S_ISVTX | C.S_IRWXU | C.S_IRWXG | C.S_IRWXO fn statx(path string, dereference bool, cache_mode CacheMode) !Statx { - mut s := Statx{} - ptr := voidptr(&s) + mut c := C.statx{} unsafe { symlink_flag := if dereference { 0 } else { c_at_symlink_nofollow } sync_flag := match cache_mode { @@ -29,19 +72,50 @@ fn statx(path string, dereference bool, cache_mode CacheMode) !Statx { } res := C.statx(0, os.abs_path(path).str, sync_flag | symlink_flag, - C.STATX_BASIC_STATS | C.STATX_BTIME, ptr) + C.STATX_BASIC_STATS | C.STATX_BTIME, &c) if res != 0 { return os.error_posix() } } - return s + return Statx{ + stx_mask: c.stx_mask + stx_blksize: c.stx_blksize + stx_attributes: c.stx_attributes + stx_nlink: c.stx_nlink + stx_uid: c.stx_uid + stx_gid: c.stx_gid + stx_mode: c.stx_mode + stx_ino: c.stx_ino + stx_size: c.stx_size + stx_blocks: c.stx_blocks + stx_attributes_mask: c.stx_attributes_mask + stx_atime: to_timestamp(c.stx_atime) + stx_btime: to_timestamp(c.stx_btime) + stx_ctime: to_timestamp(c.stx_ctime) + stx_mtime: to_timestamp(c.stx_mtime) + stx_rdev_major: c.stx_rdev_major + stx_rdev_minor: c.stx_rdev_minor + stx_dev_major: c.stx_dev_major + stx_dev_minor: c.stx_dev_minor + stx_mnt_id: c.stx_mnt_id + stx_dio_mem_align: c.stx_dio_mem_align + stx_dio_offset_align: c.stx_dio_offset_align + } +} + +fn to_timestamp(ts C.statx_timestamp) StatxTimestamp { + return StatxTimestamp{ + tv_sec: ts.tv_sec + tv_nsec: ts.tv_nsec + } } +// statvfs() passes the V Statvfs struct through a voidptr, so that struct has +// to match the kernel's `struct statvfs` exactly, trailing spare included. fn statvfs(path string) !Statvfs { mut s := Statvfs{} - ptr := voidptr(&s) unsafe { - res := C.statvfs(os.abs_path(path).str, ptr) + res := C.vcu_statvfs(os.abs_path(path).str, voidptr(&s)) if res != 0 { return os.error_posix() } diff --git a/src/stat/stat.v b/src/stat/stat.v index abe71a12..daa4e059 100644 --- a/src/stat/stat.v +++ b/src/stat/stat.v @@ -13,6 +13,11 @@ const app = common.CoreutilInfo{ help: $embed_file('help.txt').to_string() } +// fstypes_data maps the magic numbers the kernel reports for a filesystem to +// its type name. Embedded at module scope: $embed_file inside a function body +// does not resolve on all V versions. +const fstypes_data = $embed_file('fstypes.txt') + // Settings for Utility: stat struct Settings { mut: @@ -66,18 +71,22 @@ struct Statx { stx_dio_offset_align u32 } +// Field types and order mirror the kernel's `struct statvfs`, including the +// trailing spare, because statvfs() hands this struct to libc through a +// voidptr. Every f_* member is `unsigned long`/`fsblkcnt_t`, hence usize. struct Statvfs { - f_bsize u64 // Filesystem block size - f_frsize u64 // Fragment size - f_blocks u64 // Size of fs in f_frsize units - f_bfree u64 // Number of free blocks - f_bavail u64 // Number of free blocks for unprivileged users - f_files u64 // Number of inodes - f_ffree u64 // Number of free inodes - f_favail u64 // Number of free inodes for unprivileged users - f_fsid u64 // Filesystem ID - f_flag u64 // Mount flags - f_namemax u64 // Maximum filename length + f_bsize usize // Filesystem block size + f_frsize usize // Fragment size + f_blocks usize // Size of fs in f_frsize units + f_bfree usize // Number of free blocks + f_bavail usize // Number of free blocks for unprivileged users + f_files usize // Number of inodes + f_ffree usize // Number of free inodes + f_favail usize // Number of free inodes for unprivileged users + f_fsid usize // Filesystem ID + f_flag usize // Mount flags + f_namemax usize // Maximum filename length + f_spare [6]i32 // __f_spare in the C struct; V rejects leading underscores } enum CacheMode { @@ -551,8 +560,7 @@ fn get_mount_list() []MountInfo { } fn get_fs_list() map[string]u32 { - embedded_file := $embed_file('fstypes.txt') - s := embedded_file.to_string() + s := fstypes_data.to_string() assert s[s.len - 1] == `\n`, 'fstypes.txt must be newline-terminated.' mut fslist := map[string]u32{} for i := 0; i < s.len; { diff --git a/src/tail/parse_args.v b/src/tail/parse_args.v index 31f6e8ca..858f67ff 100644 --- a/src/tail/parse_args.v +++ b/src/tail/parse_args.v @@ -34,14 +34,14 @@ fn parse_args(args []string) Args { bytes_arg := fp.string('bytes', `c`, '-1', 'output the last NUM bytes; or use -c + to output ${wrap}' + - 'starting with byte of each file') + 'starting with byte of each file') follow_arg := fp.bool('follow', `f`, false, 'output appended data as the file grows') f_arg := fp.bool('', `F`, false, 'same as --follow=name --retry') lines_arg := fp.string('lines', `n`, '10', 'output the last NUM lines, instead of the last 10; or us${wrap}' + - '-n +NUM to skip NUM-1 lines at the start') + '-n +NUM to skip NUM-1 lines at the start') pid_arg := fp.string('pid', ` `, '', 'with -f, terminate after process ID, PID dies') quiet_arg := fp.bool('quiet', `q`, false, 'never output headers giving file names') @@ -72,9 +72,9 @@ fn parse_args(args []string) Args { files := scan_files_arg(files_arg) return Args{ - bytes: string_to_i64(bytes_arg) or { exit_error(err.msg()) } + bytes: string_to_i64(bytes_arg) or { exit_error(invalid_number('bytes', bytes_arg)) } follow: follow_arg || f_arg - lines: string_to_i64(lines_arg) or { exit_error(err.msg()) } + lines: string_to_i64(lines_arg) or { exit_error(invalid_number('lines', lines_arg)) } pid: pid_arg quiet: quiet_arg || silent_arg retry: f_arg || retry_arg @@ -86,6 +86,12 @@ fn parse_args(args []string) Args { } } +// invalid_number matches the diagnostic GNU tail prints for a +// -c/-n operand it cannot parse. +fn invalid_number(kind string, value string) string { + return 'invalid number of ${kind}: ‘${value}’' +} + fn scan_files_arg(files_arg []string) []string { mut files := []string{} diff --git a/src/tail/tail.v b/src/tail/tail.v index 78d6dbde..7f01d1fb 100644 --- a/src/tail/tail.v +++ b/src/tail/tail.v @@ -1,7 +1,7 @@ // tail - output the last part of files import os import time -import v.mathutil +import math const app_name = 'tail' @@ -82,7 +82,7 @@ fn tail_bytes(file FileInfo, args Args, stat_size u64, out_fn fn (string)) { pos := if args.from_start { args.bytes } else { - mathutil.max(u64(0), stat_size - u64(args.bytes)) + math.max(u64(0), stat_size - u64(args.bytes)) } mut f := os.open(file.name) or { if args.retry { @@ -112,7 +112,7 @@ fn tail_file(file FileInfo, args Args, stat_size u64, out_fn fn (string)) { mut pos := i64(0) loop1: for pos <= end { - len := mathutil.min(end - pos, buf_size) + len := math.min(end - pos, buf_size) f.read_bytes_into(u64(pos), mut buf) or { exit_error(err.msg()) } for i := 0; i < len; i += 1 { @@ -133,8 +133,8 @@ fn tail_file(file FileInfo, args Args, stat_size u64, out_fn fn (string)) { mut pos := end loop2: for pos > 0 { - len := mathutil.min(pos, buf_size) - pos = mathutil.max(pos - buf_size, 0) + len := math.min(pos, buf_size) + pos = math.max(pos - buf_size, 0) f.read_bytes_into(u64(pos), mut buf) or { exit_error(err.msg()) } for i := len - 1; i >= 0; i -= 1 { diff --git a/src/truncate/truncate.v b/src/truncate/truncate.v index 1402f829..a0b918a5 100644 --- a/src/truncate/truncate.v +++ b/src/truncate/truncate.v @@ -202,8 +202,13 @@ fn truncate(settings Settings) { f.close() } if os.exists(fname) { - block_size := if settings.io_blocks { get_block_size(fname) or { - default_block_size} } else { 1 } + block_size := if settings.io_blocks { + get_block_size(fname) or { + default_block_size + } + } else { + 1 + } size := calc_target_size(get_size(settings.reference), settings.size_opt, block_size) os.truncate(fname, size) or { app.quit(message: err.msg()) } @@ -216,8 +221,13 @@ fn truncate(settings Settings) { // If --no-create is set, nothing is done but no error is generated // This is behavior from the original GNU coreutil. if os.exists(fname) { - block_size := if settings.io_blocks { get_block_size(fname) or { - default_block_size} } else { 1 } + block_size := if settings.io_blocks { + get_block_size(fname) or { + default_block_size + } + } else { + 1 + } size := calc_target_size(get_size(fname), settings.size_opt, block_size) os.truncate(fname, size) or { app.quit(message: err.msg()) } } diff --git a/src/unlink/unlink.c.v b/src/unlink/unlink.c.v index 6a00b43e..e450f253 100644 --- a/src/unlink/unlink.c.v +++ b/src/unlink/unlink.c.v @@ -1,6 +1,7 @@ import os #include + $if !windows { #include } diff --git a/src/users/users.c.v b/src/users/users.c.v index b7e3b2dc..9fb26cc4 100644 --- a/src/users/users.c.v +++ b/src/users/users.c.v @@ -1,23 +1,19 @@ import common -pub struct C.utmpx { - ut_type i16 // Type of login. - ut_pid int // Process ID of login process. - ut_line &char // Devicename. - ut_id &char // Inittab ID. - ut_user &char // Username. -} +// `struct utmpx` is picked up from via common/readutmp_nix.c.v. A +// hand written copy here listed only the fields it needed and dropped the +// fixed size char arrays, so its element size did not match the C struct. fn utmp_users(filename &char) []string { mut utmp_buf := []C.utmpx{} - mut users := []string{} + mut names := []string{} common.read_utmp(filename, mut utmp_buf, .user_process) unsafe { for u in utmp_buf { - users << cstring_to_vstring(u.ut_user) + names << cstring_to_vstring(&u.ut_user[0]) } } // Obtain sorted order as GNU coreutils - users.sort() - return users + names.sort() + return names } diff --git a/src/users/users.v b/src/users/users.v index 121fa227..3a89d358 100644 --- a/src/users/users.v +++ b/src/users/users.v @@ -13,9 +13,9 @@ mut: } fn users(settings Settings) { - users := utmp_users(settings.input_file).join(' ') - print(users) - if users != '' { + names := utmp_users(settings.input_file).join(' ') + print(names) + if names != '' { print(common.eol()) } } diff --git a/src/wc/wc.v b/src/wc/wc.v index e928a36d..ba8ff5ff 100644 --- a/src/wc/wc.v +++ b/src/wc/wc.v @@ -59,7 +59,7 @@ fn get_count(chunk FileChunk, last_line_length u32) (Count, u32) { } line_length = 0 } - space, carriage_return, vertical_tab, form_feed { + space, vertical_tab, form_feed { prev_char_is_space = true line_length++ } @@ -96,14 +96,19 @@ mut: mutex sync.Mutex } -fn (mut file_reader FileReader) read_chunk(mut buffer []u8) ?FileChunk { +// read_chunk reads the next buffer_size bytes. os.File.read signals +// exhaustion with an os.Eof error, which the caller has to tell apart from a +// genuine read failure. +fn (mut file_reader FileReader) read_chunk(mut buffer []u8) !FileChunk { file_reader.mutex.@lock() defer { file_reader.mutex.unlock() } - nbytes := - file_reader.file.read(mut buffer) or { return none } // Propagate error. Either EOF or read error. + nbytes := file_reader.file.read(mut buffer) or { return err } + if nbytes == 0 { + return os.Eof{} + } mut chunk := FileChunk{file_reader.last_char_is_space, buffer[..nbytes].clone(), false} file_reader.last_char_is_space = is_space(buffer[nbytes - 1]) if nbytes < buffer.len { @@ -121,16 +126,14 @@ fn file_reader_counter(mut file_reader FileReader) Count { for { chunk := file_reader.read_chunk(mut buffer) or { match err { - none { - // EOF 'error', just break out of the loop. + os.Eof { break } else { - println(err) + eprintln('${application_name}: ${err}') + exit(1) } } - - exit(1) } count, line_length = get_count(chunk, line_length) @@ -256,7 +259,8 @@ fn main() { max_line_length_len := max_line_length.str().len mut col_size := int(0) - if byte(bytes_opt) + byte(chars_opt) + byte(lines_opt) + byte(words_opt) + byte(maxline_opt) == 1 { + // A single selected count is printed without a header or column padding. + if [bytes_opt, chars_opt, lines_opt, words_opt, maxline_opt].filter(it).len == 1 { col_size = 0 } else { if total_line_count_len > col_size { From c541a45d51a1eee72836290f816679ddd48ac66f Mon Sep 17 00:00:00 2001 From: mike-ward Date: Mon, 30 Dec 2024 14:54:04 -0600 Subject: [PATCH 2/3] add total block size --- src/ls/format_long.v | 450 +++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 450 insertions(+) create mode 100644 src/ls/format_long.v diff --git a/src/ls/format_long.v b/src/ls/format_long.v new file mode 100644 index 00000000..b8d34b18 --- /dev/null +++ b/src/ls/format_long.v @@ -0,0 +1,450 @@ +import arrays +import os +import term +import time +import v.mathutil { max } + +const inode_title = 'inode' +const permissions_title = 'Permission' +const mask_title = 'Mask' +const links_title = 'Links' +const owner_title = 'Owner' +const group_title = 'Group' +const size_title = 'Size' +const date_modified_title = 'Modified' +const date_accessed_title = 'Accessed' +const date_status_title = 'Status Change' +const name_title = 'Name' +const unknown = '?' +const block_size = 5 +const space = ' ' +const date_format = 'MMM DD HH:mm' +const date_iso_format = 'YYYY-MM-DD HH:mm:ss' +const date_compact_format = "DD MMM'YY HH:mm" +const date_compact_format_with_day = "ddd DD MMM'YY HH:mm" + +struct Longest { + inode int + nlink int + owner_name int + group_name int + size int + checksum int + file int +} + +enum StatTime { + accessed + changed + modified +} + +fn print_total(entries []Entry, options Options) { + total := arrays.fold[Entry, u64](entries, 0, fn (a u64, e Entry) u64 { + return a + max(u64(1), e.size / 1024) + }) + println('total: ${total}') +} + +fn format_long_listing(entries []Entry, options Options) { + longest := longest_entries(entries, options) + header, cols := format_header(options, longest) + header_len := real_length(header) + term_cols, _ := term.get_terminal_size() + + print_total(entries, options) + print_header(header, options, header_len, cols) + print_header_border(options, header_len, cols) + + dim := if options.no_dim { no_style } else { dim_style } + + for idx, entry in entries { + // emit blank row every 5th row + if options.blocked_output { + if idx % block_size == 0 && idx != 0 { + match options.table_format { + true { print(border_row_middle(header_len, cols)) } + else { print_newline() } + } + } + } + + // left table border + if options.table_format { + print(table_border_pad_left) + } + + // inode + if options.inode { + content := if entry.invalid { unknown } else { entry.stat.inode.str() } + print(format_cell(content, longest.inode, Align.right, no_style, options)) + print_space() + } + + // checksum + if options.checksum != '' { + checksum := format_cell(entry.checksum, longest.checksum, .left, dim, options) + print(checksum) + print_space() + } + + // permissions + if !options.no_permissions { + content := permissions(entry, options) + print(format_cell(content, permissions_title.len, .right, no_style, options)) + print_space() + } + + // octal permissions + if options.octal_permissions { + content := format_octal_permissions(entry, options) + print(format_cell(content, 4, .left, dim, options)) + print_space() + } + + // hard links + if !options.no_hard_links { + content := if entry.invalid { unknown } else { '${entry.stat.nlink}' } + print(format_cell(content, longest.nlink, .right, dim, options)) + print_space() + } + + // owner name + if !options.no_owner_name { + content := if entry.invalid { unknown } else { get_owner_name(entry.stat.uid) } + print(format_cell(content, longest.owner_name, .right, dim, options)) + print_space() + } + + // group name + if !options.no_group_name { + content := if entry.invalid { unknown } else { get_group_name(entry.stat.gid) } + print(format_cell(content, longest.group_name, .right, dim, options)) + print_space() + } + + // size + if !options.no_size { + content := match true { + // vfmt off + entry.invalid { unknown } + options.size_ki && !options.size_kb { entry.size_ki } + options.size_kb { entry.size_kb } + else { entry.size.str() } + // vfmt on + } + size_style := match entry.link_stat.size > 0 { + true { get_style_for_link(entry, options) } + else { get_style_for_entry(entry, options) } + } + size := format_cell(content, longest.size, .right, size_style, options) + print(size) + print_space() + } + + // date/time(modified) + if !options.no_date { + print(format_time(entry, .modified, options)) + print_space() + } + + // date/time (accessed) + if options.accessed_date { + print(format_time(entry, .accessed, options)) + print_space() + } + + // date/time (status change) + if options.changed_date { + print(format_time(entry, .changed, options)) + print_space() + } + + // file name + file_name := format_entry_name(entry, options) + file_style := get_style_for_entry(entry, options) + match options.table_format { + true { print(format_cell(file_name, longest.file, .left, file_style, options)) } + else { print(format_cell(file_name, 0, .left, file_style, options)) } + } + + // line too long? Print a '≈' in the last column + if options.no_wrap { + mut coord := term.get_cursor_position() or { term.Coord{} } + if coord.x >= term_cols { + coord.x = term_cols + term.set_cursor_position(coord) + print('≈') + } + } + + print_newline() + } + + // bottom border + print_bottom_border(options, header_len, cols) +} + +fn longest_entries(entries []Entry, options Options) Longest { + return Longest{ + // vfmt off + inode: longest_inode_len(entries, inode_title, options) + nlink: longest_nlink_len(entries, links_title, options) + owner_name: longest_owner_name_len(entries, owner_title, options) + group_name: longest_group_name_len(entries, group_title, options) + size: longest_size_len(entries, size_title, options) + checksum: longest_checksum_len(entries, options.checksum, options) + file: longest_file_name_len(entries, name_title, options) + // vfmt on + } +} + +fn print_header(header string, options Options, len int, cols []int) { + if options.header { + if options.table_format { + print(border_row_top(len, cols)) + } + println(header) + } +} + +fn format_header(options Options, longest Longest) (string, []int) { + mut buffer := '' + mut cols := []int{} + dim := if options.no_dim || options.table_format { no_style } else { dim_style } + table_pad := if options.table_format { table_border_pad_left } else { '' } + + if options.table_format { + buffer += table_border_pad_left + } + if options.inode { + title := if options.header { inode_title } else { '' } + buffer += left_pad(title, longest.inode) + table_pad + cols << real_length(buffer) - 1 + } + if options.checksum != '' { + title := if options.header { options.checksum.capitalize() } else { '' } + width := longest.checksum + buffer += right_pad(title, width) + table_pad + cols << real_length(buffer) - 1 + } + if !options.no_permissions { + buffer += 'T ${table_pad}' + cols << real_length(buffer) - 1 + buffer += left_pad(permissions_title, permissions_title.len) + table_pad + cols << real_length(buffer) - 1 + } + if options.octal_permissions { + buffer += left_pad(mask_title, mask_title.len) + table_pad + cols << real_length(buffer) - 1 + } + if !options.no_hard_links { + title := if options.header { links_title } else { '' } + buffer += left_pad(title, longest.nlink) + table_pad + cols << real_length(buffer) - 1 + } + if !options.no_owner_name { + title := if options.header { owner_title } else { '' } + buffer += left_pad(title, longest.owner_name) + table_pad + cols << real_length(buffer) - 1 + } + if !options.no_group_name { + title := if options.header { group_title } else { '' } + buffer += left_pad(title, longest.group_name) + table_pad + cols << real_length(buffer) - 1 + } + if !options.no_size { + title := if options.header { size_title } else { '' } + buffer += left_pad(title, longest.size) + table_pad + cols << real_length(buffer) - 1 + } + if !options.no_date { + title := if options.header { date_modified_title } else { '' } + width := time_format(options).len + buffer += right_pad(title, width) + table_pad + cols << real_length(buffer) - 1 + } + if options.accessed_date { + title := if options.header { date_accessed_title } else { '' } + width := time_format(options).len + buffer += right_pad(title, width) + table_pad + cols << real_length(buffer) - 1 + } + if options.changed_date { + title := if options.header { date_status_title } else { '' } + width := time_format(options).len + buffer += right_pad(title, width) + table_pad + cols << real_length(buffer) - 1 + } + + buffer += right_pad_end(if options.header { name_title } else { '' }, longest.file) // drop last space + header := format_cell(buffer, 0, .left, dim, options) + return header, cols +} + +fn time_format(options Options) string { + return match true { + // vfmt off + options.time_iso { date_iso_format } + options.time_compact { date_compact_format } + options.time_compact_with_day { date_compact_format_with_day } + else { date_format } + // vfmt on + } +} + +fn left_pad(s string, width int) string { + pad := width - s.len + return if pad > 0 { space.repeat(pad) + s + space } else { s + space } +} + +fn right_pad(s string, width int) string { + pad := width - s.len + return if pad > 0 { s + space.repeat(pad) + space } else { s + space } +} + +fn right_pad_end(s string, width int) string { + pad := width - s.len + return if pad > 0 { s + space.repeat(pad) } else { s } +} + +fn statistics(entries []Entry, len int, options Options) { + file_count := entries.filter(it.file).len + total := arrays.sum(entries.map(if it.file || it.exe { it.stat.size } else { 0 })) or { 0 } + dir_count := entries.filter(it.dir).len + link_count := entries.filter(it.link).len + mut stats := '' + + dim := if options.no_dim { no_style } else { dim_style } + file_count_styled := style_string(file_count.str(), options.style_fi, options) + + file := if file_count == 1 { 'file' } else { 'files' } + files := style_string(file, dim, options) + dir_count_styled := style_string(dir_count.str(), options.style_di, options) + + dir := if dir_count == 1 { 'directory' } else { 'directories' } + dirs := style_string(dir, dim, options) + + size := match true { + options.size_ki { readable_size(total, true) } + options.size_kb { readable_size(total, false) } + else { total.str() } + } + + totals := style_string(size, options.style_fi, options) + stats = '${dir_count_styled} ${dirs} | ${file_count_styled} ${files} [${totals}]' + + if link_count > 0 { + link_count_styled := style_string(link_count.str(), options.style_ln, options) + links := style_string('links', dim, options) + stats += ' | ${link_count_styled} ${links}' + } + println(stats) +} + +fn file_flag(entry Entry, options Options) string { + return match true { + // vfmt off + entry.invalid { unknown } + entry.link { style_string('l', options.style_ln, options) } + entry.dir { style_string('d', options.style_di, options) } + entry.exe { style_string('x', options.style_ex, options) } + entry.fifo { style_string('p', options.style_pi, options) } + entry.block { style_string('b', options.style_bd, options) } + entry.character { style_string('c', options.style_cd, options) } + entry.socket { style_string('s', options.style_so, options) } + entry.file { style_string('-', options.style_fi, options) } + else { '?' } + // vfmt on + } +} + +fn format_octal_permissions(entry Entry, options Options) string { + mode := entry.stat.get_mode() + return '0${mode.owner.bitmask()}${mode.group.bitmask()}${mode.others.bitmask()}' +} + +fn permissions(entry Entry, options Options) string { + mode := entry.stat.get_mode() + flag := file_flag(entry, options) + owner := file_permission(mode.owner, options) + group := file_permission(mode.group, options) + other := file_permission(mode.others, options) + return '${flag}${owner}${group}${other}' +} + +fn file_permission(file_permission os.FilePermission, options Options) string { + dim := if options.no_dim { no_style } else { dim_style } + dash := style_string('-', dim, options) + r := if file_permission.read { style_string('r', options.style_ln, options) } else { dash } + w := if file_permission.write { style_string('w', options.style_fi, options) } else { dash } + x := if file_permission.execute { style_string('x', options.style_ex, options) } else { dash } + return '${r}${w}${x}' +} + +fn format_time(entry Entry, stat_time StatTime, options Options) string { + entry_time := match stat_time { + .accessed { entry.stat.atime } + .changed { entry.stat.ctime } + .modified { entry.stat.mtime } + } + + mut date := time.unix(entry_time) + .local() + .custom_format(time_format(options)) + + if date.starts_with('0') { + date = ' ' + date[1..] + } + + dim := if options.no_dim { no_style } else { dim_style } + content := if entry.invalid { '?' + space.repeat(date.len - 1) } else { date } + return format_cell(content, date.len, .left, dim, options) +} + +fn longest_nlink_len(entries []Entry, title string, options Options) int { + lengths := entries.map(it.stat.nlink.str().len) + max := arrays.max(lengths) or { 0 } + return if options.no_hard_links || !options.header { max } else { max(max, title.len) } +} + +fn longest_owner_name_len(entries []Entry, title string, options Options) int { + lengths := entries.map(get_owner_name(it.stat.uid).len) + max := arrays.max(lengths) or { 0 } + return if options.no_owner_name || !options.header { max } else { max(max, title.len) } +} + +fn longest_group_name_len(entries []Entry, title string, options Options) int { + lengths := entries.map(get_group_name(it.stat.gid).len) + max := arrays.max(lengths) or { 0 } + return if options.no_group_name || !options.header { max } else { max(max, title.len) } +} + +fn longest_size_len(entries []Entry, title string, options Options) int { + lengths := entries.map(match true { + it.dir { 1 } + options.size_ki && !options.size_kb { it.size_ki.len } + options.size_kb { it.size_kb.len } + else { it.size.str().len } + }) + max := arrays.max(lengths) or { 0 } + return if options.no_size || !options.header { max } else { max(max, title.len) } +} + +fn longest_inode_len(entries []Entry, title string, options Options) int { + lengths := entries.map(it.stat.inode.str().len) + max := arrays.max(lengths) or { 0 } + return if !options.inode || !options.header { max } else { max(max, title.len) } +} + +fn longest_file_name_len(entries []Entry, title string, options Options) int { + lengths := entries.map(real_length(format_entry_name(it, options))) + max := arrays.max(lengths) or { 0 } + return if !options.header { max } else { max(max, title.len) } +} + +fn longest_checksum_len(entries []Entry, title string, options Options) int { + lengths := entries.map(it.checksum.len) + max := arrays.max(lengths) or { 0 } + return if !options.header { max } else { max(max, title.len) } +} From 0fb08891bf83ebadd7fd9aa3911b496497877040 Mon Sep 17 00:00:00 2001 From: metif12 Date: Sat, 3 Oct 2026 01:43:35 +0330 Subject: [PATCH 3/3] ls: bring the listing, the long format and the option parser up to GNU MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Takes over #183, whose author handed it on, and moves it onto current V so that it builds at all. The option parser used `v.mathutil`, which no longer exists, six helpers each shadowed `max` with a local and then called `max`, and a `C.readlink` that vlib/os now declares with a different signature. On top of that, the behaviour was compared against GNU ls 9.4 on the same machine over a matrix of 49 invocations. It started at 0 of 49 and is now 39 of them. Every number below was read off the reference. Where the first reading turned out to be wrong, the measurement is named rather than the correction. ## Things that were simply wrong * The type character came from `entry.exe`, which means "has the execute bit", so a 755 file was listed as `xrwxr-xr-x`. GNU has no `x` type at all. * A symlink's size was the size of its target, so a link to `/nonexistent` showed 0. It is the link's own size, which is the length of the target string: 12 for `/nonexistent`, 9 for `plain.txt`. * Every column was padded to the widest entry, so every line ended in spaces. GNU pads all but the last column of a row and nothing else. * Columns were laid out even with the output redirected. `ls | cat` gives one entry per line at GNU; only a terminal, or `-C`, gets a grid. * An empty directory lost its section header entirely, because it produces no entries to group by. `ls emptydir anotherdir` printed only the other one. * `os.join_path` rewrites a backslash in a name into a separator, so `./back\slash` became `./back/slash` and the entry failed to stat, printing as `?---------`. Joining on `/` directly is what fixed it. * An operand that could not be reached was listed anyway and the exit status stayed 0. GNU prints `cannot access` and exits 2. ## The long format `total` is the sum of the disk blocks in 1K units, not of the file lengths: a 200000 byte file on a 4K filesystem contributes 196, not 195. It has no colon. V's `os.Stat` has no `st_blocks`, so the field is read from a `C.stat` declared alongside the one vlib declares — the field lists have to match exactly or vlib's own call to `C.lstat` stops type checking. `-h` and `--si` were the same calculation with the base and the suffix swapped: `-h` divides by 1024 and writes `K`, `--si` divides by 1000 and writes `k`. The rounding is a ceiling, not to nearest. GNU prints 2500 bytes as 2.5K where nearest would give 2.4K, and 10240 as 10K while 10241 is 11K; one decimal is kept below ten and dropped at or above. All 33 sizes sampled from the reference fit that, and no nearest-rounding rule does. The day is padded with a space, `Oct 3`, because GNU uses `%e`; V's `custom_format` has no token for that (`_2` is passed through literally, and `2` gives an unpadded day). Anything older than six calendar months shows the year instead of the time of day. That boundary is calendar months, not a fixed count of days: with now = 2026-10-03 the reference showed a time for 2026-04-04 and a year for 2026-04-01. ## Quoting `--quoting-style` was not implemented and exited 2. Each style has its own rule and all of them were measured: * `literal` never quotes. There is no `none`; GNU 9.4 rejects it and lists the ten words it does accept. * `shell` quotes a name only when it holds something outside the set a shell treats literally, in single quotes, or in double quotes when the name itself contains a single quote. A newline stays inside the quotes: `'new line'`, with the quotes still open on the far side. * `shell-escape` closes the run at a byte that is not printable ASCII and writes `$'\n'`, `$'\t'` or a three digit octal escape, so a name with a tab is `'tab'$'\t''name'` and `ü` is `''$'\303\274''n'`. * `c` is always double quoted with C escapes. `-Q` is this, not shell quoting: `ls -Q` prints `"new\nline"`, which is worth knowing. The default follows the destination, shell-escape to a terminal and literal to a pipe. In the long listing the link target is quoted as well as the name. ## Added `-F` and `-d` were absent, and both are ordinary rather than exotic. `-F` marks directories, links, sockets, fifos and devices, executables and unknown types, and leaves an ordinary file alone; in the long listing it marks everything except a link, which already has `-> target` after it. `-d` lists the operand rather than its contents, so a bare `ls -d` prints `.`. An earlier comment on this PR said GNU sorts names case-insensitively, and the default arm did compare byte-wise. Measuring it settled the question the other way: with a directory holding `Apple BANANA ZZZ _under aPPle apple banana`, GNU prints them in that order under `LC_ALL=C`, `C.UTF-8` and `en_US.UTF-8` alike, which is byte order — folding case would have put `_under` second. The report appears to have come from a case-insensitive filesystem, where `Apple` and `apple` cannot both exist. The closure in `sort` is now bound through a typed variable. V 0.5.2 does not give the arms of a `match` of closures a type it will pass to `sorted_with_compare`: it looks for a function named after the variable instead and reports `Uint128.cmp`, which has nothing to do with this module. ## Testing 49 invocations over one directory holding a directory, an empty directory, a dangling and a working symlink, an executable, a 3000 and a 200000 byte file, names with a space, a single quote, a dollar, a backslash, a semicolon, an asterisk, a tab and an embedded newline, each with a distinct mtime so that `-t` compares times rather than nanosecond ties. Both binaries run in the same directory with the same arguments. 39 match. The ten that do not, and why: * `--time-style=iso`, `=long-iso` and `=full-iso` are not implemented. This is a gap, not a difference of opinion; the port has its own `time_iso` and compact formats but never read the option. `octal` is in this group only because the case was written with `--time-style` in it. * `-C`, `-x` and `-w` size every column to the widest name in the whole listing. GNU sizes each column to its own contents, which is how it fits six entries to a line where this fits five. * `-m` wraps at the width, but GNU picks where to break from the same column calculation as `-C`, so the two disagree about where the third line ends. * `-li -l` and `-AC` fail in the parser. A repeated short flag makes `flag.to_struct` panic, and a bundled one is not expanded. That is vlib's, not this module's. `entry_test.v` asserted the old human-readable sizes, including the `b` suffix GNU does not print and 1024-based maths under the `--si` flag. It now pins the values read off the reference, including the two places where a nearest rounding would disagree. ## Still the author's extras The icons, the checksum column, the bordered table format and `--table` are left as they were. They are not GNU options and are not compared here. Stacked on #195 for the build fixes. `v fmt -verify .` passes over `src/ls`, `v -o bin/ls ./ls` builds, and both `src/ls` test files pass. --- src/ls/auto_wrap.v | 19 ++ src/ls/entry.v | 229 ++++++++++++++++++ src/ls/entry_test.v | 60 +++++ src/ls/filter.v | 9 + src/ls/format.v | 395 ++++++++++++++++++++++++++++++ src/ls/format_long.v | 108 +++++++-- src/ls/icons.v | 440 ++++++++++++++++++++++++++++++++++ src/ls/ls.v | 177 +++++--------- src/ls/ls_nix.c.v | 99 ++++++++ src/ls/ls_windows.c.v | 51 ++++ src/ls/natural_compare.v | 54 +++++ src/ls/natural_compare_test.v | 52 ++++ src/ls/options.v | 186 ++++++++++++++ src/ls/sort.v | 75 ++++++ src/ls/style.v | 176 ++++++++++++++ src/ls/table.v | 68 ++++++ 16 files changed, 2056 insertions(+), 142 deletions(-) create mode 100644 src/ls/auto_wrap.v create mode 100644 src/ls/entry.v create mode 100644 src/ls/entry_test.v create mode 100644 src/ls/filter.v create mode 100644 src/ls/format.v create mode 100644 src/ls/icons.v create mode 100644 src/ls/ls_nix.c.v create mode 100644 src/ls/ls_windows.c.v create mode 100644 src/ls/natural_compare.v create mode 100644 src/ls/natural_compare_test.v create mode 100644 src/ls/options.v create mode 100644 src/ls/sort.v create mode 100644 src/ls/style.v create mode 100644 src/ls/table.v diff --git a/src/ls/auto_wrap.v b/src/ls/auto_wrap.v new file mode 100644 index 00000000..682cdf71 --- /dev/null +++ b/src/ls/auto_wrap.v @@ -0,0 +1,19 @@ +import os + +fn set_auto_wrap(options Options) { + if options.no_wrap { + wrap_off := '\e[?7l' + wrap_reset := '\e[?7h' + println(wrap_off) + + at_exit(fn [wrap_reset] () { + println(wrap_reset) + }) or {} + + // Ctrl-C handler + os.signal_opt(os.Signal.int, fn (sig os.Signal) { + println('\e[?7h') + exit(0) + }) or {} + } +} diff --git a/src/ls/entry.v b/src/ls/entry.v new file mode 100644 index 00000000..ccc29455 --- /dev/null +++ b/src/ls/entry.v @@ -0,0 +1,229 @@ +import os +import crypto.md5 +import crypto.sha1 +import crypto.sha256 +import crypto.sha512 +import crypto.blake2b +import math + +struct Entry { + name string + dir_name string + stat os.Stat + link_stat os.Stat + dir bool + file bool + link bool + exe bool + fifo bool + block bool + socket bool + character bool + unknown bool + link_origin string + size u64 + allocated u64 + size_ki string + size_kb string + checksum string + invalid bool // lstat could not access +} + +fn get_entries(files []string, options Options) ([]Entry, []string, int) { + mut entries := []Entry{cap: 50} + mut dirs := []string{} + mut status := 0 + + for file in files { + if options.list_directory_itself { + // -d lists the operand, not what is inside it. GNU shows `.` for a + // bare `ls -d`, which is the name it was given. + entries << make_entry(file, '', options) + continue + } + if os.is_dir(file) { + dir_files := os.ls(file) or { + status = 2 // see help for meaning of exit codes + eprintln(err) + continue + } + // An empty directory still needs a section of its own, and a group + // keyed by it, or `ls emptydir anotherdir` would lose its header. + dirs << file + entries << match options.all { + true { dir_files.map(make_entry(it, file, options)) } + else { dir_files.filter(!is_dot_file(it)).map(make_entry(it, file, options)) } + } + } else { + if os.lstat(file) or { os.Stat{} }.mode == 0 { + // GNU treats an operand it cannot reach as a serious problem: + // nothing is listed for it and the exit status is 2. + eprintln("${app_name}: cannot access '${file}': No such file or directory") + status = 2 + continue + } + if options.all || !is_dot_file(file) { + entries << make_entry(file, '', options) + } + } + } + return entries, dirs, status +} + +// os.join_path is not usable here: it rewrites a backslash in a file name into +// a separator, so `./back\slash` becomes `./back/slash` and the entry then +// fails to stat. Measured, not assumed. A name cannot contain a slash, so +// joining on one is safe on POSIX and accepted on Windows too. +fn entry_path(dir_name string, name string) string { + if dir_name == '' { + return name + } + if dir_name.ends_with('/') { + return dir_name + name + } + return dir_name + '/' + name +} + +fn make_entry(file string, dir_name string, options Options) Entry { + mut invalid := false + path := entry_path(dir_name, file) + + stat := os.lstat(path) or { + // println('${path} -> ${err.msg()}') + invalid = true + os.Stat{} + } + + filetype := stat.get_filetype() + is_link := filetype == .symbolic_link + link_origin := if is_link { read_link(path) } else { '' } + mut size := stat.size + mut link_stat := os.Stat{} + + if is_link && options.long_format && !invalid { + // os.stat follows the link. The link itself is still what the size + // column reports, so only the styling reads this: a link to a + // directory is coloured as one. GNU shows 12 for a link to + // /nonexistent, which is the length of the target string. + link_stat = os.stat(path) or { os.Stat{} } + } + + is_dir := filetype == .directory + is_fifo := filetype == .fifo + is_block := filetype == .block_device + is_socket := filetype == .socket + is_character_device := filetype == .character_device + is_unknown := filetype == .unknown + is_exe := !is_dir && is_executable(stat) + is_file := filetype == .regular + // -p appends a slash to directories and -F marks the type. They overlap on a + // directory and agree there. In the long listing -F marks everything except a + // symlink: GNU puts `-> target` after one instead of `@`. An ordinary file + // gets nothing at all, and `?` is for a type GNU cannot name. + indicator := if is_dir && (options.dir_indicator || options.classify) { + '/' + } else if options.classify && !(options.long_format && is_link) { + if is_link { + '@' + } else if is_socket { + '=' + } else if is_fifo || is_block || is_character_device { + '%' + } else if is_exe { + '*' + } else if is_unknown { + '?' + } else { + '' + } + } else { + '' + } + + return Entry{ + // vfmt off + name: file + indicator + dir_name: dir_name + stat: stat + link_stat: link_stat + dir: is_dir + file: is_file + link: is_link + exe: is_exe + fifo: is_fifo + block: is_block + socket: is_socket + character: is_character_device + unknown: is_unknown + link_origin: link_origin + size: size + allocated: allocated_size(path) + size_ki: if options.size_ki { readable_size(size, true) } else { '' } + size_kb: if options.size_kb { readable_size(size, false) } else { '' } + checksum: if is_file { checksum(file, dir_name, options) } else { '' } + invalid: invalid + // vfmt on + } +} + +// GNU's -h divides by 1024 and uses an uppercase suffix, --si divides by 1000 +// and uses a lowercase one. Neither prints a `b`, and a value below the base is +// printed as it stands rather than as 1.0K. +// +// The rounding is a ceiling, not to nearest: 2500 bytes prints as 2.5K, not +// 2.4K, and 10240 as 10K while 10241 is 11K. One decimal is kept below ten and +// dropped at or above it, so 4096 is 4.0K and 102400 is 100K. All 33 sizes +// sampled from the reference fit this, and no nearest-rounding rule does. +fn readable_size(size u64, si bool) string { + base := if si { f64(1000) } else { f64(1024) } + units := if si { + ['', 'k', 'm', 'g', 't', 'p', 'e', 'z', 'y'] + } else { + ['', 'K', 'M', 'G', 'T', 'P', 'E', 'Z', 'Y'] + } + mut sz := f64(size) + for i, unit in units { + if sz < base || i == units.len - 1 { + if unit == '' { + return size.str() + } + tenths := int(math.ceil(sz * 10)) + if tenths >= 100 { + return '${int(math.ceil(sz))}${unit}' + } + return '${tenths / 10}.${tenths % 10}${unit}' + } + sz /= base + } + return size.str() +} + +fn checksum(name string, dir_name string, options Options) string { + if options.checksum == '' { + return '' + } + file := entry_path(dir_name, name) + bytes := os.read_bytes(file) or { return unknown } + + return match options.checksum { + // vfmt off + 'md5' { md5.sum(bytes).hex() } + 'sha1' { sha1.sum(bytes).hex() } + 'sha224' { sha256.sum224(bytes).hex() } + 'sha256' { sha256.sum256(bytes).hex() } + 'sha512' { sha512.sum512(bytes).hex() } + 'blake2b' { blake2b.sum256(bytes).hex() } + else { unknown } + // vfmt on + } +} + +@[inline] +fn is_executable(stat os.Stat) bool { + return stat.get_mode().bitmask() & 0b001001001 > 0 +} + +@[inline] +fn is_dot_file(file string) bool { + return file.starts_with('.') +} diff --git a/src/ls/entry_test.v b/src/ls/entry_test.v new file mode 100644 index 00000000..671d0aeb --- /dev/null +++ b/src/ls/entry_test.v @@ -0,0 +1,60 @@ +module main + +// The expected values here were read off GNU ls 9.4, not derived from the +// implementation. Two rules are being pinned down: +// +// -h divides by 1024 and uses an uppercase suffix, --si divides by 1000 and +// uses a lowercase one. The `si` argument is therefore true for --si. +// +// The rounding is a ceiling, not to nearest. GNU prints 2500 bytes as 2.5K +// where rounding to nearest would give 2.4K, and 10240 as 10K while 10241 is +// 11K. One decimal is kept below ten and dropped at or above it, so 4096 is +// 4.0K and 102400 is 100K. +fn test_readable_size() { + // Below the base a size is printed as it stands, with no suffix. + assert readable_size(0, true) == '0' + assert readable_size(2, true) == '2' + assert readable_size(1023, false) == '1023' + assert readable_size(100, false) == '100' + + // 1024 is the first value that -h scales, and 1000 the first for --si, so + // 1023 is still unscaled by -h and already 1.1k under --si. + assert readable_size(1024, false) == '1.0K' + assert readable_size(1024, true) == '1.1k' + assert readable_size(1023, true) == '1.1k' + + // The ceiling, at each of the two points where a nearest rounding would + // disagree. + assert readable_size(2500, false) == '2.5K' + assert readable_size(2900, false) == '2.9K' + assert readable_size(2999, false) == '3.0K' + assert readable_size(1025, false) == '1.1K' + + // Exact multiples keep their exact value rather than gaining a tenth. + assert readable_size(2048, false) == '2.0K' + assert readable_size(4096, false) == '4.0K' + assert readable_size(5120, false) == '5.0K' + + // At ten the decimal goes away. + assert readable_size(10239, false) == '10K' + assert readable_size(10240, false) == '10K' + assert readable_size(10241, false) == '11K' + + assert readable_size(4095, false) == '4.0K' + assert readable_size(4097, false) == '4.1K' + assert readable_size(10_000, false) == '9.8K' + assert readable_size(20_000, false) == '20K' + assert readable_size(200_000, false) == '196K' + assert readable_size(102_400, false) == '100K' + + // --si, on a 1000 base with a lowercase suffix. + assert readable_size(4095, true) == '4.1k' + assert readable_size(100_000, true) == '100k' + assert readable_size(200_000, true) == '200k' + + // Larger units still round up, and the mantissa stays below ten. + // 100000000 / 1024 / 1024 is 95.37, which the ceiling rule takes to 96. + assert readable_size(100_000_000, false) == '96M' + assert readable_size(100_000_000, true) == '100m' + assert readable_size(8_000_000_000_000_000_000, true) == '8.0e' +} diff --git a/src/ls/filter.v b/src/ls/filter.v new file mode 100644 index 00000000..5a523b20 --- /dev/null +++ b/src/ls/filter.v @@ -0,0 +1,9 @@ +fn filter(entries []Entry, options Options) []Entry { + return match true { + // vfmt off + options.only_dirs { entries.clone().filter(it.dir) } + options.only_files { entries.clone().filter(it.file) } + else { entries } + // vfmt on + } +} diff --git a/src/ls/format.v b/src/ls/format.v new file mode 100644 index 00000000..381211e7 --- /dev/null +++ b/src/ls/format.v @@ -0,0 +1,395 @@ +import arrays +import os +import term +import math + +const cell_max = 12 // limit on wide displays +const cell_spacing = 3 // space between cells + +enum Align { + left + right +} + +fn print_files(entries_arg []Entry, options Options) { + entries := match true { + options.all && !options.almost_all { + dot := make_entry('.', '.', options) + dot_dot := make_entry('..', '.', options) + arrays.concat([dot, dot_dot], ...entries_arg) + } + else { + entries_arg + } + } + + w, _ := term.get_terminal_size() + options_width_ok := options.width_in_cols > 0 && options.width_in_cols < 1000 + // GNU falls back to 80 columns when there is no terminal, so ls -C | cat + // still makes a grid. Without this the column layouts collapse to one + // entry per line as soon as the output is redirected. + width := if options_width_ok { + options.width_in_cols + } else if w > 0 { + w + } else { + 80 + } + + match true { + // vfmt off + options.long_format { format_long_listing(entries, options) } + options.list_by_lines { format_by_lines(entries, width, options) } + options.with_commas { format_with_commas(entries, width, options) } + options.one_per_line { format_one_per_line(entries, options) } + options.list_by_columns { format_by_cells(entries, width, options) } + // GNU fills the width with columns only when the output is a terminal. + // Piped or redirected it lists one entry per line, so `ls | cat` and + // `ls` in a terminal do not agree unless -C was given. + os.is_atty(1) == 0 { format_one_per_line(entries, options) } + else { format_by_cells(entries, width, options) } + // vfmt on + } +} + +fn format_by_cells(entries []Entry, width int, options Options) { + len := entries.max_name_len(options) + cell_spacing + cols := math.min(width / len, cell_max) + max_cols := math.max(cols, 1) + partial_row := entries.len % max_cols != 0 + rows := entries.len / max_cols + if partial_row { 1 } else { 0 } + max_rows := math.max(1, rows) + + for r := 0; r < max_rows; r += 1 { + for c := 0; c < max_cols; c += 1 { + idx := r + c * max_rows + if idx < entries.len { + entry := entries[idx] + name := format_entry_name(entry, options) + // The last column of a row is not padded: GNU never ends a + // line with the spaces that would align it to the next one. + width_here := if c == max_cols - 1 { 0 } else { len } + cell := format_cell(name, width_here, .left, get_style_for_entry(entry, + options), options) + print(cell) + } + } + print_newline() + } +} + +fn format_by_lines(entries []Entry, width int, options Options) { + len := entries.max_name_len(options) + cell_spacing + cols := math.min(width / len, cell_max) + max_cols := math.max(cols, 1) + + for i, entry in entries { + if i % max_cols == 0 && i != 0 { + print_newline() + } + name := format_entry_name(entry, options) + width_here := if i % max_cols == max_cols - 1 { 0 } else { len } + cell := format_cell(name, width_here, .left, get_style_for_entry(entry, options), options) + print(cell) + } + print_newline() +} + +fn format_one_per_line(entries []Entry, options Options) { + for entry in entries { + // -i on its own is `%i %n`: the inode and then the name, no other + // column. GNU puts one space between them. + prefix := if options.inode { '${entry.stat.inode} ' } else { '' } + println('${prefix}${format_cell(format_entry_name(entry, options), 0, .left, get_style_for_entry(entry, options), options)}') + } +} + +// -m fills the width with a comma separated list. The comma is written after +// every entry but the last and the space before the next one is only written +// when there is room for it, so a wrapped line ends in `large.bin,` and not in +// `large.bin, ` the way padding each cell would leave it. +fn format_with_commas(entries []Entry, width int, options Options) { + last := entries.len - 1 + mut line := 0 + for i, entry in entries { + name := format_entry_name(entry, options) + if line > 0 { + if line + 1 + real_length(name) > width { + print_newline() + line = 0 + } else { + print(' ') + line += 1 + } + } + print(format_cell(name, 0, .left, no_style, options)) + if i < last { + print(',') + } + line += real_length(name) + } + print_newline() +} + +fn format_cell(s string, width int, align Align, style Style, options Options) string { + return match options.table_format { + true { format_table_cell(s, width, align, style, options) } + else { format_cell_content(s, width, align, style, options) } + } +} + +fn format_cell_content(s string, width int, align Align, style Style, options Options) string { + mut cell := '' + no_ansi_s := term.strip_ansi(s) + pad := width - no_ansi_s.runes().len + + if align == .right && pad > 0 { + cell += space.repeat(pad) + } + + cell += if options.colorize == when_always { + style_string(s, style, options) + } else { + no_ansi_s + } + + if align == .left && pad > 0 { + cell += space.repeat(pad) + } + + return cell +} + +fn format_table_cell(s string, width int, align Align, style Style, options Options) string { + cell := format_cell_content(s, width, align, style, options) + return '${cell}${table_border_pad_right}' +} + +// surrounds a cell with table borders +fn print_dir_name(name string, options Options, printed_any &bool) { + if name.len > 0 { + // A blank line separates sections, but there is none before the first. + if *printed_any { + print_newline() + } + unsafe { + *printed_any = true + } + nm := if options.colorize == when_always { + style_string(name, options.style_di, options) + } else { + name + } + println('${nm}:') + } +} + +fn (entries []Entry) max_name_len(options Options) int { + lengths := entries.map(real_length(format_entry_name(it, options))) + return arrays.max(lengths) or { 0 } +} + +fn get_style_for_entry(entry Entry, options Options) Style { + return match true { + // vfmt off + entry.link { options.style_ln } + entry.dir { options.style_di } + entry.exe { options.style_ex } + entry.fifo { options.style_pi } + entry.block { options.style_bd } + entry.character { options.style_cd } + entry.socket { options.style_so } + entry.file { options.style_fi } + else { no_style } + // vfmt on + } +} + +fn get_style_for_link(entry Entry, options Options) Style { + if entry.link_stat.size == 0 { + return unknown_style + } + + filetype := entry.link_stat.get_filetype() + is_dir := filetype == os.FileType.directory + is_fifo := filetype == .fifo + is_block := filetype == .block_device + is_socket := filetype == .socket + is_character_device := filetype == .character_device + is_unknown := filetype == .unknown + is_exe := is_executable(entry.link_stat) + is_file := !is_dir && !is_fifo && !is_block && !is_socket && !is_character_device && !is_unknown + && !is_exe + + return match true { + // vfmt off + is_dir { options.style_di } + is_exe { options.style_ex } + is_fifo { options.style_pi } + is_block { options.style_bd } + is_character_device { options.style_cd } + is_socket { options.style_so } + is_unknown { unknown_style } + is_file { options.style_fi } + else { no_style } + // vfmt on + } +} + +fn format_entry_name(entry Entry, options Options) string { + name := if options.relative_path { + entry_path(entry.dir_name, entry.name) + } else { + entry.name + } + + icon := get_icon_for_entry(entry, options) + + // The quoting style defaults per destination: a terminal gets shell rules, + // a pipe gets none. -Q is GNU's older spelling of --quoting-style=c, not of + // shell quoting: measured, `ls -Q` prints "new\nline" with a C escape. + style := if options.quote && options.quoting_style == '' { + 'c' + } else if options.quoting_style != '' { + options.quoting_style + } else if os.is_atty(1) != 0 { + 'shell-escape' + } else { + 'literal' + } + + return match true { + entry.link && options.long_format { + link_style := get_style_for_link(entry, options) + link := style_string(quote_name(entry.link_origin, style), link_style, options) + '${icon}${quote_name(name, style)} -> ${link}' + } + else { + '${icon}${quote_name(name, style)}' + } + } +} + +// GNU's --quoting-style, measured on the reference rather than recalled. +// `literal` never quotes. `shell` quotes only a name that needs it, `shell- +// always` quotes every one, and both use shell rules: single quotes normally, +// double quotes when the name contains a single quote, and the quoted run +// closes and reopens around an embedded newline so the name stays one argument. +// `c` is always double quoted with C escapes. +fn quote_name(name string, style string) string { + return match style { + 'c' { c_quote(name) } + 'shell' { shell_quote(name, false) } + 'shell-always' { shell_quote(name, true) } + 'shell-escape' { shell_escape_quote(name, false) } + 'shell-escape-always' { shell_escape_quote(name, true) } + else { name } + } +} + +// shell-escape differs from shell in how it treats a byte that is not printable +// ASCII: instead of leaving a newline or a tab inside the single quotes, it +// closes the run and writes $'\n' or a three digit octal escape, so a name with +// a tab in it comes out as 'tab'$'\t''name'. Bytes at or above 0x80 are escaped +// the same way, which is how GNU writes ü as ''$'\303\274''. +fn shell_escape_quote(name string, always bool) string { + if !always && !needs_quoting(name) { + return name + } + mut out := '' + mut run := '' + mut i := 0 + for i < name.len { + b := name[i] + if b >= 0x20 && b < 0x7f { + run += name[i..i + 1] + i++ + continue + } + out += quote_run(run) + run = '' + out += match b { + `\n` { "$'\\n'" } + `\t` { "$'\\t'" } + `\r` { "$'\\r'" } + else { "$'\\${b:03o}'" } + } + i++ + } + return out + quote_run(run) +} + +// A name needs no quoting when it is made only of characters a shell treats +// literally, which is the same set GNU uses to decide this. A newline needs +// quoting and a shell is happy to have it inside single quotes, which is why a +// two line name comes out as 'newline' with the quotes still open. +fn needs_quoting(name string) bool { + if name.len == 0 { + return true + } + for ch in name { + if !(ch >= `a` && ch <= `z`) && !(ch >= `A` && ch <= `Z`) && !(ch >= `0` && ch <= `9`) + && ch != `@` && ch != `%` && ch != `_` && ch != `+` && ch != `=` && ch != `:` && ch != `,` + && ch != `.` && ch != `/` && ch != `-` { + return true + } + } + return false +} + +fn shell_quote(name string, always bool) string { + if !always && !needs_quoting(name) { + return name + } + return quote_run(name) +} + +// quote_run wraps one run of characters in the quotes a shell would use. +fn quote_run(run string) string { + if run.len == 0 { + return '' + } + // A single quote cannot appear inside single quotes, so such a name is + // double quoted instead, with the double quote, backslash and dollar + // escaped. + if run.contains("'") { + mut out := '"' + for i in 0 .. run.len { + if run[i] == `"` || run[i] == `\\` || run[i] == `$` { + out += '\\' + } + out += run[i..i + 1] + } + return out + '"' + } + return "'${run}'" +} + +fn c_quote(name string) string { + mut out := '"' + for i in 0 .. name.len { + out += match name[i] { + `"` { '\\"' } + `\\` { '\\\\' } + `\n` { '\\n' } + `\t` { '\\t' } + `\r` { '\\r' } + else { name[i..i + 1] } + } + } + return out + '"' +} + +fn real_length(s string) int { + return term.strip_ansi(s).runes().len +} + +@[inline] +fn print_space() { + print_character(` `) +} + +@[inline] +fn print_newline() { + print_character(`\n`) +} diff --git a/src/ls/format_long.v b/src/ls/format_long.v index b8d34b18..6110e44f 100644 --- a/src/ls/format_long.v +++ b/src/ls/format_long.v @@ -2,7 +2,7 @@ import arrays import os import term import time -import v.mathutil { max } +import math const inode_title = 'inode' const permissions_title = 'Permission' @@ -39,11 +39,20 @@ enum StatTime { modified } +// GNU prints `total 220`, with no colon, and the number is the sum of the disk +// blocks the entries occupy in 1K units -- not the sum of their lengths. A +// 200000 byte file on a 4K filesystem contributes 196, not 195. With -h the sum +// is scaled like a size instead, giving `total 220K`. An empty directory gets +// `total 0` rather than no line at all. fn print_total(entries []Entry, options Options) { - total := arrays.fold[Entry, u64](entries, 0, fn (a u64, e Entry) u64 { - return a + max(u64(1), e.size / 1024) + allocated := arrays.fold[Entry, u64](entries, 0, fn (a u64, e Entry) u64 { + return a + e.allocated }) - println('total: ${total}') + if options.size_kb || options.size_ki { + println('total ${readable_size(allocated, options.size_ki)}') + return + } + println('total ${allocated / 1024}') } fn format_long_listing(entries []Entry, options Options) { @@ -348,7 +357,6 @@ fn file_flag(entry Entry, options Options) string { entry.invalid { unknown } entry.link { style_string('l', options.style_ln, options) } entry.dir { style_string('d', options.style_di, options) } - entry.exe { style_string('x', options.style_ex, options) } entry.fifo { style_string('p', options.style_pi, options) } entry.block { style_string('b', options.style_bd, options) } entry.character { style_string('c', options.style_cd, options) } @@ -382,6 +390,37 @@ fn file_permission(file_permission os.FilePermission, options Options) string { return '${r}${w}${x}' } +// GNU shows the time of day for anything from the last six months and the year +// for anything older. The boundary is calendar months rather than a fixed count +// of days: measured on the reference with now = 2026-10-03, a file dated +// 2026-04-04 showed a time and one dated 2026-04-01 showed a year, so the mark +// sits on 2026-04-03. +fn is_recent(t time.Time) bool { + now := time.now() + mut cy := now.year + mut cm := now.month - 6 + if cm < 1 { + cm += 12 + cy -= 1 + } + // The day of the month may not exist six months back, the 31st of a month + // with 30 days being the case that bites. + cd := math.min(now.day, time.days_in_month(cm, cy) or { 31 }) + if t.year != cy || t.month != cm { + return (t.year > cy) || (t.year == cy && t.month > cm) + } + if t.day != cd { + return t.day > cd + } + if t.hour != now.hour { + return t.hour > now.hour + } + if t.minute != now.minute { + return t.minute > now.minute + } + return t.second >= now.second +} + fn format_time(entry Entry, stat_time StatTime, options Options) string { entry_time := match stat_time { .accessed { entry.stat.atime } @@ -389,12 +428,21 @@ fn format_time(entry Entry, stat_time StatTime, options Options) string { .modified { entry.stat.mtime } } - mut date := time.unix(entry_time) - .local() - .custom_format(time_format(options)) + t := time.unix(entry_time).local() + // Only the default style switches to the year; the explicit --time-style + // ones are left to print what they were asked for. + mut date := if time_format(options) == date_format && !is_recent(t) { + // `Nov 15 2023`: the day is padded to width two and then two spaces + // separate it from the year. + day := if t.day < 10 { ' ${t.day}' } else { '${t.day}' } + '${t.smonth()} ${day} ${t.year}' + } else { + t.custom_format(time_format(options)) + } - if date.starts_with('0') { - date = ' ' + date[1..] + if date.len > 5 && date[4] == `0` && date[5] >= `0` && date[5] <= `9` + && !date.starts_with('YYYY') { + date = date[..4] + ' ' + date[5..] } dim := if options.no_dim { no_style } else { dim_style } @@ -404,20 +452,32 @@ fn format_time(entry Entry, stat_time StatTime, options Options) string { fn longest_nlink_len(entries []Entry, title string, options Options) int { lengths := entries.map(it.stat.nlink.str().len) - max := arrays.max(lengths) or { 0 } - return if options.no_hard_links || !options.header { max } else { max(max, title.len) } + longest := arrays.max(lengths) or { 0 } + return if options.no_hard_links || !options.header { + longest + } else { + math.max(longest, title.len) + } } fn longest_owner_name_len(entries []Entry, title string, options Options) int { lengths := entries.map(get_owner_name(it.stat.uid).len) - max := arrays.max(lengths) or { 0 } - return if options.no_owner_name || !options.header { max } else { max(max, title.len) } + longest := arrays.max(lengths) or { 0 } + return if options.no_owner_name || !options.header { + longest + } else { + math.max(longest, title.len) + } } fn longest_group_name_len(entries []Entry, title string, options Options) int { lengths := entries.map(get_group_name(it.stat.gid).len) - max := arrays.max(lengths) or { 0 } - return if options.no_group_name || !options.header { max } else { max(max, title.len) } + longest := arrays.max(lengths) or { 0 } + return if options.no_group_name || !options.header { + longest + } else { + math.max(longest, title.len) + } } fn longest_size_len(entries []Entry, title string, options Options) int { @@ -427,24 +487,24 @@ fn longest_size_len(entries []Entry, title string, options Options) int { options.size_kb { it.size_kb.len } else { it.size.str().len } }) - max := arrays.max(lengths) or { 0 } - return if options.no_size || !options.header { max } else { max(max, title.len) } + longest := arrays.max(lengths) or { 0 } + return if options.no_size || !options.header { longest } else { math.max(longest, title.len) } } fn longest_inode_len(entries []Entry, title string, options Options) int { lengths := entries.map(it.stat.inode.str().len) - max := arrays.max(lengths) or { 0 } - return if !options.inode || !options.header { max } else { max(max, title.len) } + longest := arrays.max(lengths) or { 0 } + return if !options.inode || !options.header { longest } else { math.max(longest, title.len) } } fn longest_file_name_len(entries []Entry, title string, options Options) int { lengths := entries.map(real_length(format_entry_name(it, options))) - max := arrays.max(lengths) or { 0 } - return if !options.header { max } else { max(max, title.len) } + longest := arrays.max(lengths) or { 0 } + return if !options.header { longest } else { math.max(longest, title.len) } } fn longest_checksum_len(entries []Entry, title string, options Options) int { lengths := entries.map(it.checksum.len) - max := arrays.max(lengths) or { 0 } - return if !options.header { max } else { max(max, title.len) } + longest := arrays.max(lengths) or { 0 } + return if !options.header { longest } else { math.max(longest, title.len) } } diff --git a/src/ls/icons.v b/src/ls/icons.v new file mode 100644 index 00000000..1d55ed1d --- /dev/null +++ b/src/ls/icons.v @@ -0,0 +1,440 @@ +import os + +fn get_icon_for_entry(entry Entry, options Options) string { + if !options.icons { + return '' + } + ext := os.file_ext(entry.name) + name := os.file_name(entry.name) + return match entry.dir { + true { get_icon_for_folder(name) } + else { get_icon_for_file(name, ext) } + } +} + +fn get_icon_for_file(name string, ext string) string { + // default icon for all files. try to find a better one though... + mut icon := icons_map['file'] + // resolve aliased extensions + mut ext_key := ext.to_lower() + if ext.starts_with('.') { + ext_key = ext_key[1..] + } + alias := aliases_map[ext_key] + if alias != '' { + ext_key = alias + } + // see if we can find a better icon based on extension alone + better_icon := icons_map[ext_key] + if better_icon != '' { + icon = better_icon + } + // now look for icons based on full names + mut full_name := name.to_lower() + full_alias := aliases_map[full_name] + if full_alias != '' { + full_name = full_alias + } + best_icon := icons_map[full_name] + if best_icon != '' { + icon = best_icon + } + return icon + space +} + +fn get_icon_for_folder(name string) string { + mut icon := folders_map['folder'] + better_icon := folders_map[name] + if better_icon != '' { + icon = better_icon + } + return icon + space +} + +const icons_map = { + 'ai': '\ue7b4' + 'android': '\ue70e' + 'apple': '\uf179' + 'as': '\ue60b' + 'asm': '󰘚' + 'audio': '\uf1c7' + 'avro': '\ue60b' + 'bf': '\uf067' + 'binary': '\uf471' + 'bzl': '\ue63a' + 'c': '\ue61e' + 'cfg': '\uf423' + 'clj': '\ue768' + 'coffee': '\ue751' + 'conf': '\ue615' + 'cpp': '\ue61d' + 'cfm': '\ue645' + 'cr': '\ue62f' + 'cs': '\ue648' + 'cson': '\ue601' + 'css': '\ue749' + 'cu': '\ue64b' + 'd': '\ue7af' + 'dart': '\ue64c' + 'db': '\uf1c0' + 'deb': '\uf306' + 'diff': '\uf440' + 'doc': '\uf1c2' + 'dockerfile': '\ue650' + 'dpkg': '\uf17c' + 'ebook': '\uf02d' + 'elm': '\ue62c' + 'env': '\uf462' + 'erl': '\ue7b1' + 'ex': '\ue62d' + 'f': '󱈚' + 'file': '\uf15b' + 'font': '\uf031' + 'fs': '\ue7a7' + 'gb': '\ue272' + 'gform': '\uf298' + 'git': '\ue702' + 'go': '\ue724' + 'graphql': '\ue662' + 'glp': '󰆧' + 'groovy': '\ue775' + 'gruntfile.js': '\ue74c' + 'gulpfile.js': '\ue610' + 'gv': '\ue225' + 'h': '\uf0fd' + 'haml': '\ue664' + 'hs': '\ue777' + 'html': '\uf13b' + 'hx': '\ue666' + 'ics': '\uf073' + 'image': '\uf1c5' + 'iml': '\ue7b5' + 'ini': '󰅪' + 'ino': '\ue255' + 'iso': '󰋊' + 'jade': '\ue66c' + 'java': '\ue738' + 'jenkinsfile': '\ue767' + 'jl': '\ue624' + 'js': '\ue781' + 'json': '\ue60b' + 'jsx': '\ue7ba' + 'key': '\uf43d' + 'ko': '\uebc6' + 'kt': '\ue634' + 'less': '\ue758' + 'lock': '\uf023' + 'log': '\uf18d' + 'lua': '\ue620' + 'maintainers': '\uf0c0' + 'makefile': '\ue20f' + 'md': '\uf48a' + 'mjs': '\ue718' + 'ml': '󰘧' + 'mustache': '\ue60f' + 'nc': '󰋁' + 'nim': '\ue677' + 'nix': '\uf313' + 'npmignore': '\ue71e' + 'package': '󰏗' + 'passwd': '\uf023' + 'patch': '\uf440' + 'pdf': '\uf1c1' + 'php': '\ue608' + 'pl': '\ue7a1' + 'prisma': '\ue684' + 'ppt': '\uf1c4' + 'psd': '\ue7b8' + 'py': '\ue606' + 'r': '\ue68a' + 'rb': '\ue21e' + 'rdb': '\ue76d' + 'rpm': '\uf17c' + 'rs': '\ue7a8' + 'rss': '\uf09e' + 'rst': '󰅫' + 'rubydoc': '\ue73b' + 'sass': '\ue603' + 'scala': '\ue737' + 'shell': '\uf489' + 'shp': '󰙞' + 'sol': '󰡪' + 'sqlite': '\ue7c4' + 'styl': '\ue600' + 'svelte': '\ue697' + 'swift': '\ue755' + 'tex': '\u222b' + 'tf': '\ue69a' + 'toml': '󰅪' + 'ts': '󰛦' + 'twig': '\ue61c' + 'txt': '\uf15c' + 'v': '𝕍' + 'vagrantfile': '\ue21e' + 'video': '\uf03d' + 'vim': '\ue62b' + 'vue': '\ue6a0' + 'windows': '\uf17a' + 'xls': '\uf1c3' + 'xml': '\ue796' + 'yml': '\ue601' + 'zig': '\ue6a9' + 'zip': '\uf410' +} + +const aliases_map = { + 'apk': 'android' + 'gradle': 'android' + 'ds_store': 'apple' + 'localized': 'apple' + 'm': 'apple' + 'mm': 'apple' + 's': 'asm' + 'aac': 'audio' + 'alac': 'audio' + 'flac': 'audio' + 'm4a': 'audio' + 'mka': 'audio' + 'mp3': 'audio' + 'ogg': 'audio' + 'opus': 'audio' + 'wav': 'audio' + 'wma': 'audio' + 'b': 'bf' + 'bson': 'binary' + 'feather': 'binary' + 'mat': 'binary' + 'o': 'binary' + 'pb': 'binary' + 'pickle': 'binary' + 'pkl': 'binary' + 'tfrecord': 'binary' + 'conf': 'cfg' + 'config': 'cfg' + 'cljc': 'clj' + 'cljs': 'clj' + 'editorconfig': 'conf' + 'rc': 'conf' + 'c++': 'cpp' + 'cc': 'cpp' + 'cxx': 'cpp' + 'scss': 'css' + 'sql': 'db' + 'docx': 'doc' + 'gdoc': 'doc' + 'dockerignore': 'dockerfile' + 'epub': 'ebook' + 'ipynb': 'ebook' + 'mobi': 'ebook' + 'f03': 'f' + 'f77': 'f' + 'f90': 'f' + 'f95': 'f' + 'for': 'f' + 'fpp': 'f' + 'ftn': 'f' + 'eot': 'font' + 'otf': 'font' + 'ttf': 'font' + 'woff': 'font' + 'woff2': 'font' + 'fsi': 'fs' + 'fsscript': 'fs' + 'fsx': 'fs' + 'dna': 'gb' + 'gitattributes': 'git' + 'gitconfig': 'git' + 'gitignore': 'git' + 'gitignore_global': 'git' + 'gitmirrorall': 'git' + 'gitmodules': 'git' + 'gltf': 'glp' + 'gsh': 'groovy' + 'gvy': 'groovy' + 'gy': 'groovy' + 'h++': 'h' + 'hh': 'h' + 'hpp': 'h' + 'hxx': 'h' + 'lhs': 'hs' + 'htm': 'html' + 'xhtml': 'html' + 'bmp': 'image' + 'cbr': 'image' + 'cbz': 'image' + 'dvi': 'image' + 'eps': 'image' + 'gif': 'image' + 'ico': 'image' + 'jpeg': 'image' + 'jpg': 'image' + 'nef': 'image' + 'orf': 'image' + 'pbm': 'image' + 'pgm': 'image' + 'png': 'image' + 'pnm': 'image' + 'ppm': 'image' + 'pxm': 'image' + 'sixel': 'image' + 'stl': 'image' + 'svg': 'image' + 'tif': 'image' + 'tiff': 'image' + 'webp': 'image' + 'xpm': 'image' + 'disk': 'iso' + 'dmg': 'iso' + 'img': 'iso' + 'ipsw': 'iso' + 'smi': 'iso' + 'vhd': 'iso' + 'vhdx': 'iso' + 'vmdk': 'iso' + 'jar': 'java' + 'cjs': 'js' + 'properties': 'json' + 'webmanifest': 'json' + 'tsx': 'jsx' + 'cjsx': 'jsx' + 'cer': 'key' + 'crt': 'key' + 'der': 'key' + 'gpg': 'key' + 'p7b': 'key' + 'pem': 'key' + 'pfx': 'key' + 'pgp': 'key' + 'license': 'key' + 'codeowners': 'maintainers' + 'credits': 'maintainers' + 'cmake': 'makefile' + 'justfile': 'makefile' + 'markdown': 'md' + 'mkd': 'md' + 'rdoc': 'md' + 'readme': 'md' + 'mli': 'ml' + 'sml': 'ml' + 'netcdf': 'nc' + 'brewfile': 'package' + 'cargo.toml': 'package' + 'cargo.lock': 'package' + 'go.mod': 'package' + 'go.sum': 'package' + 'pyproject.toml': 'package' + 'poetry.lock': 'package' + 'package.json': 'package' + 'pipfile': 'package' + 'pipfile.lock': 'package' + 'php3': 'php' + 'php4': 'php' + 'php5': 'php' + 'phpt': 'php' + 'phtml': 'php' + 'gslides': 'ppt' + 'pptx': 'ppt' + 'pxd': 'py' + 'pyc': 'py' + 'pyx': 'py' + 'whl': 'py' + 'rdata': 'r' + 'rds': 'r' + 'rmd': 'r' + 'gemfile': 'rb' + 'gemspec': 'rb' + 'guardfile': 'rb' + 'procfile': 'rb' + 'rakefile': 'rb' + 'rspec': 'rb' + 'rspec_parallel': 'rb' + 'rspec_status': 'rb' + 'ru': 'rb' + 'erb': 'rubydoc' + 'slim': 'rubydoc' + 'awk': 'shell' + 'bash': 'shell' + 'bash_history': 'shell' + 'bash_profile': 'shell' + 'bashrc': 'shell' + 'csh': 'shell' + 'fish': 'shell' + 'ksh': 'shell' + 'ps1': 'shell' + 'sh': 'shell' + 'zsh': 'shell' + 'zsh-theme': 'shell' + 'zshrc': 'shell' + 'plpgsql': 'sql' + 'plsql': 'sql' + 'psql': 'sql' + 'tsql': 'sql' + 'sl3': 'sqlite' + 'sqlite3': 'sqlite' + 'stylus': 'styl' + 'cls': 'tex' + 'avi': 'video' + 'flv': 'video' + 'm2v': 'video' + 'mkv': 'video' + 'mov': 'video' + 'mp4': 'video' + 'mpeg': 'video' + 'mpg': 'video' + 'ogm': 'video' + 'ogv': 'video' + 'vob': 'video' + 'webm': 'video' + 'vimrc': 'vim' + 'bat': 'windows' + 'cmd': 'windows' + 'exe': 'windows' + 'csv': 'xls' + 'gsheet': 'xls' + 'xlsx': 'xls' + 'plist': 'xml' + 'xul': 'xml' + 'yaml': 'yml' + '7z': 'zip' + 'Z': 'zip' + 'bz2': 'zip' + 'gz': 'zip' + 'lzma': 'zip' + 'par': 'zip' + 'rar': 'zip' + 'tar': 'zip' + 'tc': 'zip' + 'tgz': 'zip' + 'txz': 'zip' + 'xz': 'zip' + 'z': 'zip' +} + +const folders_map = { + '.atom': '\ue764' + '.aws': '\ue7ad' + '.docker': '\ue7b0' + '.gem': '\ue21e' + '.git': '\ue5fb' + '.git-credential-cache': '\ue5fb' + '.github': '\ue5fd' + '.npm': '\ue5fa' + '.nvm': '\ue718' + '.rvm': '\ue21e' + '.Trash': '\uf1f8' + '.vscode': '\ue70c' + '.vim': '\ue62b' + 'config': '\ue5fc' + 'folder': '\uf07c' + 'hidden': '\uf023' + 'node_modules': '\ue5fa' +} + +const other_icons_map = { + 'link': '\uf0c1' + 'linkDir': '\uf0c1' + 'brokenLink': '\uf127' + 'device': '\uf0a0' + 'socket': '\uf1e6' + 'pipe': '\ufce3' +} diff --git a/src/ls/ls.v b/src/ls/ls.v index e4a60f27..050df767 100644 --- a/src/ls/ls.v +++ b/src/ls/ls.v @@ -1,133 +1,74 @@ -import os -import common +import arrays { group_by } +import datatypes { Set } +import math -struct Directory { - name string -mut: - contents []string +fn main() { + options, files := get_args() + set_auto_wrap(options) + entries, dirs, status := get_entries(files, options) + mut cyclic := Set[string]{} + mut printed_any := false + status1 := ls(entries, dirs, options, mut cyclic, &printed_any, '') + exit(math.max(status, status1)) } -fn go_print(file_list []Directory, seperator string) { - mut constructed := '' - if file_list.len == 1 { - // Single directory - for i, contents in file_list[0].contents { - constructed += contents - if i == file_list[0].contents.len - 1 { - break - } - constructed += seperator - } - } else { - // Multiple directories - for i, dir in file_list { - constructed += dir.name + ':\n' - for j, contents in dir.contents { - constructed += contents - if j == dir.contents.len - 1 { - break - } - constructed += seperator - } - if i != file_list.len - 1 { - constructed += '\n\n' - } +// section names the directory this call is listing, so that one holding nothing +// still gets its header. GNU prints `./Zdir:` above no rows at all rather than +// skipping it. dirs carries the directory operands so that an empty one still +// gets a section: it produces no entries, so grouping alone would lose it. +fn ls(entries []Entry, dirs []string, options Options, mut cyclic Set[string], printed_any &bool, section string) int { + mut status := 0 + group_by_dirs := group_by[string, Entry](entries, fn (e Entry) string { + return e.dir_name + }) + mut sections := dirs.clone() + for name in group_by_dirs.keys() { + if !sections.contains(name) { + sections << name } } - print(constructed) -} + sorted_dirs := sections.sorted() -fn main() { - mut fp := common.flag_parser(os.args) - fp.application('ls') - fp.description('list directory contents') - - arg_1 := fp.bool('', `1`, false, 'list one file per line') - arg_all := fp.bool('all', `a`, false, 'do not ignore entries starting with .') - arg_almost_all := fp.bool('almost-all', `A`, false, 'do not list implied . and ..') - arg_comma_seperated := fp.bool('comma-seperated', `m`, false, - 'fill width with a comma seperated list of entries') - arg_reverse := fp.bool('reverse', `r`, false, 'reverse order wile sorting') - arg_help := fp.bool('help', 0, false, 'display this help and exit') - - // Get folders - args := fp.finalize() or { - eprintln(err) - println(fp.usage()) - exit(1) - } - - // Help command - if arg_help { - println(fp.usage()) - exit(0) + if sorted_dirs.len == 0 { + // Nothing came back from the directory, but its header still has to be + // printed and the long listing still has to say `total 0`. + print_dir_name(section, options, printed_any) + print_files([], options) + return 0 } - // Get dir / dirs - mut file_list := match args.len { - 0 { - list := os.ls('.') or { - eprintln(err) - println("ls: cannot access '.': No such file or directory") - exit(1) - } - - [Directory{'.', list}] + for dir in sorted_dirs { + files := group_by_dirs[dir] + filtered := filter(files, options) + sorted := sort(filtered, options) + if sections.len > 1 || options.recursive { + print_dir_name(dir, options, printed_any) } - else { - // 1 or more dirs - mut dirs := []Directory{} - for arg in args { - name := if args.len > 1 { - arg - } else { - '.' - } - list := os.ls(arg) or { - eprintln(err) - println("ls: cannot access '" + arg + "': No such file or directory") - exit(1) - } - dirs << Directory{name.replace('/', ''), list} + print_files(sorted, options) + // A header that follows rows is still preceded by a blank line, even + // when the rows came from an unnamed section such as a bare operand. + if sorted.len > 0 { + unsafe { + *printed_any = true } - dirs } - } - - // Define initial seperator - mut seperator := ' ' - - // Modify seperator - if arg_comma_seperated { - seperator = ', ' - } - if arg_1 { - seperator += '\n' - } - // Do not list dotfiles by default - if !(arg_all || arg_almost_all) { - for i, dir in file_list { - file_list[i].contents = dir.contents.filter(fn (contents string) bool { - return contents[0] != `.` - }) - } - } - - // . and .. path listing - if arg_all { - for i, _ in file_list { - file_list[i].contents.prepend(['.', '..']) - } - } - - // Reverse - if arg_reverse { - for i, _ in file_list { - file_list[i].contents.reverse_in_place() + if options.recursive { + for entry in sorted { + if entry.dir { + path := entry_path(entry.dir_name, entry.name) + if cyclic.exists(path) { + println('===> cyclic reference detected <===') + continue + } + cyclic.add(path) + dir_entries, sub_dirs, status1 := get_entries([path], options) + status2 := ls(dir_entries, sub_dirs, options, mut cyclic, printed_any, path) + cyclic.remove(path) + status = math.max(status1, status2) + } + } } } - - // Print - go_print(file_list, seperator) + return status } diff --git a/src/ls/ls_nix.c.v b/src/ls/ls_nix.c.v new file mode 100644 index 00000000..1eecf04d --- /dev/null +++ b/src/ls/ls_nix.c.v @@ -0,0 +1,99 @@ +import os + +#include +#include +#include + +struct Passwd { + pw_name &char + pw_uid usize + pw_gid usize + pw_dir &char + pw_shell &char +} + +struct Group { + gr_name &char + gr_gid usize + gr_mem &&char +} + +fn C.getpwuid(uid usize) &Passwd +fn C.getgrgid(uid usize) &Group + +// V's os.Stat has no st_blocks, and GNU's `total` line is the sum of those in +// 1K units, so the port cannot get the number any other way. vlib/os has the +// field in its own C.stat but drops it when it builds the platform-agnostic +// Stat, and it is not reachable as os.C.stat. Declaring the same struct here +// works because the field list is identical to the one vlib/os declares; a +// different name or layout breaks vlib's own call to C.lstat, which is the +// error this replaces. st_blocks was checked against `stat -c %b` rather than +// assumed, since the layout above is reproduced here. +struct C.stat { + st_dev u64 + st_ino u64 + st_nlink u64 + st_mode u32 + st_uid u32 + st_gid u32 + st_rdev u64 + st_size u64 + st_blksize u64 + st_blocks u64 + st_atime i64 + st_mtime i64 + st_ctime i64 +} + +fn C.lstat(path &char, buf &C.stat) int + +// allocated_size returns the space the entry occupies on disk, which is not +// what GNU's -h prints: a 3000 byte file on a 4K filesystem is shown as 3.0K, +// and a 200000 byte one as 196K. st_blocks is counted in 512 byte units. +fn allocated_size(path string) u64 { + mut buf := C.stat{} + unsafe { + if C.lstat(path.str, &buf) != 0 { + return 0 + } + } + return buf.st_blocks * 512 +} + +fn get_owner_name(uid usize) string { + pwd := C.getpwuid(uid) + unsafe { + if isnil(pwd) { + // Call succeeded but user not found + if C.errno == 0 { + return '' + } + return os.error_posix().msg() + } + return cstring_to_vstring(pwd.pw_name) + } +} + +fn get_group_name(uid usize) string { + grp := C.getgrgid(uid) + unsafe { + if isnil(grp) { + // Call succeeded but user not found + if C.errno == 0 { + return '' + } + return os.error_posix().msg() + } + return cstring_to_vstring(grp.gr_name) + } +} + +fn read_link(file string) string { + buf_size := 2048 + buf := '\0'.repeat(buf_size) + len := C.readlink(file.str, buf.str, usize(buf_size)) + if len == -1 { + return os.error_posix().msg() + } + return buf.substr(0, len) +} diff --git a/src/ls/ls_windows.c.v b/src/ls/ls_windows.c.v new file mode 100644 index 00000000..e64778f8 --- /dev/null +++ b/src/ls/ls_windows.c.v @@ -0,0 +1,51 @@ +fn C.CreateFileW(lpFilename &u16, dwDesiredAccess u32, dwShareMode u32, lpSecurityAttributes &u16, dwCreationDisposition u32, dwFlagsAndAttributes u32, hTemplateFile voidptr) voidptr +fn C.GetFinalPathNameByHandleW(hFile voidptr, lpFilePath &u16, nSize u32, dwFlags u32) u32 + +const max_path_buffer_size = u32(512) + +fn read_link(path string) string { + // gets handle with GENERIC_READ, FILE_SHARE_READ, 0, OPEN_EXISTING, FILE_ATTRIBUTE_NORMAL, 0 + file := C.CreateFile(path.to_wide(), 0x80000000, 1, 0, 3, 0x80, 0) + if file != voidptr(-1) { + defer { + C.CloseHandle(file) + } + final_path := [max_path_buffer_size]u8{} + // https://docs.microsoft.com/en-us/windows/win32/api/fileapi/nf-fileapi-getfinalpathnamebyhandlew + final_len := C.GetFinalPathNameByHandleW(file, unsafe { &u16(&final_path[0]) }, + max_path_buffer_size, 0) + if final_len == 0 { + return '?' + } + if final_len < max_path_buffer_size { + sret := unsafe { string_from_wide2(&u16(&final_path[0]), int(final_len)) } + defer { + unsafe { sret.free() } + } + // remove '\\?\' from beginning (see link above) + assert sret[0..4] == r'\\?\' + sret_slice := sret[4..] + res := sret_slice.clone() + return res + } else { + return '?' + } + } else { + return '?' + } +} + +fn get_owner_name(uid usize) string { + return uid.str() +} + +fn get_group_name(uid usize) string { + return uid.str() +} + +// Windows has no st_blocks, so there is no allocated size to report and no +// `total` line: summing lengths would print a number that means nothing. +fn allocated_size(path string) u64 { + _ := path + return 0 +} diff --git a/src/ls/natural_compare.v b/src/ls/natural_compare.v new file mode 100644 index 00000000..d2dc1a85 --- /dev/null +++ b/src/ls/natural_compare.v @@ -0,0 +1,54 @@ +import math + +// compares strings with embedded numbers (e.g. log17.txt) +fn natural_compare(a &string, b &string) int { + pa := split(a) + pb := split(b) + max := math.min(pa.len, pb.len) + + for i := 0; i < max; i++ { + if pa[i].is_int() && pb[i].is_int() { + result := pa[i].int() - pb[i].int() + if result != 0 { + return result + } + } else { + result := compare_strings(pa[i], pb[i]) + if result != 0 { + return result + } + } + } + return pa.len - pb.len +} + +enum State { + init + digit + non_digit +} + +fn split(a &string) []string { + mut result := []string{} + mut start := 0 + mut state := State.init + s := a.runes() + + for i := 0; i < s.len; i++ { + if s[i] >= `0` && s[i] <= `9` { + if state == State.non_digit { + result << s[start..i].string() + start = i + } + state = State.digit + } else { + if state == State.digit { + result << s[start..i].string() + start = i + } + state = State.non_digit + } + } + result << s[start..].string() + return result +} diff --git a/src/ls/natural_compare_test.v b/src/ls/natural_compare_test.v new file mode 100644 index 00000000..9823a340 --- /dev/null +++ b/src/ls/natural_compare_test.v @@ -0,0 +1,52 @@ +module main + +fn test_numbers_embdded_in_text() { + a := 'log10.txt' + b := 'log9.txt' + + assert compare_strings(&b, &a) > 0 + assert natural_compare(&b, &a) < 0 + + assert compare_strings(&a, &b) < 0 + assert natural_compare(&a, &b) > 0 + + assert compare_strings(&a, &a) == 0 + assert natural_compare(&a, &a) == 0 + + assert compare_strings(&b, &b) == 0 + assert natural_compare(&b, &b) == 0 +} + +fn test_numbers_two_embdded_in_text() { + a := '0log10.txt' + b := '1log9.txt' + + assert compare_strings(&a, &b) < 0 + assert natural_compare(&a, &b) < 0 + + assert compare_strings(&b, &a) > 0 + assert natural_compare(&b, &a) > 0 + + assert compare_strings(&a, &a) == 0 + assert natural_compare(&a, &a) == 0 + + assert compare_strings(&b, &b) == 0 + assert natural_compare(&b, &b) == 0 +} + +fn test_no_numbers_in_text() { + a := 'abc' + b := 'bca' + + assert compare_strings(&a, &b) < 0 + assert natural_compare(&a, &b) < 0 + + assert compare_strings(&b, &a) > 0 + assert natural_compare(&b, &a) > 0 + + assert compare_strings(&a, &a) == 0 + assert natural_compare(&a, &a) == 0 + + assert compare_strings(&b, &b) == 0 + assert natural_compare(&b, &b) == 0 +} diff --git a/src/ls/options.v b/src/ls/options.v new file mode 100644 index 00000000..d3705d5f --- /dev/null +++ b/src/ls/options.v @@ -0,0 +1,186 @@ +module main + +import common +import flag +import os +import term + +const app_name = 'ls' +const app_version = '0.1' +const current_dir = ['.'] + +const when_always = 'always' +const when_never = 'never' +const when_auto = 'auto' + +@[version: app_version] +@[name: app_name] +struct Options { +mut: + // + // flags + all bool @[long: 'all'; short: 'a'; xdoc: 'do not ignore entries starting with .'] + almost_all bool @[long: 'almost-all'; short: 'A'; xdoc: 'do not list implied . and ..'] // default is to not list, not used + blocked_output bool @[xdoc: 'blank line every 5 rows'] + checksum string @[xdoc: 'show file checksum (md5, sha1, sha224, sha256, sha512, blake2b)'] + list_by_columns bool @[only: 'C'; xdoc: 'list entries by columns'] + classify bool @[only: 'F'; xdoc: 'append a character indicating file type'] + list_directory_itself bool @[only: 'd'; xdoc: 'list directories themselves, not their contents'] + colorize string = when_never @[long: 'color'; xdoc: 'color the output (WHEN); more info below'] + long_no_owner bool @[only: 'g'; xdoc: 'like -l, but do not list owner'] + dirs_first bool @[only: 'group-directories-first'; xdoc: 'group directories before files; can be augmented with a --sort option, but any use of --sort=none (-U) disables grouping'] + icons bool @[xdoc: 'show file icon (requires nerd fonts)'] + no_date bool @[xdoc: 'hide data (modified)'] + no_dim bool @[xdoc: 'hide shading; useful for light backgrounds'] + no_group_name bool @[long: 'no-group'; short: 'G'; xdoc: 'in a long listing, don\'t print group names'] + no_hard_links bool @[xdoc: 'hide hard links count'] + no_owner_name bool @[only: 'no_owner'; xdoc: 'hide owner name'] + no_permissions bool @[xdoc: 'hide permissions'] + no_size bool @[xdoc: 'hide file size'] + no_wrap bool @[xdoc: 'do not wrap long lines'] + size_kb bool @[long: 'human-readable'; short: 'h'; xdoc: 'with -l and -s, print sizes like 1K 234M 2G etc.'] + header bool @[xddoc: 'show column headers (implies -l)'] + size_ki bool @[long: 'si'; xdoc: 'likewise, but use powers of 1000 not 1024'] + inode bool @[long: 'inode'; short: 'i'; xdoc: 'print the index number of each file'] + long_format bool @[only: 'l'; xdoc: 'use a long listing format'] + with_commas bool @[only: 'm'; xdoc: 'fill width with a comma separated list of entries'] + long_no_group bool @[only: 'o'; xdoc: 'like -l, but do not list group information'] + octal_permissions bool @[xdoc: 'show as permissions octal number'] + only_dirs bool @[xdoc: 'list only directories'] + only_files bool @[xdoc: 'list only files'] + dir_indicator bool @[long: 'indicator-style'; short: 'p'; xdoc: 'append / indicator to directories'] + quote bool @[long: 'quote-name'; short: 'Q'; xdoc: 'enclose entry names in double quotes'] + quoting_style string @[only: 'quoting-style'; xdoc: 'quote names