diff --git a/src/sort/compare.v b/src/sort/compare.v new file mode 100644 index 00000000..2b3bfb08 --- /dev/null +++ b/src/sort/compare.v @@ -0,0 +1,225 @@ +// The ordering options, which decide how one key is compared with another. Any of +// them may be given on the command line, and a key definition may override the +// ones set globally. +enum OrderMode { + ascii + numeric + general + human + month + version +} + +struct Ordering { +mut: + mode OrderMode + ignore_blanks bool + dictionary bool + fold_case bool + ignore_nonprint bool + reverse bool + // stable turns off the last-resort comparison, so that lines which compare + // equal keep the order they came in with. + stable bool + // random is -R, which shuffles but keeps lines with equal keys together. + random bool +} + +// compare_key compares a and b under one ordering, including the comparison of +// the part that follows a number, so that a non-zero result means the two keys +// are ordered and a zero result means the next key, or the last-resort +// comparison, has to decide. +fn compare_key(ordering &Ordering, a &string, b &string) int { + x := order_key(ordering, a) + y := order_key(ordering, b) + + match ordering.mode { + .general, .numeric, .human { + // Only the number weighs here. GNU does not look at the text that + // follows it, which is why "1a" and "01b" sort as their whole lines + // do rather than by the a and b. + rn := compare_values(x.value, y.value) + if rn != 0 { + return flip(ordering.reverse, rn) + } + } + .ascii { + rt := compare_bytes(x.text, y.text) + if rt != 0 { + return flip(ordering.reverse, rt) + } + } + .month { + rm := compare_values(f64(x.month), f64(y.month)) + if rm != 0 { + return flip(ordering.reverse, rm) + } + // A month weighs with the text after it, so "JAN" comes before "JANUARY". + rs2 := compare_bytes(x.rest, y.rest) + if rs2 != 0 { + return flip(ordering.reverse, rs2) + } + } + .version { + rv := compare_version(x.text, y.text) + if rv != 0 { + return flip(ordering.reverse, rv) + } + // A version sort compares the whole key, digits and all, so there is + // nothing left over for the last resort but the lines themselves. + return 0 + } + else { + rt := compare_bytes(x.text, y.text) + if rt != 0 { + return flip(ordering.reverse, rt) + } + } + } + return 0 +} + +// order_key is one key as the ordering wants to see it: the characters this +// ordering weighs, and the value of the number or month that may start it. +struct OrderedKey { + text string + value f64 + month int + rest string +} + +// order_key applies the ordering options to s. The transformations are the ones +// GNU applies, in this order: leading blanks, then the characters -d and -i skip, +// then the case folding. A numeric ordering reads what is left. +fn order_key(ordering &Ordering, s string) OrderedKey { + mut text := s + if ordering.ignore_blanks { + text = text.trim_left(' \t\n\v\f\r') + } + if ordering.dictionary { + mut out := []u8{} + for c in text { + if c.is_digit() || (c >= `a` && c <= `z`) || (c >= `A` && c <= `Z`) || is_blank(c) { + out << c + } + } + text = out.bytestr() + } + if ordering.ignore_nonprint { + mut out := []u8{} + for c in text { + if c >= ` ` && c <= `~` { + out << c + } + } + text = out.bytestr() + } + if ordering.fold_case { + text = fold_case(text) + } + + match ordering.mode { + .numeric { + n := number_prefix(text, .string) + return OrderedKey{ text: text, value: n.value, rest: n.rest } + } + .general { + n := number_prefix(text, .general) + return OrderedKey{ text: text, value: n.value, rest: n.rest } + } + .human { + n := number_prefix(text, .human) + return OrderedKey{ text: text, value: n.value, rest: n.rest } + } + .month { + month, rest := month_prefix(text) + return OrderedKey{ text: text, month: month, rest: rest } + } + else { + return OrderedKey{ text: text } + } + } +} + +// compare_values orders two f64, putting a NaN before everything and an infinity +// at either end, which is the order GNU documents for -g. +fn compare_values(a f64, b f64) int { + // A NaN is the only value that is not equal to itself. + a_nan := a != a + b_nan := b != b + if a_nan { + return if b_nan { 0 } else { -1 } + } + if b_nan { + return 1 + } + if a < b { + return -1 + } + if a > b { + return 1 + } + return 0 +} + +// compare_bytes is a plain byte comparison, the order LC_ALL=C gives. +fn compare_bytes(a string, b string) int { + mut i := 0 + for i < a.len && i < b.len { + if a[i] != b[i] { + return if a[i] < b[i] { -1 } else { 1 } + } + i++ + } + if i >= a.len && i >= b.len { + return 0 + } + return if i >= a.len { -1 } else { 1 } +} + +fn flip(reverse bool, result int) int { + return if reverse { -result } else { result } +} + +// fold_case maps the ASCII letters to one case, so that a comparison under -f +// weighs them the same whichever case they came in. +fn fold_case(s string) string { + mut out := []u8{len: s.len, cap: s.len} + for i, c in s { + if c >= `A` && c <= `Z` { + out[i] = c + 32 + } else if c >= 0xc0 && c <= 0xde && c != 0xd7 { + out[i] = c + 32 + } else { + out[i] = c + } + } + return out.bytestr() +} + +const month_names = ['jan', 'feb', 'mar', 'apr', 'may', 'jun', 'jul', 'aug', 'sep', 'oct', 'nov', + 'dec'] + +// month_prefix reads a month name at the start of s, ignoring case, and returns +// its number and the rest. A line that does not start with a month gets -1, so +// that it comes before every month, as at GNU. +fn month_prefix(s string) (int, string) { + if s.len < 3 { + return -1, s + } + mut name := []u8{cap: 3} + for i in 0 .. 3 { + c := s[i] | 0x20 + if c < `a` || c > `z` { + return -1, s + } + name << c + } + folded := name.bytestr() + for m, candidate in month_names { + if folded == candidate { + return m, s[3..] + } + } + return -1, s +} diff --git a/src/sort/driver.v b/src/sort/driver.v new file mode 100644 index 00000000..3ea208b0 --- /dev/null +++ b/src/sort/driver.v @@ -0,0 +1,147 @@ +// The comparison the whole program is built on. It applies each key definition in +// turn, and when they all call the lines equal it falls back on comparing the lines +// themselves, which is the last-resort comparison that -s turns off. +struct Comparator { + options Options + keys []SortKey +} + +// compare returns how a should be ordered against b. +fn (c &Comparator) compare(a &InputLine, b &InputLine) int { + mut result := compare_keys(c.options, c.keys, a, b) + if result != 0 { + return result + } + if !c.options.ordering.stable { + if c.options.ordering.reverse { + result = compare_bytes(b.text, a.text) + } else { + result = compare_bytes(a.text, b.text) + } + if result != 0 { + return result + } + } + // Lines that weigh the same keep the order they arrived in, which is what -s + // asks for and what -u relies on to decide which line of a run it keeps. + return compare_values(f64(a.order), f64(b.order)) +} + +fn make_comparator(options Options, keys []SortKey) &Comparator { + return &Comparator{ + options: options + keys: keys + } +} + +// compare_keys compares two lines by their keys alone, without the last-resort +// step. A zero result means the keys are equal. +// +// With no -k at all the whole line is the key, and the ordering options given on +// the command line are what weigh it. +fn compare_keys(options Options, keys []SortKey, a &InputLine, b &InputLine) int { + if keys.len == 0 { + return compare_key(&options.ordering, &a.text, &b.text) + } + for key in keys { + ak := extract_key(a.text, &key, options.separator, options.has_separator, + options.ordering.ignore_blanks) + bk := extract_key(b.text, &key, options.separator, options.has_separator, + options.ordering.ignore_blanks) + result := compare_key(&key.ordering, &ak, &bk) + if result != 0 { + return result + } + } + return 0 +} + +// sort_lines orders the whole input. Without -s the comparison is total, because +// of the last-resort step, so a plain sort is enough. +fn sort_lines(lines []InputLine, options Options, keys []SortKey) []InputLine { + cmp := make_comparator(sort_options(options), keys) + mut out := lines.clone() + out.sort_with_compare(fn [cmp] (a &InputLine, b &InputLine) int { + return cmp.compare(a, b) + }) + return out +} + +// unique_lines keeps the first line of every run that compares equal, which is +// what -u means: the lines may differ, the keys may not. +fn unique_lines(lines []InputLine, options Options, keys []SortKey) []InputLine { + if !options.unique { + return lines + } + mut out := []InputLine{} + for line in lines { + if out.len > 0 { + previous := out[out.len - 1] + // Only the keys count here. -s is deliberately not consulted: a run of + // equal keys is a run of equal keys however the lines were ordered. + if compare_keys(options, keys, &previous, &line) == 0 { + continue + } + } + out << line + } + return out +} + +// check_input reports the first line that is out of order, the way GNU's -c does. +fn check_input(lines []InputLine, options Options, keys []SortKey) { + cmp := make_comparator(options, keys) + mut i := 1 + for i < lines.len { + previous := lines[i - 1] + current := lines[i] + result := cmp.compare(&previous, ¤t) + // The comparison breaks a tie by the order the lines arrived in, so a + // non-zero result here means the later line weighs less than the earlier + // one, which is a disorder. With -u a line whose key equals the one before + // it is a disorder as well. + equal_keys := compare_keys(options, keys, &previous, ¤t) == 0 + disordered := result > 0 || (options.unique && equal_keys) + if disordered { + if options.check == .diagnose { + println('${app_name}: ${current.file}:${current.number}: disorder: ${current.text}') + } + exit(1) + } + i++ + } +} + +// merge_input combines already sorted inputs the way -m does. The promise of -m is +// that each input is sorted, so the lines are merged without being sorted again, +// and a line that is out of place in its own file stays where it was. +fn merge_input(streams [][]InputLine, options Options, keys []SortKey) []InputLine { + cmp := make_comparator(options, keys) + mut out := []InputLine{} + mut pos := []int{len: streams.len, init: 0} + + for { + // Find the stream whose next line weighs least. When the lines weigh the + // same the earlier stream wins, which is the order GNU merges in. + mut best := -1 + for i, stream in streams { + if pos[i] >= stream.len { + continue + } + if best < 0 { + best = i + continue + } + chosen := cmp.compare(stream[pos[i]], streams[best][pos[best]]) + if chosen < 0 { + best = i + } + } + if best < 0 { + break + } + out << streams[best][pos[best]] + pos[best]++ + } + return out +} diff --git a/src/sort/input.v b/src/sort/input.v new file mode 100644 index 00000000..ac48dc0d --- /dev/null +++ b/src/sort/input.v @@ -0,0 +1,120 @@ +import os + +// input_files is the list of inputs to read, which --files0-from can supply. +fn input_files(options Options) []string { + mut files := options.files.clone() + if options.files0_from.len > 0 { + files = read_files0(options) + } + if files.len == 0 { + files = ['-'] + } + return files +} + +// read_all_input gathers the lines from every input into one stream, which is what +// GNU sorts as a whole unless -m is given. +fn read_all_input(options Options) []InputLine { + mut lines := []InputLine{} + for file in input_files(options) { + lines << read_one(file, options) + } + return lines +} + +// read_each_input keeps the inputs apart, which -m needs in order to merge them +// rather than sort them. +fn read_each_input(options Options) [][]InputLine { + mut all := [][]InputLine{} + for file in input_files(options) { + all << read_one(file, options) + } + return all +} + +fn read_one(file string, options Options) []InputLine { + mut f := if file == '-' { + os.stdin() + } else { + os.open(file) or { read_error(file) } + } + return read_stream(f, file, options) +} + +// read_files0 reads the list of names that --files0-from points at. The names are +// separated by NUL, so a name may contain a newline. +fn read_files0(options Options) []string { + mut list := if options.files0_from == '-' { + os.stdin() + } else { + os.open(options.files0_from) or { read_error(options.files0_from) } + } + content := read_all(list) + mut names := []string{} + mut start := 0 + mut i := 0 + for i < content.len { + if content[i] == 0 { + names << content[start..i] + start = i + 1 + } + i++ + } + if start < content.len { + names << content[start..] + } + return names +} + +// read_stream splits a file into lines. A line keeps the text without its +// delimiter, and the position GNU reports it at. +fn read_stream(f os.File, name string, options Options) []InputLine { + content := read_all(f) + return split_records(content, name, line_delimiter(options)) +} + +// read_all reads a file to its end. +fn read_all(f os.File) string { + mut out := []u8{} + mut buf := []u8{len: 64 * 1024} + for { + n := f.read(mut buf) or { break } + if n == 0 { + break + } + out << buf[..n] + } + return out.bytestr() +} + +// split_records cuts content at every delimiter byte. +fn split_records(content string, name string, delim string) []InputLine { + mut lines := []InputLine{} + mut start := 0 + mut i := 0 + mut number := 1 + for i < content.len { + if content[i..i + delim.len] == delim { + lines << InputLine{ + text: content[start..i] + file: name + number: number + order: lines.len + } + number++ + start = i + 1 + } + i++ + } + // A file that ends with a delimiter has no extra empty line, but a file that + // does not ends with one line that has no delimiter. + if start < content.len { + lines << InputLine{ + text: content[start..] + file: name + number: number + order: lines.len + } + } + return lines +} diff --git a/src/sort/key.v b/src/sort/key.v new file mode 100644 index 00000000..add3d311 --- /dev/null +++ b/src/sort/key.v @@ -0,0 +1,317 @@ +// A key definition, as -k takes it: F[.C][OPTS][,F[.C][OPTS]], where F is a field +// number and C a character position in that field, both counting from one. +struct SortKey { +mut: + from_field int + from_char int + to_field int + to_char int + has_to bool + ordering Ordering + has_ordering bool +} + +// parse_keys turns the -k arguments into key definitions. The orderings a key +// names override the global ones; it inherits each global option that it does not +// mention. +fn parse_keys(specs []string, base Ordering) []SortKey { + mut keys := []SortKey{} + for spec in specs { + keys << parse_key(spec, base) + } + return keys +} + +fn parse_key(spec string, base Ordering) SortKey { + mut key := SortKey{ + from_field: 0 + from_char: 0 + to_field: 0 + to_char: 0 + ordering: base + } + + // The definition is read as: a number, an optional .number, any ordering + // letters, then an optional ,number with an optional .number. The ordering + // letters may sit anywhere among the numbers, which is what makes -k2n,2 and + // -k2,2n mean the same thing. + mut from_field := 0 + mut from_char := 0 + mut to_field := 0 + mut to_char := 0 + mut has_to := false + mut ordering := base + mut saw_field := false + mut i := 0 + // which number comes next: 0 for the start, 1 for the end + mut next := 0 + + for i < spec.len { + c := spec[i] + if c.is_digit() { + // A field number is the digits, and nothing may come between them and + // the dot or comma that may follow. An ordering letter may sit in + // between, though: -k2n,2 and -k2,2n mean the same thing. + n, j := read_field_number(spec, i) + if next == 0 { + from_field = n + from_char = 0 + saw_field = true + } else { + to_field = n + to_char = 0 + has_to = true + } + i = j + continue + } + if c == `.` { + // A position after the dot, so a digit run must follow. + n, j := read_field_number(spec, i + 1) + if j == i + 1 { + key_error(spec) + } + if next == 0 { + from_char = n + } else { + to_char = n + } + i = j + continue + } + if c == `,` { + if !saw_field { + key_error(spec) + } + next = 1 + i++ + continue + } + // An ordering letter may only follow the field number, which is why -k2n + // reads but -kn does not. + if !saw_field { + error_only("invalid number at field start: invalid count at start of '${spec}'") + } + ordering = apply_ordering_letter(mut ordering, c, spec) + i++ + } + // A definition that does not start with a field number is not a key at all, + // which is the mistake GNU calls an invalid count at the start of it. + // A definition that does not start with a field number is not a key at all. + if !saw_field { + error_only("invalid number at field start: invalid count at start of '${spec}'") + } + // A field number of zero is not a position. GNU rejects the whole definition + // rather than reading it as the end of the line. + if from_field == 0 || (has_to && to_field == 0) { + error_only("field number is zero: invalid field specification '${spec}'") + } + + key.from_field = from_field + key.from_char = from_char + key.to_field = to_field + key.to_char = to_char + key.has_to = has_to + key.ordering = ordering + return key +} + +// read_field_number reads the digits at i, and returns the number and where it +// stopped. V does not allow a mutable scalar parameter, hence the pair. +fn read_field_number(spec string, i int) (int, int) { + mut j := i + for j < spec.len && spec[j].is_digit() { + j++ + } + mut n := 0 + for c in spec[i..j] { + n = n * 10 + int(c - `0`) + } + return n, j +} + +// ordering_letters are the ones a key definition may name after its positions. +const ordering_letters = 'bdfghiMhnRrV' + +// apply_ordering_letter applies one ordering letter to a key's ordering. The +// Ordering fields are declared mut, so a local copy is enough. +fn apply_ordering_letter(mut ordering Ordering, c u8, spec string) Ordering { + match c { + `b` { ordering.ignore_blanks = true } + `d` { ordering.dictionary = true } + `f` { ordering.fold_case = true } + `i` { ordering.ignore_nonprint = true } + `n` { ordering.mode = .numeric } + `g` { ordering.mode = .general } + `h` { ordering.mode = .human } + `M` { ordering.mode = .month } + `V` { ordering.mode = .version } + `r` { ordering.reverse = true } + `R` { ordering.random = true } + else { key_error(spec) } + } + return ordering +} + +// key_error reports a malformed key definition. GNU does not follow this with the +// advice about --help. +@[noreturn] +fn key_error(spec string) { + error_only('invalid key ${spec}') +} + +// extract_key returns the part of line that a key definition covers. +fn extract_key(line string, key &SortKey, separator string, has_separator bool, + ignore_blanks bool) string { + mut start := field_start(line, key.from_field, separator, has_separator, ignore_blanks) + // A position of zero means the start of the field, so that is what a missing + // .C has to mean. + if key.from_char > 1 { + start += key.from_char - 1 + } + if start > line.len { + start = line.len + } + + // Without an end position the key runs to the end of the line, which is what + // -k2 on its own does. + mut end := line.len + if key.has_to && key.to_field > 0 { + if key.to_char > 0 { + // A character position is a place in the field, counted from its + // start, and it is the last character of the key rather than a count + // from it. A line shorter than that simply stops early. + end = field_start(line, key.to_field, separator, has_separator, + ignore_blanks) + key.to_char + } else { + end = field_end(line, key.to_field, separator, has_separator, ignore_blanks) + } + } + if end > line.len { + end = line.len + } + if end < start { + end = start + } + return line[start..end] +} + +// field_start returns where a field begins. With -t it is the start of the field +// itself; without -t a field runs from the blank that introduced it, unless -b says +// to ignore those blanks. +fn field_start(line string, field int, separator string, has_separator bool, + ignore_blanks bool) int { + mut pos := 0 + mut f := 1 + mut n := line.len + // Whether a field with characters in it has been stepped over. Without it, a + // line of nothing but blanks would have an empty first field and a second field + // covering the whole line, where GNU gives an empty second field. + mut consumed := false + + for f < field { + if pos >= n { + return n + } + if has_separator { + next := find_from(line, pos, separator) + if next < 0 { + return n + } + pos = next + separator.len + consumed = true + } else { + // A field is a run of non-blanks, so the blanks before it belong to + // it only when -b is not in effect. Stepping over them first is what + // makes a line that begins with blanks have no leading field. + for pos < n && is_blank(line[pos]) { + pos++ + } + if pos < n { + consumed = true + } + for pos < n && !is_blank(line[pos]) { + pos++ + } + } + f++ + } + if pos > n { + return n + } + + if has_separator { + return pos + } + if ignore_blanks { + // The loop above stops on the blank that introduces this field, so -b has + // to step past it. + for pos < n && is_blank(line[pos]) { + pos++ + } + return pos + } + if field == 1 || !consumed { + // The first field has no whitespace before it to count from, and neither + // has a field that does not exist. + return pos + } + // Any other field starts at the blank that introduced it. + mut back := pos + for back > 0 && is_blank(line[back - 1]) { + back-- + } + return back +} + +// field_end returns where a field ends, which is after its last character. +fn field_end(line string, field int, separator string, has_separator bool, + ignore_blanks bool) int { + mut n := line.len + start := field_start(line, field, separator, has_separator, ignore_blanks) + if start >= n { + return n + } + + if has_separator { + next := find_from(line, start, separator) + return if next < 0 { n } else { next } + } + + // Without -t a field runs from the blank before it to the blank after it, so + // the blanks at its start have to be stepped over before its end can be found. + mut pos := start + for pos < n && is_blank(line[pos]) { + pos++ + } + for pos < n && !is_blank(line[pos]) { + pos++ + } + return pos +} + +// find_next_non_blank returns the position of the next character that is not a +// blank, or -1 when there is none. +fn find_next_non_blank(line string, from int) int { + mut pos := from + for pos < line.len { + if !is_blank(line[pos]) { + return pos + } + pos++ + } + return -1 +} + +// find_from returns where sep first appears at or after pos, or -1 when it does +// not. V's string module has no search that starts at a position. +fn find_from(line string, pos int, sep string) int { + mut i := pos + for i + sep.len <= line.len { + if line[i..i + sep.len] == sep { + return i + } + i++ + } + return -1 +} diff --git a/src/sort/number.v b/src/sort/number.v new file mode 100644 index 00000000..f383a520 --- /dev/null +++ b/src/sort/number.v @@ -0,0 +1,264 @@ +import math +import strconv + +// A numeric prefix, as the -n, -g and -h orderings use it: the value found at +// the start of the key and the rest of the key from there. +struct NumberPrefix { + value f64 + rest string +} + +// Numeric prefixes follow three different grammars, so that is how they are told +// apart. -g reads what strtod does, -h reads that plus a size suffix, and -n is +// the narrowest of the three. +enum NumberStyle { + string + general + human +} + +// number_prefix reads the numeric prefix of s in the given style. +fn number_prefix(s string, style NumberStyle) NumberPrefix { + mut i := 0 + // A leading minus is read everywhere, a leading plus only by -g. GNU's probe + // confirms it: with -n and with -h, "+5" weighs nothing and sorts as zero, + // which puts it before "10" but after the letters. + allow_plus := style == .general + mut allow_sign := true + mut allow_blanks := true + mut negative := false + + for i < s.len { + c := s[i] + if allow_blanks && is_blank(c) { + i++ + continue + } + if allow_sign && (c == `-` || (allow_plus && c == `+`)) { + negative = c == `-` + allow_sign = false + allow_blanks = false + i++ + continue + } + break + } + + body_start := i + // The words inf and nan are numbers to -g even though no digit starts them, so + // they have to be recognised before the scan for digits, which would find + // nothing and report the line as having no number at all. + if style == .general { + word := s[body_start..].to_lower() + if word == 'inf' { + return NumberPrefix{ + value: math.inf(1) + rest: s[body_start + 3..] + } + } else if word == 'nan' { + // A NaN weighs as the largest negative finite number, which puts it + // after the lines that have no number at all and before every number + // there is. That is where GNU puts it, even though its manual lists + // NaNs first. + return NumberPrefix{ + value: -math.max_f64 + rest: s[body_start + 3..] + } + } + } + + body_end := scan_number_body(s, i, style) + if body_end == body_start { + // Nothing numeric at all. -g calls that minus infinity so that such lines + // come first; -n and -h call it zero. + return NumberPrefix{ + value: if style == .general { math.inf(-1) } else { 0.0 } + rest: s + } + } + + // The sign was stepped over above, so it is put back for the parser to read, + // which is what keeps -5 negative rather than five. + mut body := s[body_start..body_end] + if negative { + body = '-' + body + } + mut value := if style == .general { + parse_general(body) + } else { + parse_plain(body) + } + + // -h lets a size suffix follow the digits. + mut end := body_end + if style == .human { + scale, suffix_len := size_suffix(s, end) + if scale != 1.0 { + value *= scale + end += suffix_len + } + } + + return NumberPrefix{ + value: value + rest: s[end..] + } +} + +// scan_number_body returns where the numeric text ends, starting at i. It never +// runs past a character that is not part of a number. +fn scan_number_body(s string, i int, style NumberStyle) int { + mut j := i + mut seen_digit := false + + if style == .general && (s[j..].starts_with('0x') || s[j..].starts_with('0X')) { + j += 2 + for j < s.len && is_hex_digit(s[j]) { + j++ + seen_digit = true + } + return if seen_digit { j } else { i } + } + + for j < s.len && s[j].is_digit() { + j++ + seen_digit = true + } + // -n accepts a decimal point but no exponent, and -h the same. + if j < s.len && s[j] == `.` && (style != .general || true) { + mut k := j + 1 + mut frac := false + for k < s.len && s[k].is_digit() { + k++ + frac = true + } + if frac || seen_digit { + j = k + seen_digit = seen_digit || frac + } + } + if !seen_digit { + return i + } + // An exponent is read by -g in either case, and by -h only when it is written + // with a capital E. GNU's own behaviour: -h weighs 1e3 as one, because the + // lower case e stops the number, but weighs 1E3 as a thousand. -n reads no + // exponent at all. + if j < s.len && ((style == .general && (s[j] == `e` || s[j] == `E`)) + || (style == .human && s[j] == `E`)) { + mut k := j + 1 + if k < s.len && (s[k] == `-` || s[k] == `+`) { + k++ + } + mut exp_digits := false + for k < s.len && s[k].is_digit() { + k++ + exp_digits = true + } + if exp_digits { + j = k + } + } + return j +} + +// size_suffix reads the multiplier that -h allows after the digits. The units up +// to T are accepted in either case, but P, E, Z and Y only in capitals: GNU weighs +// "1p" as one and "1P" as a petabyte, and the same for E, Z and Y. A lower case e +// is not a unit at all, which is why -h reads "1e3" as one. +// +// One combination is not modelled. GNU weighs a suffixed number against a plain +// one of seven digits or more by something this does not reproduce: it puts +// "1000000" before "1K", although every comparison of a suffix against a smaller +// plain number agrees, including 1K against 1024 and 2K against 1M. No value for +// the K in "1K" is consistent with both, since it has to be above 1e7 and at most +// 1024^4, which is what makes this look like a fault on GNU's side rather than a +// rule. +fn size_suffix(s string, i int) (f64, int) { + if i >= s.len { + return 1.0, 0 + } + scale := match s[i] { + `k`, `K` { f64(1024) } + `m`, `M` { f64(1024) * 1024 } + `g`, `G` { f64(1024) * 1024 * 1024 } + `t`, `T` { f64(1024) * 1024 * 1024 * 1024 } + `P` { f64(1024) * 1024 * 1024 * 1024 * 1024 } + `E` { f64(1024) * 1024 * 1024 * 1024 * 1024 * 1024 } + `Z` { f64(1024) * 1024 * 1024 * 1024 * 1024 * 1024 * 1024 } + `Y` { f64(1024) * 1024 * 1024 * 1024 * 1024 * 1024 * 1024 * 1024 } + else { return 1.0, 0 } + } + mut used := 1 + // GNU also takes the optional i and B of 2KiB and 2KB, and so does this. + if i + 1 < s.len && (s[i + 1] == `i` || s[i + 1] == `I`) { + used++ + if i + 2 < s.len && (s[i + 2] == `B` || s[i + 2] == `b`) { + used++ + } + } else if i + 1 < s.len && (s[i + 1] == `B` || s[i + 1] == `b`) { + used++ + } + return scale, used +} + +fn parse_plain(body string) f64 { + return strconv.atof64(body) or { 0.0 } +} + +// parse_general reads what -g accepts: an ordinary float, a hexadecimal float, or +// one of the words inf and nan, which C's strtod understands and V does not. +fn parse_general(body string) f64 { + if body.len > 2 && (body.starts_with('inf') || body.starts_with('INF')) { + return math.inf(1) + } + if body.len > 2 && (body.starts_with('nan') || body.starts_with('NAN')) { + // A NaN weighs as the largest negative finite number, which puts it after + // the lines that have no number at all and before every number there is. + // That is where GNU puts it, even though its manual lists NaNs first. + return -math.max_f64 + } + if body.len > 2 && (body.starts_with('0x') || body.starts_with('0X')) { + digits := body[2..] + // A hexadecimal float may carry a fractional part after a point, which + // strtof64 does not read, so it is scaled here. + if point := digits.index('.') { + int_part := digits[..point] + frac_part := digits[point + 1..] + mut value := f64(0.0) + mut scale := 1.0 + for k in 0 .. int_part.len { + value = value * 16.0 + f64(hex_value(int_part[k])) + } + for k in 0 .. frac_part.len { + scale /= 16.0 + value += f64(hex_value(frac_part[k])) * scale + } + return value + } + // Plain hexadecimal without a fractional part. + mut hex := u64(0) + for c in digits { + hex = hex * 16 + u64(hex_value(c)) + } + return f64(hex) + } + return strconv.atof64(body) or { 0.0 } +} + +fn hex_value(c u8) int { + return match c { + `0`...`9` { int(c - `0`) } + `a`...`f` { int(c - `a`) + 10 } + `A`...`F` { int(c - `A`) + 10 } + else { 0 } + } +} + +fn is_hex_digit(c u8) bool { + return c.is_digit() || (c >= `a` && c <= `f`) || (c >= `A` && c <= `F`) +} + +fn is_blank(c u8) bool { + return c == ` ` || c == `\t` || c == `\n` || c == `\v` || c == `\f` || c == `\r` +} diff --git a/src/sort/options.v b/src/sort/options.v index 52317bf9..cdc981ce 100644 --- a/src/sort/options.v +++ b/src/sort/options.v @@ -1,131 +1,369 @@ import common -import flag import os -import time const app_name = 'sort' +// CheckMode is what -c and -C ask for. +enum CheckMode { + none + diagnose + quiet +} + struct Options { - ignore_leading_blanks bool - dictionary_order bool - ignore_case bool - ignore_non_printing bool - numeric bool - reverse bool - // other optoins - check_diagnose bool - check_quiet bool - sort_keys []string - field_separator string = ' ' - merge bool - output_file string +mut: + ordering Ordering + key_specs []string + check CheckMode unique bool + merge bool + output string + separator string + has_separator bool + zero_terminated bool + debug bool files []string + files0_from string } -fn get_options() Options { - mut fp := flag.new_flag_parser(os.args) - fp.application(app_name) - fp.version(common.coreutils_version()) - fp.skip_executable() - fp.arguments_description('[FILE]') - fp.description('\nWrite sorted concatenation of all FILE(s) to standard output.' + - '\nWith no FILE, or when FILE is -, read standard input.') - - ignore_leading_blanks := fp.bool('ignore-leading-blanks', `b`, false, 'ignore leading blanks') - dictionary_order := fp.bool('dictionary-order', `d`, false, - 'consider only blanks and alphanumeric characters') - ignore_case := fp.bool('ignore-case', `f`, false, 'fold lower case to upper case characters') - ignore_non_printing := fp.bool('ignore-non-printing', `i`, false, - 'consider only printable characters') - numeric := fp.bool('numeric-sort', `n`, false, - 'Restrict the sort key to an initial numeric\n${flag.space}' + - 'string, consisting of optional characters,\n${flag.space}' + - 'optional character, and zero or\n${flag.space}' + - 'more digits, which shall be sorted by arithmetic\n${flag.space}' + - 'value. An empty digit string shall be treated as\n${flag.space}' + - 'zero. Leading zeros shall not affect ordering.') - reverse := fp.bool('reverse', `r`, false, 'reverse the result of comparisons\n\nOther options:') - - check_diagnose := fp.bool('', `c`, false, 'check for sorted input; do not sort') - check_quiet := fp.bool('', `C`, false, 'like -c, but do not report first bad line') - sort_keys := fp.string_multi('key', `k`, 'sort via a key(s); gives location and type') - merge := fp.bool('merge', `m`, false, 'merge already sorted files; do not sort') - field_separator := fp.string('', `t`, ' ', 'use as field separator') - output_file := fp.string('output', `o`, '', 'write result to FILE instead of standard output') - unique := fp.bool('unique', `u`, false, 'with -c, check for strict ordering;\n${flag.space}' + - 'without -c, output only the first of an equal run') - - fp.footer(" - - KEYDEF is F[.C][OPTS][,F[.C][OPTS]] for start and stop position, - where F is a field number and C a character position in the - field; both are origin 1, and the stop position defaults to the - line's end. If neither -t nor -b is in effect, characters in a - field are counted from the beginning of the preceding whitespace. - OPTS is one or more single-letter ordering options [bdfir], which - override global ordering options for that key. If no key is - given, use the entire line as the key.".trim_indent()) - - fp.footer(common.coreutils_footer()) - files := fp.finalize() or { exit_error(err.msg()) } - - return Options{ - ignore_leading_blanks: ignore_leading_blanks - dictionary_order: dictionary_order - ignore_case: ignore_case - ignore_non_printing: ignore_non_printing - numeric: numeric - reverse: reverse - // other options - check_diagnose: check_diagnose - check_quiet: check_quiet - sort_keys: sort_keys - field_separator: field_separator - merge: merge - output_file: output_file - unique: unique - files: scan_files_arg(files) - } -} - -fn scan_files_arg(files_arg []string) []string { - mut files := []string{} - for file in files_arg { - if file == '-' { - files << stdin_to_tmp() +// long options that take a value +const long_value_options = ['key', 'field-separator', 'output', 'sort', 'files0-from', 'buffer-size', + 'temporary-directory', 'parallel', 'batch-size', 'compress-program', 'random-source'] + +// short options that take a value +const short_value_options = 'kto' + +fn get_options() (Options, []SortKey) { + mut o := Options{} + mut args := []string{} + for a in os.args[1..] { + args << a + } + mut i := 0 + mut no_more_options := false + + for i < args.len { + arg := args[i] + i++ + if no_more_options || arg == '-' || !arg.starts_with('-') { + o.files << arg continue } - files << file + if arg == '--' { + no_more_options = true + continue + } + if arg.starts_with('--') { + i = parse_long_option(mut o, arg[2..], args, i) + continue + } + i = parse_short_options(mut o, arg[1..], args, i) + } + + // GNU's default separator is the transition from non-blank to blank. + if !o.has_separator { + o.separator = ' ' + } + return o, parse_keys(o.key_specs, o.ordering) +} + +// sort_options is the ordering the sort itself runs under. -u keeps the first line +// of a run of equal keys, and which line that is comes from the order the lines +// arrived in, so -u orders stably even though -s was not given. The check modes +// do not take this, because there -u means that an equal line is a disorder. +fn sort_options(options Options) Options { + if options.unique && !options.ordering.stable { + mut out := options + out.ordering.stable = true + return out + } + return options +} + +// take_value returns the value of an option, which may be attached to it, along +// with the index of the next argument. A short option with no value gets the +// message GNU uses, a long one its own. +fn take_value(name string, long_form bool, rest string, args []string, i int) (string, int) { + if rest.len > 0 { + return rest, i } - if files.len == 0 { - files << stdin_to_tmp() + if i < args.len { + return args[i], i + 1 } - return files + if long_form { + option_error("option '${app_name} --${name}' requires an argument") + } else { + option_error("option requires an argument -- '${name}'") + } + return '', i } -const tmp_pattern = '/${app_name}-tmp-' +fn parse_long_option(mut o Options, arg string, args []string, start_i int) int { + mut i := start_i + mut name := arg + mut value := '' + mut has_value := false + if eq := arg.index('=') { + name = arg[..eq] + value = arg[eq + 1..] + has_value = true + } + + needs_value := name in long_value_options + if needs_value && !has_value { + mut taken := '' + taken, i = take_value(name, true, '', args, i) + value = taken + } -fn stdin_to_tmp() string { - tmp := '${os.temp_dir()}/${tmp_pattern}${time.ticks()}' - os.create(tmp) or { exit_error(err.msg()) } - mut f := os.open_append(tmp) or { exit_error(err.msg()) } - defer { f.close() } - for { - s := os.get_raw_line() - if s.len == 0 { - break + match name { + 'help' { + print_help() + exit(0) + } + 'version' { + // The test rig checks for this exact shape, so the version goes through + // the shared helper rather than being spelled out here. + println('${app_name} ${common.coreutils_version()}') + exit(0) + } + 'ignore-leading-blanks' { o.ordering.ignore_blanks = true } + 'dictionary-order' { o.ordering.dictionary = true } + 'ignore-case' { o.ordering.fold_case = true } + 'ignore-nonprinting', 'ignore-non-printing' { o.ordering.ignore_nonprint = true } + 'reverse' { o.ordering.reverse = true } + 'stable' { o.ordering.stable = true } + 'merge' { o.merge = true } + 'unique' { o.unique = true } + 'debug' { o.debug = true } + 'zero-terminated' { o.zero_terminated = true } + 'numeric-sort' { o.ordering.mode = .numeric } + 'general-numeric-sort' { o.ordering.mode = .general } + 'human-numeric-sort' { o.ordering.mode = .human } + 'month-sort' { o.ordering.mode = .month } + 'version-sort' { o.ordering.mode = .version } + 'random-sort' { o.ordering.random = true } + 'sort' { + if value in ['general-numeric', 'human-numeric', 'month', 'numeric', 'random', 'version'] { + o.ordering.mode = sort_word(value) + } else { + option_error("invalid argument '${value}' for --sort") + } + if value == 'random' { + o.ordering.random = true + } + } + 'key' { o.key_specs << value } + 'field-separator' { set_separator(mut o, value) } + 'output' { o.output = value } + 'files0-from' { o.files0_from = value } + 'buffer-size', 'temporary-directory', 'parallel', 'batch-size', + 'compress-program', 'random-source' { + } + 'check' { + match value { + '' { o.check = .diagnose } + 'diagnose-first' { o.check = .diagnose } + 'quiet', 'silent' { o.check = .quiet } + else { error_only("invalid argument '${value}' for '--check'") } + } + } + else { + option_error("unrecognized option '--${name}'") } - f.write_string(s) or { exit_error(err.msg()) } } - return tmp + return i } -@[noreturn] -fn exit_error(msg string) { - if msg.len > 0 { - eprintln('${app_name}: ${error}') +// parse_short_options walks a bundle of short options such as -rn, where one of +// them may take the rest of the bundle as its value. +fn parse_short_options(mut o Options, arg string, args []string, start_i int) int { + mut i := start_i + mut c := 0 + for c < arg.len { + letter := arg[c] + c++ + if in_bytes(short_value_options, letter) { + rest := arg[c..] + mut value := '' + value, i = take_value(short_name(letter), false, rest, args, i) + apply_short_with_value(mut o, letter, value) + return i + } + apply_short(mut o, letter) } + return i +} + +// short_name spells a short option letter for a message. +// short_name spells one short option letter for a message. +fn short_name(letter u8) string { + mut out := []u8{} + out << letter + return out.bytestr() +} + +fn in_bytes(set string, c u8) bool { + for x in set { + if x == c { + return true + } + } + return false +} + +fn in_string(set []string, needle string) bool { + for s in set { + if s == needle { + return true + } + } + return false +} + +fn apply_short(mut o Options, letter u8) { + match letter { + `b` { o.ordering.ignore_blanks = true } + `d` { o.ordering.dictionary = true } + `f` { o.ordering.fold_case = true } + `i` { o.ordering.ignore_nonprint = true } + `r` { o.ordering.reverse = true } + `s` { o.ordering.stable = true } + `m` { o.merge = true } + `u` { o.unique = true } + `c` { o.check = .diagnose } + `C` { o.check = .quiet } + `z` { o.zero_terminated = true } + `n` { o.ordering.mode = .numeric } + `g` { o.ordering.mode = .general } + `h` { o.ordering.mode = .human } + `M` { o.ordering.mode = .month } + `V` { o.ordering.mode = .version } + `R` { o.ordering.random = true } + else { + option_error("invalid option -- '${letter.str()}'") + } + } +} + +fn apply_short_with_value(mut o Options, letter u8, value string) { + match letter { + `k` { o.key_specs << value } + `t` { set_separator(mut o, value) } + `o` { o.output = value } + else { option_error("invalid option -- '${letter.str()}'") } + } +} + +// set_separator applies -t. GNU takes the rest of the option argument, or the +// next argument if that is empty, and refuses anything longer than one character. +// This message gets no advice about --help. +fn set_separator(mut o Options, value string) { + if value.len != 1 { + error_only("multi-character tab '${value}'") + } + o.separator = value + o.has_separator = true +} + +// sort_word maps a --sort argument onto a mode. An unknown word is rejected by the +// caller, since only it can say which option the word came from. +fn sort_word(word string) OrderMode { + mode := match word { + 'general-numeric' { OrderMode.general } + 'human-numeric' { OrderMode.human } + 'month' { OrderMode.month } + 'numeric' { OrderMode.numeric } + 'random' { + // --sort=random only sets the mode; the shuffling needs -R, which the + // caller of this function turns on through the same option. + OrderMode.ascii + } + 'version' { OrderMode.version } + else { OrderMode.ascii } + } + return mode +} + +// option_error reports a bad option, a bad argument or a missing value the way GNU +// does, and ends the program with the status GNU uses for those. +@[noreturn] +fn option_error(message string) { + eprintln('${app_name}: ${message}') eprintln("Try '${app_name} --help' for more information.") - exit(2) // exit(1) is used with the -c option + exit(2) +} + +// error_only reports a bad argument that GNU does not follow with the advice +// about --help, which is the case for a bad key and a bad --check argument. +@[noreturn] +fn error_only(message string) { + eprintln('${app_name}: ${message}') + exit(2) +} + +fn print_help() { + println('Usage: ${app_name} [OPTION]... [FILE]...') + println(' or: ${app_name} [OPTION]... --files0-from=F') + println('Write sorted concatenation of all FILE(s) to standard output.') + println('') + println('With no FILE, or when FILE is -, read standard input.') + println('') + println('Mandatory arguments to long options are mandatory for short options too.') + println('Ordering options:') + println('') + println(' -b, --ignore-leading-blanks ignore leading blanks') + println(' -d, --dictionary-order consider only blanks and alphanumeric characters') + println(' -f, --ignore-case fold lower case to upper case characters') + println(' -g, --general-numeric-sort compare according to general numerical value') + println(' -i, --ignore-nonprinting consider only printable characters') + println(" -M, --month-sort compare (unknown) < 'JAN' < ... < 'DEC'") + println(' -h, --human-numeric-sort compare human readable numbers (e.g., 2K 1G)') + println(' -n, --numeric-sort compare according to string numerical value;') + println(' see full documentation for supported strings') + println(' -R, --random-sort shuffle, but group identical keys. See shuf(1)') + println(' --random-source=FILE get random bytes from FILE') + println(' -r, --reverse reverse the result of comparisons') + println(' --sort=WORD sort according to WORD:') + println(' general-numeric -g, human-numeric -h, month -M,') + println(' numeric -n, random -R, version -V') + println(' -V, --version-sort natural sort of (version) numbers within text') + println('') + println('Other options:') + println('') + println(' --batch-size=NMERGE merge at most NMERGE inputs at once;') + println(' for more use temp files') + println(' -c, --check, --check=diagnose-first check for sorted input; do not sort') + println(' -C, --check=quiet, --check=silent like -c, but do not report first bad line') + println(' --compress-program=PROG compress temporaries with PROG;') + println(' decompress them with PROG -d') + println(' --debug annotate the part of the line used to sort, and') + println(' warn about questionable usage to standard error') + println(' --files0-from=F read input from the files specified by') + println(' NUL-terminated names in file F;') + println(' If F is - then read names from standard input') + println(' -k, --key=KEYDEF sort via a key; KEYDEF gives location and type') + println(' -m, --merge merge already sorted files; do not sort') + println(' -o, --output=FILE write result to FILE instead of standard output') + println(' -s, --stable stabilize sort by disabling last-resort comparison') + println(' -S, --buffer-size=SIZE use SIZE for main memory buffer') + println(' -t, --field-separator=SEP use SEP instead of non-blank to blank transition') + println(' -T, --temporary-directory=DIR use DIR for temporaries, not $TMPDIR or /tmp;') + println(' multiple options specify multiple directories') + println(' --parallel=N change the number of sorts run concurrently to N') + println(' -u, --unique output only the first of lines with equal keys;') + println(' with -c, check for strict ordering') + println(' -z, --zero-terminated line delimiter is NUL, not newline') + println(' --help display this help and exit') + println(' --version output version information and exit') + println('') + println('KEYDEF is F[.C][OPTS][,F[.C][OPTS]] for start and stop position, where F is a') + println('field number and C a character position in the field; both are origin 1, and') + println("the stop position defaults to the line's end. If neither -t nor -b is in") + println('effect, characters in a field are counted from the beginning of the preceding') + println('whitespace. OPTS is one or more single-letter ordering options [bdfgiMhnRrV],') + println('which override global ordering options for that key. If no key is given, use') + println('the entire line as the key. Use --debug to diagnose incorrect key usage.') + println('') + println(common.coreutils_footer()) } diff --git a/src/sort/sort.v b/src/sort/sort.v index bd5ca215..ce3b0944 100644 --- a/src/sort/sort.v +++ b/src/sort/sort.v @@ -1,180 +1,84 @@ import os -import arrays -import strconv -const space = ` ` -const tab = `\t` - -fn main() { - options := get_options() - results := sort(options) - if options.output_file == '' { - for result in results { - println(result) - } - } else { - os.write_lines(options.output_file, results) or { exit_error(err.msg()) } - } +// A line of input, and where it came from, which the check mode needs in order to +// report a disorder. +struct InputLine { + text string + file string + // number counts from one, so that the first line is line 1. + number int + // order is where the line came in. V's sort is not stable, and -s and -u both + // depend on lines that compare equal keeping the order they arrived in, so the + // position has to be part of the comparison. + order int } -fn sort(options Options) []string { - mut results := []string{} - for file in options.files { - results << do_sort(file, options) - } - return results -} - -fn do_sort(file string, options Options) []string { - mut lines := os.read_lines(file) or { exit_error(err.msg()) } - original := if options.check_diagnose || options.check_quiet { - lines.clone() - } else { - []string{} - } - match true { - // order matters here - options.sort_keys.len > 0 { sort_key(mut lines, options) } - options.numeric { sort_general_numeric(mut lines, options) } - options.ignore_case { sort_ignore_case(mut lines, options) } - options.dictionary_order { sort_dictionary_order(mut lines, options) } - options.ignore_non_printing { sort_ignore_non_printing(mut lines, options) } - options.ignore_leading_blanks { sort_ignore_leading_blanks(mut lines, options) } - else { sort_lines(mut lines, options) } - } +fn main() { + options, keys := get_options() - if options.unique { - lines = arrays.distinct(lines) - } - if original.len > 0 { - if lines != original { - if options.check_diagnose { - println('sort: not sorted') - } - exit(1) - } else { - if options.check_diagnose { - println('sort: already sorted') - } - exit(0) - } + // -c looks for one disorder and reports where it is, so there is only one input. + // GNU quotes the extra name and gives no advice about --help. + if options.check != .none && options.files.len > 1 { + error_only("extra operand '${options.files[1]}' not allowed with -c") } - return lines -} - -fn sort_lines(mut lines []string, options Options) { - cmp := if options.reverse { compare_strings_reverse } else { compare_strings } - lines.sort_with_compare(fn [cmp] (a &string, b &string) int { - return cmp(a, b) - }) -} - -fn compare_strings_reverse(a &string, b &string) int { - return compare_strings(b, a) -} -fn sort_ignore_case(mut lines []string, options Options) { - lines.sort_ignore_case() - if options.reverse { - lines.reverse_in_place() + if options.merge { + // -m promises each input is sorted already, so the streams are merged as + // they stand rather than sorted together. + emit(merge_input(read_each_input(options), options, keys), options) + return } -} - -// Ignore leading blanks when finding sort keys in each line. -// By default a blank is a space or a tab -fn sort_ignore_leading_blanks(mut lines []string, options Options) { - cmp := if options.reverse { compare_strings_reverse } else { compare_strings } - lines.sort_with_compare(fn [cmp] (a &string, b &string) int { - return cmp(trim_leading_spaces(a), trim_leading_spaces(b)) - }) -} - -fn trim_leading_spaces(s string) string { - return s.trim_left(' \n\t\v\f\r') -} -// Sort in phone directory order: ignore all characters except letters, digits -// and blanks when sorting. By default letters and digits are those of ASCII -fn sort_dictionary_order(mut lines []string, options Options) { - cmp := if options.reverse { compare_strings_reverse } else { compare_strings } - lines.sort_with_compare(fn [cmp] (a &string, b &string) int { - aa := a.bytes().map(is_dictionary_char).bytestr() - bb := b.bytes().map(is_dictionary_char).bytestr() - return cmp(aa, bb) - }) -} + lines := read_all_input(options) -fn is_dictionary_char(e u8) u8 { - return match e.is_digit() || e.is_letter() || e == space || e == tab { - true { e } - else { space } + if options.check != .none { + check_input(lines, options, keys) + return } -} -// Sort numerically, converting a prefix of each line to a long double-precision -// floating point number. See Floating point numbers. Do not report overflow, -// underflow, or conversion errors. Use the following collating sequence: -// Lines that do not start with numbers (all considered to be equal). -// - NaNs (“Not a Number” values, in IEEE floating point arithmetic) in a -// consistent but machine-dependent order. -// - Minus infinity. -// - Finite numbers in ascending numeric order (with -0 and +0 equal). -// - Plus infinity -fn sort_general_numeric(mut lines []string, options Options) { - cmp := if options.reverse { compare_strings_reverse } else { compare_strings } - lines.sort_with_compare(fn [cmp, options] (a &string, b &string) int { - numeric_a, rest_a := numeric_rest(a) - numeric_b, rest_b := numeric_rest(b) - numeric_diff := if options.reverse { numeric_b - numeric_a } else { numeric_a - numeric_b } - return if numeric_diff != 0 { - if numeric_diff > 0 { 1 } else { -1 } - } else { - cmp(rest_a, rest_b) - } - }) + mut sorted := sort_lines(lines, options, keys) + sorted = unique_lines(sorted, options, keys) + emit(sorted, options) } -const minus_infinity = f64(-0xFFFFFFFFFFFFFFF) - -fn numeric_rest(s string) (f64, string) { - mut allow_blanks := true - mut allow_sign := true - mut end := s.len - for i := 0; i < s.len; i++ { - c := s[i] - if allow_blanks && c == space { - continue +// emit writes the result to -o's file, or to standard output when no file was +// given. +fn emit(lines []InputLine, options Options) { + delim := line_delimiter(options) + if options.output.len > 0 { + mut file := os.create(options.output) or { + eprintln('${app_name}: cannot write ${options.output}: ${os.error_posix().msg()}') + exit(2) } - if allow_sign && (c == `-` || c == `+`) { - allow_sign = false - allow_blanks = false - continue + for line in lines { + file.write_string(line.text + delim) or { + exit_error('write failed') + } } - if c.is_digit() || c == strconv.c_dpoint { - allow_sign = false - allow_blanks = false - continue + file.close() + return + } + mut out := os.stdout() + for line in lines { + out.write_string(line.text + delim) or { + exit_error('write failed') } - // non-numeric char found - end = i - break } - num := strconv.atof64(s[0..end]) or { minus_infinity } - rest := s[end..] - return num, rest + out.flush() +} + +fn line_delimiter(options Options) string { + return if options.zero_terminated { '\x00' } else { '\n' } } -// This option has no effect if the stronger --dictionary-order (-d) option -// is also given. -fn sort_ignore_non_printing(mut lines []string, options Options) { - cmp := if options.reverse { compare_strings_reverse } else { compare_strings } - lines.sort_with_compare(fn [cmp] (a &string, b &string) int { - aa := a.bytes().map(is_printable).bytestr() - bb := b.bytes().map(is_printable).bytestr() - return cmp(aa, bb) - }) +@[noreturn] +fn exit_error(message string) { + eprintln('${app_name}: ${message}') + exit(2) } -fn is_printable(e u8) u8 { - return if e >= u8(` `) && e <= u8(`~`) { e } else { space } +@[noreturn] +fn read_error(file string) { + eprintln('${app_name}: cannot read: ${file}: ${os.error_posix().msg()}') + exit(2) } diff --git a/src/sort/sort_key.v b/src/sort/sort_key.v deleted file mode 100644 index cd705c7a..00000000 --- a/src/sort/sort_key.v +++ /dev/null @@ -1,195 +0,0 @@ -import strconv - -enum SortType { - ascii - numeric - leading - dictionary - ignore_case - ignore_non_printing - reverse -} - -struct SortKey { - f1 int - c1 int - f2 int - c2 int - sort_type SortType -} - -fn sort_key(mut lines []string, options Options) { - mut sort_keys := []SortKey{} - for sort_key in options.sort_keys { - sort_keys << parse_sort_key(sort_key) - } - - lines.sort_with_compare(fn [sort_keys, options] (a &string, b &string) int { - for key in sort_keys { - aa := find_field(a, key, options) - bb := find_field(b, key, options) - // println('${aa}, ${bb}') - result := match key.sort_type { - .numeric { compare_numeric(aa, bb) } - .leading { compare_leading(aa, bb) } - .dictionary { compare_dictionary(aa, bb) } - .ignore_case { compare_ignore_case(aa, bb) } - .ignore_non_printing { compare_ignore_non_printing(aa, bb) } - .reverse { compare_strings(bb, aa) } - else { compare_strings(aa, bb) } - } - - if result != 0 { - return result - } - } - return compare_strings(a, b) - }) -} - -fn compare_numeric(a &string, b &string) int { - af, ar := numeric_rest(a) - bf, br := numeric_rest(b) - diff := af - bf - return if diff != 0 { - match diff > 0 { - true { 1 } - else { -1 } - } - } else { - compare_strings(ar, br) - } -} - -fn compare_leading(a &string, b &string) int { - aa := trim_leading_spaces(a) - bb := trim_leading_spaces(b) - return compare_strings(aa, bb) -} - -fn compare_dictionary(a &string, b &string) int { - aa := a.bytes().map(is_dictionary_char).bytestr() - bb := b.bytes().map(is_dictionary_char).bytestr() - return compare_strings(aa, bb) -} - -fn compare_ignore_case(a &string, b &string) int { - return compare_strings(a.to_upper(), b.to_upper()) -} - -fn compare_ignore_non_printing(a &string, b &string) int { - aa := a.bytes().map(is_printable).bytestr() - bb := b.bytes().map(is_printable).bytestr() - return compare_strings(aa, bb) -} - -fn find_field(s string, key SortKey, options Options) string { - parts := s.split(options.field_separator) - f1 := key.f1 - 1 - c1 := if key.c1 > 0 { key.c1 - 1 } else { 0 } - f2 := key.f2 // from the end, don't subtrace 1 - c2 := key.c2 // from the end, don't subtrace 1 - start := if f1 < parts.len { f1 } else { 0 } - end := if f2 >= f1 && f2 < parts.len { f2 } else { parts.len } - join := parts[start..end].join('') - begin := join[c1..] - field := if c2 > 0 { - c := begin.len - c2 - begin[..c] - } else { - begin - } - return field -} - -fn parse_sort_key(k string) SortKey { - mut i := 0 - mut f1 := 0 - mut c1 := 0 - mut f2 := 0 - mut c2 := 0 - mut start := 0 - - // field - for ; i < k.len; i++ { - if !k[i].is_digit() { - f1 = strconv.atoi(k[start..i]) or { exit_error(err.msg()) } - break - } - } - - if f1 == 0 { - f1 = strconv.atoi(k[start..i]) or { exit_error(err.msg()) } - } - - // column - if i < k.len && k[i] == `.` { - i += 1 - start = i - for ; i < k.len; i++ { - if !k[i].is_digit() { - c1 = strconv.atoi(k[start..i]) or { exit_error(err.msg()) } - break - } - } - - if c1 == 0 { - c1 = strconv.atoi(k[start..i]) or { exit_error(err.msg()) } - } - } - - // sort option - sort_t := if i < k.len { k[i] } else { space } - - sort_type := match sort_t { - `b` { SortType.leading } - `d` { SortType.dictionary } - `f` { SortType.ignore_case } - `i` { SortType.ignore_non_printing } - `n` { SortType.numeric } - `r` { SortType.reverse } - else { SortType.ascii } - } - - if sort_type != .ascii { - i += 1 - } - - if i < k.len && k[i] == `,` { - i += 1 - start = i - for ; i < k.len; i++ { - if !k[i].is_digit() { - f2 = strconv.atoi(k[start..i]) or { exit_error(err.msg()) } - break - } - } - - if f2 == 0 { - f2 = strconv.atoi(k[start..i]) or { exit_error(err.msg()) } - } - - if i < k.len && k[i] == `.` { - i += 1 - start = i - for ; i < k.len; i++ { - if !k[i].is_digit() { - c2 = strconv.atoi(k[start..i]) or { exit_error(err.msg()) } - break - } - } - - if c2 == 0 { - c2 = strconv.atoi(k[start..i]) or { exit_error(err.msg()) } - } - } - } - - return SortKey{ - f1: f1 - c1: c1 - f2: f2 - c2: c2 - sort_type: sort_type - } -} diff --git a/src/sort/sort_keys_test.v b/src/sort/sort_keys_test.v index 141da909..0a2a8120 100644 --- a/src/sort/sort_keys_test.v +++ b/src/sort/sort_keys_test.v @@ -1,143 +1,133 @@ module main -import os - -const test_aa = os.temp_dir() + '/test_aa.txt' -const test_bb = os.temp_dir() + '/test_bb.txt' - -fn testsuite_begin() { - create_test_data() -} - -fn create_test_data() { - os.write_lines(test_aa, [ - 'Now is the time', - 'for all good men', - 'to come to the aid', - 'of their country', - ]) or {} - os.write_lines(test_bb, [ - ' 4.0 Now is the time', - ' 3.0 for all good men', - ' 2.0 to come to the aid', - ' 01. of their country', - ]) or {} -} - -// parse field tests - -fn test_parse_simple_field() { - assert parse_sort_key('2') == SortKey{ - f1: 2 - c1: 0 - f2: 0 - c2: 0 - sort_type: .ascii - } -} - -fn test_parse_field_column() { - assert parse_sort_key('2.1') == SortKey{ - f1: 2 - c1: 1 - f2: 0 - c2: 0 - sort_type: .ascii - } -} - -fn test_parse_field_column_sort_field() { - assert parse_sort_key('2.1b,3') == SortKey{ - f1: 2 - c1: 1 - f2: 3 - c2: 0 - sort_type: .leading - } -} - -fn test_parse_field_column_sort_field_column() { - assert parse_sort_key('2.1i,3.3') == SortKey{ - f1: 2 - c1: 1 - f2: 3 - c2: 3 - sort_type: .ignore_non_printing - } -} - -// find field tests -// -fn test_find_field_simple() { - key := SortKey{ - f1: 2 - c1: 0 - f2: 2 - c2: 0 - sort_type: .ascii - } - assert find_field('Now is the time', key, Options{}) == 'is' -} - -fn test_find_field_no_f2() { - key := SortKey{ - f1: 2 - c1: 0 - f2: 0 - c2: 0 - sort_type: .ascii - } - assert find_field('Now is the time', key, Options{}) == 'isthetime' -} - -fn test_find_field_full_spec() { - key := SortKey{ - f1: 1 - c1: 2 - f2: 4 - c2: 1 - sort_type: .ascii - } - assert find_field('Now is the time', key, Options{}) == 'owisthetim' -} - -// sorting - -fn test_sort_simple_column() { - options := Options{ - sort_keys: ['2'] - files: [test_aa] - } - assert sort(options) == [ - 'for all good men', - 'to come to the aid', - 'Now is the time', - 'of their country', - ] -} - -fn test_sort_full_spec() { - options := Options{ - sort_keys: ['1.2,4.1'] - files: [test_aa] - } - assert sort(options) == [ - 'of their country', - 'to come to the aid', - 'for all good men', - 'Now is the time', - ] -} - -fn test_sort_numeric_simple() { - options := Options{ - sort_keys: ['1n'] - files: [test_bb] - } - assert sort(options) == [ - ' 01. of their country', - ' 2.0 to come to the aid', - ' 3.0 for all good men', - ' 4.0 Now is the time', - ] +import common.testing + +// These are unit tests of the key parser, so they need no built program. +const rig = testing.prepare_rig(util: 'sort') + +// A key definition is F[.C][OPTS][,F[.C][OPTS]]. These tests read the positions +// and the ordering out of one, and then the text a key covers. + +fn base_ordering() Ordering { + return Ordering{} +} + +fn test_key_start_field_only() { + key := parse_key('2', base_ordering()) + + assert key.from_field == 2 + assert key.from_char == 0 + assert !key.has_to +} + +fn test_key_start_field_and_char() { + key := parse_key('2.1', base_ordering()) + + assert key.from_field == 2 + assert key.from_char == 1 + assert !key.has_to +} + +fn test_key_with_an_end() { + key := parse_key('2,3', base_ordering()) + + assert key.from_field == 2 + assert key.to_field == 3 + assert key.has_to +} + +fn test_key_with_end_chars() { + key := parse_key('2.1,3.3', base_ordering()) + + assert key.from_field == 2 + assert key.from_char == 1 + assert key.to_field == 3 + assert key.to_char == 3 +} + +// The ordering letters may sit on either side of the comma, and mean the same. + +fn test_ordering_letter_before_the_comma() { + key := parse_key('2n,2', base_ordering()) + + assert key.from_field == 2 + assert key.has_to + assert key.ordering.mode == OrderMode.numeric +} + +fn test_ordering_letter_after_the_end() { + key := parse_key('2,2n', base_ordering()) + + assert key.from_field == 2 + assert key.has_to + assert key.ordering.mode == OrderMode.numeric +} + +fn test_key_inherits_the_global_ordering() { + mut global := Ordering{} + global.fold_case = true + key := parse_key('2', global) + + assert key.ordering.fold_case +} + +fn test_key_ordering_overrides_the_global_one() { + mut global := Ordering{} + global.mode = .numeric + key := parse_key('2f', global) + + // A key names only the options it mentions, so -f adds the folding and leaves + // the -n that was given on the command line in place. + assert key.ordering.mode == .numeric + assert key.ordering.fold_case +} + +// Without -t, a field starts at the blank before it, and the first field has no +// blank before it to start from. + +fn test_key_of_the_second_field_includes_the_blank_before_it() { + key := parse_key('2,2', base_ordering()) + // GNU counts a field from the whitespace before it, so the blank is part of + // the key. + assert extract_key('Now is the time', &key, ' ', false, false) == ' is' +} + +fn test_key_of_the_first_field_is_the_whole_field() { + key := parse_key('1,1', base_ordering()) + assert extract_key('Now is the time', &key, ' ', false, false) == 'Now' +} + +fn test_ignore_blanks_drops_the_blank_before_the_field() { + key := parse_key('2,2', base_ordering()) + assert extract_key('Now is the time', &key, ' ', false, true) == 'is' +} + +fn test_field_separator_splits_the_fields() { + key := parse_key('2,2', base_ordering()) + assert extract_key('Now:is:the', &key, ':', true, false) == 'is' +} + +// A key with no end runs to the end of the line. + +fn test_key_without_an_end_runs_to_the_end() { + key := parse_key('2', base_ordering()) + assert extract_key('Now is the time', &key, ' ', false, false) == ' is the time' +} + +fn test_key_past_the_end_of_the_line_is_empty() { + key := parse_key('3,3', base_ordering()) + assert extract_key('Now is', &key, ' ', false, false) == '' +} + +// A character position is the last character of the key, counted from the start +// of the field. + +fn test_key_end_char_cuts_the_field_short() { + key := parse_key('1.2,1.4', base_ordering()) + assert extract_key('abcdef', &key, ' ', false, false) == 'bcd' +} + +fn test_key_start_char_moves_the_start() { + key := parse_key('1.2', base_ordering()) + assert extract_key('abcdef', &key, ' ', false, false) == 'bcdef' } diff --git a/src/sort/sort_test.v b/src/sort/sort_test.v index 062b6bb6..e5924dd3 100644 --- a/src/sort/sort_test.v +++ b/src/sort/sort_test.v @@ -1,207 +1,334 @@ module main import os +import common.testing -const test_a = os.temp_dir() + '/test_a.txt' -const test_b = os.temp_dir() + '/test_b.txt' -const test_c = os.temp_dir() + '/test_c.txt' -const test_d = os.temp_dir() + '/test_d.txt' -const test_e = os.temp_dir() + '/test_e.txt' +const rig = testing.prepare_rig(util: 'sort') +const executable_under_test = rig.executable_under_test +const eol = testing.output_eol() + +const words_path = os.join_path(rig.temp_dir, 'words.txt') +const numbers_path = os.join_path(rig.temp_dir, 'numbers.txt') +const pairs_path = os.join_path(rig.temp_dir, 'pairs.txt') +const months_path = os.join_path(rig.temp_dir, 'months.txt') +const versions_path = os.join_path(rig.temp_dir, 'versions.txt') +const dups_path = os.join_path(rig.temp_dir, 'dups.txt') +const plain_path = os.join_path(rig.temp_dir, 'plain.txt') +const od_path = os.join_path(rig.temp_dir, 'od.txt') +const sorted_path = os.join_path(rig.temp_dir, 'sorted.txt') fn testsuite_begin() { - create_test_data() -} - -fn create_test_data() { - os.write_lines(test_a, [ - 'Now is the time', - 'for all good men', - 'to come to the aid', - 'of their country', - ]) or {} - - os.write_lines(test_b, [ - ' Now is the time', - ' for all good men', - ' to come to the aid', - ' of their country', - ]) or {} - os.write_lines(test_c, [ - '% to come to the aid', - '* for all good men', - '# of their country', - '! Now is the time', - ]) or {} - os.write_lines(test_d, [ - '\xf1 Now is the time', - '\xf2 for all good men', - '\xf3 to come to the aid', - '\xf4 of their country', - ]) or {} - os.write_lines(test_e, [ - '100.1 Now is the time', - '50.2 for all good men', - 'to come to the aid', - '-24.3 of their country', - ]) or {} -} - -fn test_no_options() { - options := Options{ - files: [test_a] - } - assert sort(options) == [ - 'Now is the time', - 'for all good men', - 'of their country', - 'to come to the aid', - ] + os.write_lines(words_path, ['banana', 'Apple', 'cherry', 'apple', 'Banana']) or {} + os.write_lines(numbers_path, ['10', '9', '-5', 'abc', '100', '2']) or {} + os.write_lines(pairs_path, ['b 2', 'a 10', 'b 1', 'a 2']) or {} + os.write_lines(months_path, ['JAN', 'jan', 'Feb', 'DEC', 'unknown', 'Mar']) or {} + os.write_lines(versions_path, ['v1.10', 'v1.9', 'v1.2', 'v10.0']) or {} + os.write_lines(dups_path, ['dup', 'dup', 'dup', 'other', 'dup']) or {} + os.write_lines(plain_path, ['b 2', 'a 10', 'b 1', 'a 2']) or {} + os.write_file(od_path, 'b\0c\nd\0e\n') or {} + os.write_lines(sorted_path, ['a', 'b', 'c']) or {} +} + +fn testsuite_end() { + os.rm(words_path)! + os.rm(numbers_path)! + os.rm(pairs_path)! + os.rm(months_path)! + os.rm(versions_path)! + os.rm(dups_path)! + os.rm(plain_path)! + os.rm(od_path)! + os.rm(sorted_path)! +} + +fn test_help_and_version() { + rig.assert_help_and_version_options_work() +} + +// The byte comparison is the order LC_ALL=C gives, which is what the rig runs +// under, so a plain sort is the uppercase letters before the lower case ones. + +fn test_plain_order() { + res := os.execute('${executable_under_test} ${words_path}') + + assert res.exit_code == 0 + assert res.output == 'Apple${eol}Banana${eol}apple${eol}banana${eol}cherry${eol}' } fn test_reverse() { - options := Options{ - reverse: true - files: [test_a] - } - assert sort(options) == [ - 'to come to the aid', - 'of their country', - 'for all good men', - 'Now is the time', - ] + res := os.execute('${executable_under_test} -r ${words_path}') + + assert res.exit_code == 0 + assert res.output == 'cherry${eol}banana${eol}apple${eol}Banana${eol}Apple${eol}' } +// -f folds the case, so the two spellings of a word weigh the same and the +// last-resort comparison puts the upper case first. + fn test_ignore_case() { - options := Options{ - ignore_case: true - files: [test_a] - } - assert sort(options) == [ - 'for all good men', - 'Now is the time', - 'of their country', - 'to come to the aid', - ] -} - -fn test_ignore_case_reverse() { - options := Options{ - reverse: true - ignore_case: true - files: [test_a] - } - assert sort(options) == [ - 'to come to the aid', - 'of their country', - 'Now is the time', - 'for all good men', - ] -} - -fn test_ignore_leading_blanks() { - options := Options{ - ignore_leading_blanks: true - files: [test_b] - } - assert sort(options) == [ - ' Now is the time', - ' for all good men', - ' of their country', - ' to come to the aid', - ] -} - -fn test_ignore_leading_blanks_reverse() { - options := Options{ - reverse: true - ignore_leading_blanks: true - files: [test_b] - } - assert sort(options) == [ - ' to come to the aid', - ' of their country', - ' for all good men', - ' Now is the time', - ] -} - -fn test_dictionary_order() { - options := Options{ - dictionary_order: true - files: [test_c] - } - assert sort(options) == [ - '! Now is the time', - '* for all good men', - '# of their country', - '% to come to the aid', - ] -} - -fn test_dictionary_order_everse() { - options := Options{ - reverse: true - dictionary_order: true - files: [test_c] - } - assert sort(options) == [ - '% to come to the aid', - '# of their country', - '* for all good men', - '! Now is the time', - ] -} - -fn test_non_printing() { - options := Options{ - ignore_non_printing: true - files: [test_d] - } - assert sort(options) == [ - '\xf1 Now is the time', - '\xf2 for all good men', - '\xf4 of their country', - '\xf3 to come to the aid', - ] -} - -fn test_non_printing_reverse() { - options := Options{ - reverse: true - ignore_non_printing: true - files: [test_d] - } - assert sort(options) == [ - '\xf3 to come to the aid', - '\xf4 of their country', - '\xf2 for all good men', - '\xf1 Now is the time', - ] + res := os.execute('${executable_under_test} -f ${words_path}') + + assert res.exit_code == 0 + assert res.output == 'Apple${eol}apple${eol}Banana${eol}banana${eol}cherry${eol}' } +// -n weighs the number at the front. A line with no number weighs zero, which is +// why "abc" lands between -5 and 2. + fn test_numeric() { - options := Options{ - numeric: true - files: [test_e] - } - assert sort(options) == [ - 'to come to the aid', - '-24.3 of their country', - '50.2 for all good men', - '100.1 Now is the time', - ] + res := os.execute('${executable_under_test} -n ${numbers_path}') + + assert res.exit_code == 0 + assert res.output == '-5${eol}abc${eol}2${eol}9${eol}10${eol}100${eol}' } fn test_numeric_reverse() { - options := Options{ - numeric: true - reverse: true - files: [test_e] - } - assert sort(options) == [ - '100.1 Now is the time', - '50.2 for all good men', - '-24.3 of their country', - 'to come to the aid', - ] + res := os.execute('${executable_under_test} -n -r ${numbers_path}') + + assert res.exit_code == 0 + assert res.output == '100${eol}10${eol}9${eol}2${eol}abc${eol}-5${eol}' +} + +// -g reads what strtod does, so it takes an exponent and a hexadecimal float. + +fn test_general_numeric() { + res := os.execute('${executable_under_test} -g ${numbers_path}') + + assert res.exit_code == 0 + assert res.output == 'abc${eol}-5${eol}2${eol}9${eol}10${eol}100${eol}' +} + +// -h reads a size suffix, and only the capitals from P upwards. + +fn test_human_numeric() { + list := os.join_path(rig.temp_dir, 'human.txt') + os.write_lines(list, ['1K', '512', '1M', '2k', '1G', '10']) or {} + res := os.execute('${executable_under_test} -h ${list}') + + assert res.exit_code == 0 + assert res.output == '10${eol}512${eol}1K${eol}2k${eol}1M${eol}1G${eol}' + os.rm(list)! +} + +fn test_month_sort() { + res := os.execute('${executable_under_test} -M ${months_path}') + + assert res.exit_code == 0 + assert res.output == 'unknown${eol}JAN${eol}jan${eol}Feb${eol}Mar${eol}DEC${eol}' +} + +// -V weighs runs of digits as numbers, so 1.10 comes after 1.9. + +fn test_version_sort() { + res := os.execute('${executable_under_test} -V ${versions_path}') + + assert res.exit_code == 0 + assert res.output == 'v1.2${eol}v1.9${eol}v1.10${eol}v10.0${eol}' +} + +// A key weighs one field. The keys here are the second fields as strings, so 1 +// comes before 10 before 2. + +fn test_key_second_field() { + res := os.execute('${executable_under_test} -k2,2 ${pairs_path}') + + assert res.exit_code == 0 + assert res.output == 'b 1${eol}a 10${eol}a 2${eol}b 2${eol}' +} + +fn test_key_two_keys() { + res := os.execute('${executable_under_test} -k2,2 -k1,1 ${pairs_path}') + + assert res.exit_code == 0 + assert res.output == 'b 1${eol}a 10${eol}a 2${eol}b 2${eol}' +} + +fn test_key_with_numeric_ordering() { + res := os.execute('${executable_under_test} -k2,2n ${pairs_path}') + + assert res.exit_code == 0 + assert res.output == 'b 1${eol}a 2${eol}b 2${eol}a 10${eol}' +} + +// -t names the separator, and it has to be a single character. + +fn test_field_separator() { + list := os.join_path(rig.temp_dir, 'colon.txt') + os.write_lines(list, ['b:12', 'a:10', 'a:2']) or {} + res := os.execute('${executable_under_test} -t: -k2,2n ${list}') + + assert res.exit_code == 0 + assert res.output == 'a:2${eol}a:10${eol}b:12${eol}' + os.rm(list)! +} + +fn test_field_separator_rejects_two_characters() { + res := os.execute('${executable_under_test} -t:: ${pairs_path}') + + assert res.exit_code == 2 + assert res.output.trim_space() == "sort: multi-character tab '::'" +} + +// A field number of zero is not a position, so GNU rejects the definition. + +fn test_key_zero_field_is_an_error() { + res := os.execute('${executable_under_test} -k1,0 ${pairs_path}') + + assert res.exit_code == 2 + assert res.output.trim_space() == "sort: field number is zero: invalid field specification '1,0'" +} + +// Several files are one stream that is sorted as a whole, not one stream per file. + +fn test_several_files_are_one_stream() { + res := os.execute('${executable_under_test} ${numbers_path} ${words_path}') + + assert res.exit_code == 0 + lines := res.output.trim_space().split('\n') + assert lines.len == 11 + assert lines[0] == '-5' + assert lines[10] == 'cherry' +} + +// -m promises each input is sorted, so it merges without sorting and a line out +// of place in its own file stays there. + +fn test_merge_keeps_unsorted_input_as_it_is() { + unsorted := os.join_path(rig.temp_dir, 'unsorted.txt') + os.write_lines(unsorted, ['3', '1', '2']) or {} + res := os.execute('${executable_under_test} -m ${unsorted}') + + assert res.exit_code == 0 + assert res.output == '3${eol}1${eol}2${eol}' + os.rm(unsorted)! +} + +// -u keeps the first line of a run of equal keys, which with -n means equal +// numbers rather than equal lines. + +fn test_unique_by_key_under_numeric() { + res := os.execute('${executable_under_test} -n -u ${numbers_path}') + + assert res.exit_code == 0 + assert res.output == '-5${eol}abc${eol}2${eol}9${eol}10${eol}100${eol}' +} + +fn test_unique_plain() { + res := os.execute('${executable_under_test} -u ${dups_path}') + + assert res.exit_code == 0 + assert res.output == 'dup${eol}other${eol}' +} + +// -s turns off the last-resort comparison, so equal keys keep the order they +// arrived in. + +fn test_stable() { + list := os.join_path(rig.temp_dir, 'stable.txt') + os.write_lines(list, ['b 1', 'a 2', 'b 0', 'a 1']) or {} + res := os.execute('${executable_under_test} -s -k1,1 ${list}') + + assert res.exit_code == 0 + assert res.output == 'a 2${eol}a 1${eol}b 1${eol}b 0${eol}' + os.rm(list)! +} + +fn test_without_stable_the_lines_are_reordered() { + list := os.join_path(rig.temp_dir, 'unstable.txt') + os.write_lines(list, ['b 1', 'a 2', 'b 0', 'a 1']) or {} + res := os.execute('${executable_under_test} -k1,1 ${list}') + + assert res.exit_code == 0 + assert res.output == 'a 1${eol}a 2${eol}b 0${eol}b 1${eol}' + os.rm(list)! +} + +// -c names the file, the line and the line itself, and reports the first disorder +// only. -C says nothing. + +fn test_check_reports_the_first_disorder() { + res := os.execute('${executable_under_test} -c ${numbers_path}') + + assert res.exit_code == 1 + assert res.output == 'sort: ${numbers_path}:3: disorder: -5${eol}' +} + +fn test_check_quiet_on_a_disorder() { + res := os.execute('${executable_under_test} -C ${numbers_path}') + + assert res.exit_code == 1 + assert res.output == '' +} + +fn test_check_succeeds_on_sorted_input() { + res := os.execute('${executable_under_test} -c ${sorted_path}') + + assert res.exit_code == 0 + assert res.output == '' +} + +// With -c only one input is allowed, because the lines would otherwise have no +// single position to report. + +fn test_check_takes_one_file() { + res := os.execute('${executable_under_test} -c ${numbers_path} ${words_path}') + + assert res.exit_code == 2 + assert res.output.trim_space() == "sort: extra operand '${words_path}' not allowed with -c" +} + +fn test_output_to_a_file() { + out := os.join_path(rig.temp_dir, 'out.txt') + res := os.execute('${executable_under_test} -o ${out} ${numbers_path}') + + assert res.exit_code == 0 + assert res.output == '' + assert os.read_file(out)! == '-5\n10\n100\n2\n9\nabc\n' + os.rm(out)! +} + +fn test_reads_standard_input() { + res := os.execute('cat ${numbers_path} | ${executable_under_test}') + + assert res.exit_code == 0 + assert res.output == '-5${eol}10${eol}100${eol}2${eol}9${eol}abc${eol}' +} + +fn test_missing_file() { + res := os.execute('${executable_under_test} ${rig.temp_dir}/nosuchfile') + + assert res.exit_code == 2 + assert res.output.trim_space().starts_with('sort: cannot read:') +} + +// -z makes the NUL the record separator, which is how a file whose names contain +// newlines is sorted. + +fn test_zero_terminated() { + res := os.execute('${executable_under_test} -z ${od_path}') + + assert res.exit_code == 0 + assert res.output == 'b\x00c${eol}d\x00e${eol}\x00' +} + +// --files0-from reads the list of names from a file of NUL separated names. + +fn test_files0_from() { + list := os.join_path(rig.temp_dir, 'names') + os.write_file(list, '${words_path}\x00${numbers_path}\x00') or {} + res := os.execute('${executable_under_test} --files0-from=${list}') + + assert res.exit_code == 0 + lines := res.output.trim_space().split('\n') + assert lines.len == 11 + assert lines[0] == '-5' + os.rm(list)! +} + +fn test_unknown_option() { + res := os.execute('${executable_under_test} --no-such-option ${words_path}') + + assert res.exit_code == 2 + assert res.output.trim_space() == "sort: unrecognized option '--no-such-option' +Try 'sort --help' for more information." } diff --git a/src/sort/test.txt b/src/sort/test.txt deleted file mode 100644 index 73794140..00000000 --- a/src/sort/test.txt +++ /dev/null @@ -1,4 +0,0 @@ -Now is the time -for all good men -to come to the aid -of their country \ No newline at end of file diff --git a/src/sort/version.v b/src/sort/version.v new file mode 100644 index 00000000..438d9631 --- /dev/null +++ b/src/sort/version.v @@ -0,0 +1,127 @@ +// -V sorts the way version numbers read: runs of digits compare as numbers, so +// that 1.10 comes after 1.9, and the other characters weigh in an order that is +// not their byte order. That order is below. + +// compare_version compares two keys under -V. Runs of digits weigh as numbers, so +// that 1.10 comes after 1.9, and other characters weigh by version_rank: GNU puts +// the punctuation that introduces a version number first, then digits, then +// letters, then the punctuation that is only part of a name. The table was read off +// GNU 9.4 by sorting every printable character as a one character name. +fn compare_version(a string, b string) int { + mut i := 0 + mut j := 0 + for { + // A zero weighs as nothing only when it leads a number. Skipping every zero + // would make "0x10" weigh as "x10", which sorts after "3KiB" where GNU puts + // it first, and a blank weighs as itself, which is why " lead" sorts after + // "x:y" and before "#h". + for i < a.len && a[i] == `0` && leads_number(a, i) { + i++ + } + for j < b.len && b[j] == `0` && leads_number(b, j) { + j++ + } + if i >= a.len || j >= b.len { + break + } + if a[i].is_digit() && b[j].is_digit() { + // Both are at a digit run: compare the runs as numbers. + mut ai := i + for ai < a.len && a[ai].is_digit() { + ai++ + } + mut bj := j + for bj < b.len && b[bj].is_digit() { + bj++ + } + r := compare_digit_runs(a[i..ai], b[j..bj]) + if r != 0 { + return r + } + i = ai + j = bj + continue + } + if a[i] != b[j] { + return compare_ranks(version_rank(a[i], i), version_rank(b[j], j)) + } + i++ + j++ + } + // Whatever is left of one side decides: a prefix comes first. + if i >= a.len && j >= b.len { + return 0 + } + return if i >= a.len { -1 } else { 1 } +} + +// leads_number reports whether the zero at pos starts a run of digits, and so +// stands for a number rather than being the character itself. +fn leads_number(s string, pos int) bool { + return pos + 1 < s.len && s[pos + 1].is_digit() +} + +// version_rank is where a character sits in the order -V uses. A dot and a tilde +// come before everything only where they start the key, because there they +// introduce a version number; later in the key they are part of a name and rank +// with the other punctuation. The letters come next, and that last group keeps +// the byte order, so those characters are ranked by adding a constant. +fn version_rank(c u8, pos int) int { + if pos == 0 { + if c == `.` { + return 0 + } + if c == `~` { + return 1 + } + } + if c == `.` || c == `~` { + return 1000 + int(c) + } + if c.is_digit() { + return 2 + int(c - `0`) + } + if c >= `A` && c <= `Z` { + return 12 + int(c - `A`) + } + if c >= `a` && c <= `z` { + return 38 + int(c - `a`) + } + return 1000 + int(c) +} + +fn compare_ranks(a int, b int) int { + if a < b { + return -1 + } + if a > b { + return 1 + } + return 0 +} + +// compare_digit_runs compares two strings of digits as numbers. The leading zeros +// have already been skipped, so the longer run is the larger number unless the +// extra digits are zeros. +fn compare_digit_runs(a string, b string) int { + mut i := 0 + mut j := 0 + // Find the significant length of each run. + for i < a.len && a[i] == `0` { + i++ + } + for j < b.len && b[j] == `0` { + j++ + } + alen := a.len - i + blen := b.len - j + if alen != blen { + return if alen > blen { 1 } else { -1 } + } + for k in 0 .. alen { + if a[i + k] != b[j + k] { + return if a[i + k] < b[j + k] { -1 } else { 1 } + } + } + return 0 +}