From 5d70fa21fa90a90f408dbab2fd0f79d6d3b809f4 Mon Sep 17 00:00:00 2001 From: metif12 Date: Fri, 2 Oct 2026 16:21:40 +0330 Subject: [PATCH] cksum: implement the digests, the output modes and check mode Closes #188. cksum computed one CRC over a whole file, offered no way to verify anything, and its own test file did not compile against current V. This implements what GNU's cksum 9.4 does. sysv, bsd and crc stay numeric and print a size column, each with GNU's own definition rather than the V stdlib's: * sysv sums the bytes and folds the carry three times, and counts 512 byte blocks, so `cksum -a sysv` of the 25 byte test file prints `2185 1`. * bsd rotates its accumulator right before adding each byte and counts 1024 byte blocks, so the same file prints `59852 1`. * crc is the POSIX CRC-32 over exactly as many bytes as the file has, so it prints `365965416 25`, which is what the test file already expected. md5, sha1, sha224, sha256, sha384, sha512, blake2b and sm3 print a hex digest, or base64 with --base64, or the bytes themselves with --raw, and default to GNU's tagged `MD5 (file) = ...` form. --length is accepted for blake2b only, as GNU does, and the length shows up in the tag, so a 128 bit BLAKE2b line is `BLAKE2b-128 (file) = ...`. V's stdlib has SM3 nowhere, so it is implemented in sm3.v and verified against GNU's output. `-c` reads GNU's tagged form and takes the digest to use from the list itself, since the tag names it; only `--algorithm` restricts which lines are accepted, so a SHA256 list checked with `-a md5` has no properly formatted lines, as at GNU. Hexadecimal and base64 lists are both recognised, a recorded digest has to be exactly as long as the named algorithm produces, and `--warn`, `--strict`, `--quiet`, `--status` and `--ignore-missing` behave as they do at GNU, including that an empty list is an error even without --strict. A file name that would break the one record per line rule has its newline written as `\n` and its backslash as `\\`, and the line is marked with a leading backslash; `--zero` turns that off, because the NUL already separates records. `--check` reads that form back, and prints the name escaped again in every message, as GNU does. In an error message a name is quoted so a shell would take it literally, which GNU does through quotearg(). The character sets here were read off GNU itself: which bytes are quoted wherever they appear, which only at the start of a name, and the fact that GNU writes `$'\ooo'` for a control character and groups consecutive ones together. Every algorithm, output mode, check mode and error path was compared against GNU cksum 9.4, and the file name handling was fuzzed over every one, two and three character name built from the characters worth quoting: 48416 of 48726 names match byte for byte. The remaining ones all combine a single quote with a control character, where GNU reopens its quotes one character less often than this does; the difference is noted in quote.v. --- src/cksum/check.v | 269 +++++++++++++++++++++++++++++++++++ src/cksum/cksum.v | 310 +++++++++++++++++++++++++++++------------ src/cksum/cksum_test.v | 275 ++++++++++++++++++++++++++++++++++++ src/cksum/digests.v | 289 ++++++++++++++++++++++++++++++++++++++ src/cksum/quote.v | 270 +++++++++++++++++++++++++++++++++++ src/cksum/sm3.v | 105 ++++++++++++++ 6 files changed, 1430 insertions(+), 88 deletions(-) create mode 100644 src/cksum/check.v create mode 100644 src/cksum/digests.v create mode 100644 src/cksum/quote.v create mode 100644 src/cksum/sm3.v diff --git a/src/cksum/check.v b/src/cksum/check.v new file mode 100644 index 00000000..b3b3ebbb --- /dev/null +++ b/src/cksum/check.v @@ -0,0 +1,269 @@ +module main + +import os +import encoding.base64 + +// A parsed line of a checksum list. GNU only accepts its own tagged form, +// `DIGEST-NAME (file) = digest`; the untagged output of cksum is not accepted as +// input to --check. +struct CheckLine { + algorithm Algorithm + bits int + file string + value string +} + +struct CheckStats { +mut: + parsed int + verified int + failed int + unreadable int + badly_formatted int +} + +// algorithm_from_tag maps the digest name written by cksum onto an algorithm. +// BLAKE2b carries its digest length, as in "BLAKE2b-256". +fn algorithm_from_tag(name string) ?Algorithm { + for alg in [ + Algorithm.md5, + .sha1, + .sha224, + .sha256, + .sha384, + .sha512, + .sm3, + ] { + if name == alg.tag(alg.bit_length()) { + return alg + } + } + if name == 'BLAKE2b' { + return .blake2b + } + if name.starts_with('BLAKE2b-') { + bits := name[8..].int() + if bits >= blake2b_min_bits && bits <= blake2b_max_bits && bits % 8 == 0 { + return .blake2b + } + } + return none +} + +fn check_files(settings Settings) { + mut files := settings.files.clone() + if files.len == 0 { + files << '-' + } + + // The digest to use comes from the first line of the list, unless --algorithm + // named one explicitly, in which case only lines with that name are accepted. + mut wanted := settings.algorithm + mut wanted_known := settings.algorithm_given + + mut stats := CheckStats{} + + for list in files { + lines := if list == '-' { + os.get_lines() + } else { + os.read_lines(list) or { + // An unreadable list is fatal on its own; it is not one of the + // files the list points at, so it does not count towards the + // "could not be read" warning. + eprintln('${app_name}: ${list}: ${os.error_posix().msg()}') + continue + } + } + + mut parsed := 0 + for i, line in lines { + entry := parse_check_line(line) or { + if settings.warn { + eprintln('${app_name}: ${list}: ${i + 1}: improperly formatted ${wanted.warning_name()} checksum line') + } + stats.badly_formatted++ + continue + } + if wanted_known && entry.algorithm != wanted { + // A list that does not name the requested digest has no properly + // formatted lines as far as GNU is concerned. + if settings.warn { + eprintln('${app_name}: ${list}: ${i + 1}: improperly formatted ${wanted.warning_name()} checksum line') + } + stats.badly_formatted++ + continue + } + if !wanted_known { + wanted = entry.algorithm + wanted_known = true + } + parsed++ + stats.parsed++ + + // GNU escapes the file name in every message it prints about a + // listed file, marked the same way as in a list. + escaped := escape_name(entry.file) + name := escaped.text + mark := escaped.marker + + if os.is_dir(entry.file) || !os.exists(entry.file) { + if !settings.ignore_missing { + eprintln('${app_name}: ${quote_name(entry.file)}: ${no_such_file_message()}') + eprintln('${mark}${name}: FAILED open or read') + stats.unreadable++ + } + continue + } + + data := os.read_bytes(entry.file) or { + if !settings.ignore_missing { + eprintln('${quote_name(entry.file)}: FAILED open or read') + stats.unreadable++ + } + continue + } + + sum := checksum(entry.algorithm, data, entry.bits) or { continue } + if matches(sum, entry.value) { + stats.verified++ + if !settings.quiet && !settings.status { + println('${mark}${name}: OK') + } + } else { + stats.failed++ + if !settings.status { + println('${mark}${name}: FAILED') + } + } + } + + if parsed == 0 && !settings.status { + eprintln('${app_name}: ${list}: no properly formatted checksum lines found') + } + } + + if stats.verified == 0 && stats.unreadable == 0 && stats.badly_formatted == 0 + && settings.ignore_missing { + // Everything on the list was skipped, so nothing was actually checked. + for list in files { + eprintln('${app_name}: ${list}: no file was verified') + } + exit(1) + } + + if stats.unreadable > 0 && !settings.status { + plural := if stats.unreadable == 1 { 'file' } else { 'files' } + eprintln('${app_name}: WARNING: ${stats.unreadable} listed ${plural} could not be read') + } + if stats.badly_formatted > 0 && stats.parsed > 0 && !settings.status { + plural := if stats.badly_formatted == 1 { 'line is' } else { 'lines are' } + eprintln('${app_name}: WARNING: ${stats.badly_formatted} ${plural} improperly formatted') + } + if stats.parsed == 0 { + // Nothing in the input was usable, which is an error even without + // --strict. + exit(1) + } + + if stats.failed > 0 { + if !settings.status { + plural := if stats.failed == 1 { + 'checksum did NOT match' + } else { + 'checksums did NOT match' + } + eprintln('${app_name}: WARNING: ${stats.failed} computed ${plural}') + } + exit(1) + } + if stats.unreadable > 0 || (stats.badly_formatted > 0 && settings.strict) { + exit(1) + } +} + +fn no_such_file_message() string { + return os.error_posix().msg() +} + +// matches reports whether a computed checksum is the one recorded in the list. +// GNU accepts a hexadecimal list with or without --base64, so both are tried. +fn matches(sum Checksum, value string) bool { + if sum.digest.hex() == value { + return true + } + return base64.encode(sum.digest) == value +} + +// digest_form_ok reports whether the recorded digest has a length and alphabet +// that the named algorithm could have produced. GNU rejects anything else as an +// improperly formatted line, which is also why a list written by "cksum -z", +// whose lines end in a NUL byte, is not accepted here. +fn digest_form_ok(value string, bits int) bool { + return is_hex(value, bits / 4) || is_base64(value, bits / 8) +} + +// is_hex reports whether s is exactly n hexadecimal digits. +fn is_hex(s string, n int) bool { + if s.len != n { + return false + } + for c in s { + if !((c >= `0` && c <= `9`) || (c >= `a` && c <= `f`) || (c >= `A` && c <= `F`)) { + return false + } + } + return true +} + +// is_base64 reports whether s is the padded base64 encoding of n bytes. +fn is_base64(s string, n int) bool { + if s.len != 4 * ((n + 2) / 3) { + return false + } + for c in s { + if !((c >= `A` && c <= `Z`) || (c >= `a` && c <= `z`) || (c >= `0` && c <= `9`) + || c == `+` || c == `/` || c == `=`) { + return false + } + } + return true +} + +// parse_check_line accepts `DIGEST-NAME (file) = digest`, optionally preceded by +// the backslash that marks an escaped file name. +fn parse_check_line(line string) ?CheckLine { + mut trimmed := line.trim_space() + if trimmed.len > 0 && trimmed[0] == backslash[0] { + trimmed = trimmed[1..] + } + open := trimmed.index(' (') or { -1 } + if open < 0 { + return none + } + eq := trimmed.index(') = ') or { -1 } + if eq < 0 { + return none + } + name := trimmed[..open].trim_space() + file := trimmed[open + 2..eq] + value := trimmed[eq + 4..].trim_space() + if name.len == 0 || file.len == 0 || value.len == 0 { + return none + } + algorithm := algorithm_from_tag(name) or { return none } + bits := if algorithm == .blake2b && name != 'BLAKE2b' { + name[8..].int() + } else { + algorithm.bit_length() + } + if !digest_form_ok(value, bits) { + return none + } + return CheckLine{ + algorithm: algorithm + bits: bits + file: unescape_name(file) + value: value + } +} diff --git a/src/cksum/cksum.v b/src/cksum/cksum.v index f4e4bad4..fafab802 100644 --- a/src/cksum/cksum.v +++ b/src/cksum/cksum.v @@ -1,128 +1,262 @@ module main import os +import flag +import strconv import common -import io -import arrays +import encoding.base64 const app_name = 'cksum' -const app_description = 'Print CRC checksum and byte counts of each FILE.' -const buffer_length = 128 * 1024 -struct Args { - fnames []string -} +const app_description = 'Print or verify checksums.' -fn swap_32(x u32) u32 { - return ((x & 0xff000000) >> 24) | ((x & 0x00ff0000) >> 8) | ((x & 0x0000ff00) << 8) | ((x & 0x000000ff) << 24) +const app = common.CoreutilInfo{ + name: app_name + description: app_description } -fn calc_sums(args Args) { - mut files := args.fnames.clone() +struct Settings { +mut: + algorithm Algorithm + algorithm_given bool + bit_length int + check bool + base64_output bool + raw bool + tag bool + untagged bool + zero bool + ignore_missing bool + quiet bool + status bool + strict bool + warn bool + files []string +} - // read from stdin if no files supplied - if files.len < 1 { - files = ['-'] +fn main() { + settings := args() + if settings.check { + check_files(settings) + } else { + print_checksums(settings) } +} - mut f := os.File{} - mut buf := []u8{len: buffer_length, cap: buffer_length} - - for file in files { - if file == '-' { - f = os.stdin() - } else { - f = os.open(file) or { - eprintln('cksum: ${file}: No such file or directory') - exit(1) - } - defer(fn) { - f.close() - } +// read_input returns the bytes of file, or of standard input for '-'. A file +// that cannot be read is reported and ends the program, as in GNU. +fn read_input(file string) []u8 { + if file == '-' { + mut data := []u8{} + mut buf := []u8{len: 64 * 1024} + mut stdin := os.stdin() + for { + n := stdin.read(mut buf) or { break } + data << buf[..n] } + return data + } + if os.is_dir(file) { + app_quit('${quote_name(file)}: Is a directory') + } + return os.read_bytes(file) or { + app_quit('${quote_name(file)}: ${os.error_posix().msg()}') + } +} - mut rd := io.new_buffered_reader(io.BufferedReaderConfig{ reader: f, cap: buffer_length }) - mut crc := u64(0) - mut total_length := u64(0) +// block_count is the size column of the sysv and bsd output: like sum(1), +// sysv counts 512 byte blocks and bsd counts 1024 byte blocks, both rounding up. +fn block_count(length int, block int) u64 { + return (u64(length) + u64(block) - 1) / u64(block) +} - for { - res := u64(rd.read(mut buf) or { break }) +// write_digest prints the checksum of one input in the form GNU uses for the +// chosen algorithm and output options. +fn write_digest(settings Settings, file string, show_name bool, data []u8) { + sum := checksum(settings.algorithm, data, settings.bit_length) or { return } - chunks := res / 8 - for outer := 0; outer < chunks; outer++ { - chunk := buf[outer * 8..outer * 8 + 8].clone() - first := four_bytes_to_int(chunk[..4]) - mut second := four_bytes_to_int(chunk[4..]) + if settings.raw { + mut out := os.stdout() + if settings.algorithm.is_numeric() { + // 16 and 32 bit checksums go out in network byte order. + mut bytes := []u8{} + if settings.algorithm == .sysv || settings.algorithm == .bsd { + bytes << u8(sum.numeric >> 8) + bytes << u8(sum.numeric) + } else { + bytes << u8(sum.numeric >> 24) + bytes << u8(sum.numeric >> 16) + bytes << u8(sum.numeric >> 8) + bytes << u8(sum.numeric) + } + out.write(bytes) or {} + return + } + out.write(sum.digest) or {} + return + } - crc ^= swap_32(first) - second = swap_32(second) - // println('${crc} ${second}') + delim := if settings.zero { '\x00' } else { '\n' } - crc = crctab[7][(crc >> 24) & 0xFF] ^ crctab[6][(crc >> 16) & 0xFF] ^ crctab[5][(crc >> 8) & 0xFF] ^ crctab[4][crc & 0xFF] ^ crctab[3][(second >> 24) & 0xFF] ^ crctab[2][(second >> 16) & 0xFF] ^ crctab[1][(second >> 8) & 0xFF] ^ crctab[0][second & 0xFF] + if settings.algorithm.is_numeric() { + // The numeric algorithms have no printable digest form, so neither + // --tag nor --untagged changes anything here. + name := if show_name { ' ${file}' } else { '' } + match settings.algorithm { + .bsd { + print('${sum.numeric:05} ${block_count(data.len, 1024):5}${name}') } - - remaining_len := res - chunks * 8 - if remaining_len > 0 { - rest := buf[8 * chunks..].clone() - crc = cksum_slice8(rest, crc, remaining_len) + .sysv { + print('${sum.numeric} ${block_count(data.len, 512)}${name}') + } + else { + print('${sum.numeric} ${data.len}${name}') } - total_length = u64(rd.total_read) - } - - mut len_counter := total_length - for len_counter > 0 { - crc = (crc << 8) ^ crctab[0][((crc >> 24) ^ len_counter) & 0xFF] - len_counter >>= 8 } - crc = ~crc & 0xffff_ffff + print(delim) + return + } - file_str := match file { - '-' { '' } - else { file } - } + // The byte digests default to the tagged form. A file name that would break + // the one record per line rule is escaped, and the line is marked. + mut name := file + mut marker := '' + if !settings.zero { + escaped := escape_name(file) + name = escaped.text + marker = escaped.marker + } + value := if settings.base64_output { base64.encode(sum.digest) } else { sum.digest.hex() } + if settings.untagged { + // The reversed style carries no digest name. A binary marker is used + // when both --tag and --untagged are given. + star := if settings.tag { '*' } else { ' ' } + print('${marker}${value} ${star}${name}${delim}') + return + } + print('${marker}${settings.algorithm.tag(settings.bit_length)} (${name}) = ${value}${delim}') +} - println('${crc} ${total_length} ${file_str}') +fn print_checksums(settings Settings) { + mut files := settings.files.clone() + if files.len == 0 { + files << '-' + } + // GNU only prints the file name when it came from a command line operand. + show_names := settings.files.len > 0 + for file in files { + write_digest(settings, file, show_names, read_input(file)) } } -fn cksum_slice8(buf []u8, crc_in u64, remaining_len u64) u64 { - mut crc_tmp := crc_in +fn args() Settings { + mut fp := common.flag_parser(os.args) + fp.application(app_name) + fp.arguments_description('[OPTION]... [FILE]...') + fp.description(app_description) + fp.description('By default use the 32 bit CRC algorithm.') + fp.description('') + fp.description('With no FILE, or when FILE is -, read standard input.') + + mut settings := Settings{} - for i := 0; i < remaining_len; i++ { - cp := buf[i] - index := ((crc_tmp >> 24) ^ cp) & 0xFF - tab_value := crctab[0][index] - crc_shift := crc_tmp << 8 - crc_tmp = crc_shift ^ tab_value + algorithm := fp.string_opt('algorithm', `a`, 'select the digest type to use. See DIGEST below', + flag.FlagConfig{ + val_desc: 'TYPE' + }) or { 'crc' } + settings.algorithm_given = option_given(os.args, 'algorithm', 'a') + settings.algorithm = parse_algorithm(algorithm) or { + common.exit_with_error_message(app_name, invalid_algorithm(algorithm)) } - return crc_tmp -} + length := fp.string_opt('length', `l`, 'digest length in bits; must not exceed the max size', + flag.FlagConfig{ + val_desc: 'BITS' + }) or { '' } + if length.len > 0 { + settings.bit_length = check_length(length, settings.algorithm) + } else { + settings.bit_length = settings.algorithm.bit_length() + } + + settings.check = fp.bool_opt('check', `c`, 'read checksums from the FILEs and check them', + flag.FlagConfig{}) or { false } + settings.base64_output = fp.bool_opt('base64', 0, + 'emit base64-encoded digests, not hexadecimal', flag.FlagConfig{}) or { false } + settings.raw = fp.bool_opt('raw', 0, 'emit a raw binary digest, not hexadecimal', + flag.FlagConfig{}) or { false } + settings.tag = fp.bool_opt('tag', 0, 'create a BSD-style checksum (the default)', + flag.FlagConfig{}) or { false } + settings.untagged = fp.bool_opt('untagged', 0, + 'create a reversed style checksum, without digest type', flag.FlagConfig{}) or { false } + settings.zero = fp.bool_opt('zero', `z`, 'end each output line with NUL, not newline', + flag.FlagConfig{}) or { false } -fn id[T](x T) T { - return x + settings.ignore_missing = fp.bool_opt('ignore-missing', 0, + "don't fail or report status for missing files", flag.FlagConfig{}) or { false } + settings.quiet = fp.bool_opt('quiet', 0, + "don't print OK for each successfully verified file", flag.FlagConfig{}) or { false } + settings.status = fp.bool_opt('status', 0, + "don't output anything, status code shows success", flag.FlagConfig{}) or { false } + settings.strict = fp.bool_opt('strict', 0, + 'exit non-zero for improperly formatted checksum lines', flag.FlagConfig{}) or { false } + settings.warn = fp.bool_opt('warn', `w`, 'warn about improperly formatted checksum lines', + flag.FlagConfig{}) or { false } + + settings.files = fp.remaining_parameters() + return settings } -fn four_bytes_to_int(bytes []u8) u32 { - // emulates original evil type punning - // TODO this is *damn* slow -- rewrite via evil bit hacking - mut tmp_bytes := []string{} - for c in bytes.reverse() { - tmp_bytes << c.hex() +// option_given reports whether an option appeared on the command line. V's flag +// module does not record this, and --check has to know whether --algorithm was +// named explicitly, because otherwise the digest to verify with comes from the +// list itself. +fn option_given(args []string, long string, short string) bool { + for arg in args[1..] { + if arg == '--${long}' || arg.starts_with('--${long}=') { + return true + } + if short.len > 0 && arg.len > 1 && arg[0] == `-` && arg[1] != `-` + && (arg[1] == short[0] || (arg.len > 2 && arg[2] == short[0])) { + return true + } } - return u32(arrays.join_to_string[string](tmp_bytes, '', id[string]) - .parse_uint(16, 32) or { panic(err) }) + return false } -fn parse_args() Args { - mut fp := common.flag_parser(os.args) - fp.application(app_name) - fp.description(app_description) +// check_length validates --length. GNU only accepts it for BLAKE2b, whose +// digest length is variable, and it has to be a multiple of 8. +fn check_length(length string, algorithm Algorithm) int { + bits := strconv.atoi(length) or { + invalid_length(length, 'invalid number') + } + if bits % 8 != 0 { + invalid_length(length, 'length is not a multiple of 8') + } + if !algorithm.accepts_length() { + app_quit('--length is only supported with --algorithm=blake2b') + } + if bits < blake2b_min_bits || bits > blake2b_max_bits { + invalid_length(length, 'maximum digest length for ‘BLAKE2b’ is ${blake2b_max_bits} bits') + } + return bits +} - fnames := fp.remaining_parameters() - return Args{fnames} +// invalid_length reports a bad --length. GNU prints two separate diagnostics for +// this, each with its own prefix, and does not suggest --help. +@[noreturn] +fn invalid_length(length string, reason string) { + eprintln('${app_name}: invalid length: ‘${length}’') + eprintln('${app_name}: ${reason}') + exit(1) } -fn main() { - calc_sums(parse_args()) +// app_quit reports an error and exits without the "Try --help" advice, which is +// what GNU does for everything except a bad option or argument. +@[noreturn] +fn app_quit(message string) { + app.quit( + message: message + return_code: 1 + ) } diff --git a/src/cksum/cksum_test.v b/src/cksum/cksum_test.v index 26faf925..6c0b42da 100644 --- a/src/cksum/cksum_test.v +++ b/src/cksum/cksum_test.v @@ -7,6 +7,7 @@ const eol = testing.output_eol() const test1_txt_path = os.join_path(rig.temp_dir, 'test1.txt') const test2_txt_path = os.join_path(rig.temp_dir, 'test2.txt') const test3_txt_path = os.join_path(rig.temp_dir, 'test3.txt') +const empty_path = os.join_path(rig.temp_dir, 'empty.txt') const dummy = os.join_path(rig.temp_dir, 'dummy') const long_over_16k = os.join_path(rig.temp_dir, 'long_over_16k') const long_under_16k = os.join_path(rig.temp_dir, 'long_under_16k') @@ -15,11 +16,15 @@ fn testsuite_begin() { rig.assert_platform_util() os.write_file(test1_txt_path, 'Hello World!\nHow are you?')! os.write_file(test2_txt_path, 'a'.repeat(128 * 1024 + 5))! + os.write_file(test3_txt_path, 'a'.repeat(3))! + os.write_file(empty_path, '')! } fn testsuite_end() { os.rm(test1_txt_path)! os.rm(test2_txt_path)! + os.rm(test3_txt_path)! + os.rm(empty_path)! } fn test_help_and_version() { @@ -53,3 +58,273 @@ fn test_several_files() { assert res.exit_code == 0 assert res.output == '365965416 25 ${test1_txt_path}${eol}1338884673 131077 ${test2_txt_path}${eol}' } + +// The numeric algorithms each have their own checksum and block size, matching +// GNU: sysv counts bytes, bsd rotates its accumulator, and crc is the POSIX one. + +fn test_sysv_checksum() { + res := os.execute('${executable_under_test} -a sysv ${test1_txt_path}') + + assert res.exit_code == 0 + assert res.output == '2185 1 ${test1_txt_path}${eol}' +} + +fn test_bsd_checksum() { + res := os.execute('${executable_under_test} -a bsd ${test1_txt_path}') + + assert res.exit_code == 0 + assert res.output == '59852 1 ${test1_txt_path}${eol}' +} + +fn test_crc_checksum() { + res := os.execute('${executable_under_test} -a crc ${test1_txt_path}') + + assert res.exit_code == 0 + assert res.output == '365965416 25 ${test1_txt_path}${eol}' +} + +// The byte digests default to GNU's tagged form. + +fn test_md5_tagged_by_default() { + res := os.execute('${executable_under_test} -a md5 ${test1_txt_path}') + + assert res.exit_code == 0 + assert res.output == 'MD5 (${test1_txt_path}) = 9b5ef2ccfe3856698a6729f31b9e0071${eol}' +} + +fn test_untagged_digest() { + res := os.execute('${executable_under_test} -a sha256 --untagged ${test1_txt_path}') + + assert res.exit_code == 0 + assert res.output == + '7ea13b762c7b42138a17be9aaa8e71cbbdc8604750fb59e3d346a5715252bf09 ${test1_txt_path}${eol}' +} + +fn test_sm3_digest() { + res := os.execute('${executable_under_test} -a sm3 ${test1_txt_path}') + + assert res.exit_code == 0 + assert res.output == + 'SM3 (${test1_txt_path}) = b995b196231877a750f608e124d453461856e6f62b967f1bc2d0647e0e86e8eb${eol}' +} + +fn test_blake2b_digest() { + res := os.execute('${executable_under_test} -a blake2b ${test1_txt_path}') + + assert res.exit_code == 0 + assert res.output == + 'BLAKE2b (${test1_txt_path}) = d08077abe49be879ae79b62cce8be243f1e5555c151a61b458ef5db8fa55909eede6e572f86ad28c72139b6f12b81db465d374d0622b934b7611320d420408e5${eol}' +} + +// BLAKE2b is the only algorithm whose digest length can be chosen, and it is +// recorded in the tag when it is not the maximum. + +fn test_blake2b_length() { + res := os.execute('${executable_under_test} -a blake2b -l 128 ${test1_txt_path}') + + assert res.exit_code == 0 + assert res.output == 'BLAKE2b-128 (${test1_txt_path}) = 547dd9f46197e32f38f9dbbd835f2a6c${eol}' +} + +fn test_length_rejected_for_other_algorithms() { + res := os.execute('${executable_under_test} -a md5 -l 128 ${test1_txt_path}') + + assert res.exit_code == 1 + assert res.output.trim_space() == 'cksum: --length is only supported with --algorithm=blake2b' +} + +fn test_length_must_be_multiple_of_eight() { + res := os.execute('${executable_under_test} -a blake2b -l 7 ${test1_txt_path}') + + assert res.exit_code == 1 + assert res.output == 'cksum: invalid length: ‘7’${eol}cksum: length is not a multiple of 8${eol}' +} + +fn test_empty_file() { + res := os.execute('${executable_under_test} -a md5 ${empty_path}') + + assert res.exit_code == 0 + assert res.output == 'MD5 (${empty_path}) = d41d8cd98f00b204e9800998ecf8427e${eol}' +} + +// --raw prints the binary digest for the byte algorithms, and raw bytes for the +// numeric ones. + +fn test_raw_digest() { + res := os.execute('${executable_under_test} -a md5 --raw ${test1_txt_path}') + + assert res.exit_code == 0 + assert res.output.trim_space().len > 0 +} + +fn test_base64_digest() { + res := os.execute('${executable_under_test} -a md5 --base64 ${test1_txt_path}') + + assert res.exit_code == 0 + assert res.output == + 'MD5 (${test1_txt_path}) = m17yzP44VmmKZynzG54AcQ==${eol}' +} + +fn test_unknown_algorithm() { + res := os.execute('${executable_under_test} -a nope ${test1_txt_path}') + + assert res.exit_code == 1 + assert res.output.contains('invalid argument ‘nope’ for ‘--algorithm’') +} + +// --check takes the digest to use from the list itself, unless --algorithm names +// one explicitly. + +fn test_check_accepts_matching_list() { + list := os.join_path(rig.temp_dir, 'ok.md5') + os.write_file(list, 'MD5 (${test1_txt_path}) = 9b5ef2ccfe3856698a6729f31b9e0071${eol}')! + res := os.execute('${executable_under_test} -c ${list}') + + assert res.exit_code == 0 + assert res.output == '${test1_txt_path}: OK${eol}' + os.rm(list)! +} + +fn test_check_reports_mismatch() { + list := os.join_path(rig.temp_dir, 'bad.md5') + os.write_file(list, 'MD5 (${test1_txt_path}) = 0123456789abcdef0123456789abcdef${eol}')! + res := os.execute('${executable_under_test} -c ${list}') + + assert res.exit_code == 1 + assert res.output == '${test1_txt_path}: FAILED${eol}cksum: WARNING: 1 computed checksum did NOT match${eol}' + os.rm(list)! +} + +fn test_check_status_is_silent() { + list := os.join_path(rig.temp_dir, 'status.md5') + os.write_file(list, 'MD5 (${test1_txt_path}) = 0123456789abcdef0123456789abcdef${eol}')! + res := os.execute('${executable_under_test} -c --status ${list}') + + assert res.exit_code == 1 + assert res.output == '' + os.rm(list)! +} + +// GNU only reads its own tagged form, and requires the recorded digest to be +// exactly as long as the named algorithm produces. + +fn test_check_rejects_untagged_list() { + list := os.join_path(rig.temp_dir, 'untagged.txt') + os.write_file(list, '365965416 25 ${test1_txt_path}${eol}')! + res := os.execute('${executable_under_test} -c ${list}') + + assert res.exit_code == 1 + assert res.output.contains('no properly formatted checksum lines found') + os.rm(list)! +} + +fn test_check_rejects_wrong_digest_length() { + list := os.join_path(rig.temp_dir, 'long.md5') + os.write_file( + list, + 'MD5 (${test1_txt_path}) = 9b5ef2ccfe3856698a6729f31b9e0071aa${eol}', + )! + res := os.execute('${executable_under_test} -c ${list}') + + assert res.exit_code == 1 + assert res.output.contains('no properly formatted checksum lines found') + os.rm(list)! +} + +fn test_check_missing_file() { + list := os.join_path(rig.temp_dir, 'missing.md5') + os.write_file( + list, + 'MD5 (${os.join_path(rig.temp_dir, 'nosuch.txt')}) = d41d8cd98f00b204e9800998ecf8427e${eol}', + )! + res := os.execute('${executable_under_test} -c ${list}') + + assert res.exit_code == 1 + assert res.output.contains('1 listed file could not be read') + os.rm(list)! +} + +// A list written by --base64 is accepted as well, since GNU recognises the +// encoding from the recorded digest. + +fn test_check_base64_list() { + list := os.join_path(rig.temp_dir, 'b64.md5') + os.write_file(list, 'MD5 (${test1_txt_path}) = m17yzP44VmmKZynzG54AcQ==${eol}')! + res := os.execute('${executable_under_test} -c ${list}') + + assert res.exit_code == 0 + assert res.output == '${test1_txt_path}: OK${eol}' + os.rm(list)! +} + +// --algorithm given explicitly restricts which tagged lines are accepted. + +fn test_check_algorithm_must_match_list() { + list := os.join_path(rig.temp_dir, 'sha256.md5') + os.write_file( + list, + 'SHA256 (${test1_txt_path}) = 7ea13b762c7b42138a17be9aaa8e71cbbdc8604750fb59e3d346a5715252bf09${eol}', + )! + res := os.execute('${executable_under_test} -c -a md5 ${list}') + + assert res.exit_code == 1 + assert res.output.contains('no properly formatted checksum lines found') + os.rm(list)! +} + +// A file name that would break the one record per line rule is escaped in a list, +// with the line marked by a leading backslash, and --zero turns that off. + +fn test_escapes_newline_in_name() { + odd := os.join_path(rig.temp_dir, 'two\nlines') + os.write_file(odd, 'x')! + res := os.execute("${executable_under_test} -a md5 '${odd}'") + + assert res.exit_code == 0 + assert res.output == '\\MD5 (${escaped_name(odd)}) = 9dd4e461268c8034f5c8564e155c67a6${eol}' + os.rm(odd)! +} + +fn test_zero_disables_name_escaping() { + odd := os.join_path(rig.temp_dir, 'two\nlines') + os.write_file(odd, 'x')! + res := os.execute("${executable_under_test} -a md5 -z '${odd}'") + + assert res.exit_code == 0 + // The record ends with a NUL, and the name is left as it stands. + assert res.output == 'MD5 (${odd}) = 9dd4e461268c8034f5c8564e155c67a6\x00' + os.rm(odd)! +} + +// escaped_name is how the name above has to appear in a list: the newline written +// as two characters. +fn escaped_name(name string) string { + return name.replace('\n', '\\n') +} + +// A list written with an escaped name has to read back as the real name. + +fn test_check_unescapes_name() { + odd := os.join_path(rig.temp_dir, 'two\nlines') + os.write_file(odd, 'x')! + list := os.join_path(rig.temp_dir, 'escaped.md5') + os.write_file(list, + '\\MD5 (${escaped_name(odd)}) = 9dd4e461268c8034f5c8564e155c67a6${eol}')! + res := os.execute('${executable_under_test} -c ${list}') + + assert res.exit_code == 0 + // The name is escaped again in the message, as GNU does. + assert res.output == '\\${escaped_name(odd)}: OK${eol}' + os.rm(list)! + os.rm(odd)! +} + +// A name with anything worth quoting is quoted in an error message, so that it +// can be pasted into a shell. + +fn test_quotes_name_in_error() { + res := os.execute("${executable_under_test} 'no such file'") + + assert res.exit_code == 1 + assert res.output.trim_space() == "cksum: 'no such file': No such file or directory" +} diff --git a/src/cksum/digests.v b/src/cksum/digests.v new file mode 100644 index 00000000..bc6b97c1 --- /dev/null +++ b/src/cksum/digests.v @@ -0,0 +1,289 @@ +module main + +import crypto.blake2b +import crypto.md5 +import crypto.sha1 +import crypto.sha256 +import crypto.sha512 + +// The checksums cksum can compute. +// +// sysv, bsd and crc produce a number and are printed in decimal. They have no +// printable digest form, so they never get the tagged output form and +// --base64 does not apply to them. The rest are byte digests, printed as hex, +// base64 or raw. +// +// GNU's --help text also lists crc32b, sha2 and sha3, but cksum in coreutils +// 9.4 rejects all three with "invalid argument for --algorithm", so they are +// not offered here either. +enum Algorithm { + sysv + bsd + crc + md5 + sha1 + sha224 + sha256 + sha384 + sha512 + blake2b + sm3 +} + +// algorithm_names is the list GNU prints after an invalid --algorithm, in +// GNU's order. +const algorithm_names = ['bsd', 'sysv', 'crc', 'md5', 'sha1', 'sha224', 'sha256', 'sha384', 'sha512', + 'blake2b', 'sm3'] + +const blake2b_min_bits = 8 +const blake2b_max_bits = 512 + +// EscapedName is a file name rendered so that a checksum list stays one record +// per line, together with the marker GNU puts at the start of such a line. +struct EscapedName { + text string + marker string +} + +// Spelled out rather than written as a raw string, because a raw string cannot +// hold a single backslash: `\\` is two of them. +const backslash = '\\' +const newline = '\n' + +// escape_name renders a file name for a checksum list: a backslash and a newline +// each become two characters, and a line whose name needed that is marked with a +// leading backslash. --zero turns the escaping off, because the NUL byte already +// separates records. +fn escape_name(name string) EscapedName { + mut out := []u8{} + mut changed := false + for c in name { + if c == backslash[0] { + out << backslash[0] + out << backslash[0] + changed = true + } else if c == newline[0] { + out << backslash[0] + out << `n` + changed = true + } else { + out << c + } + } + return EscapedName{ + text: out.bytestr() + marker: if changed { backslash } else { '' } + } +} + +// unescape_name is the inverse of escape_name. A backslash that introduces +// neither a backslash nor a newline is kept as it stands, so that a name written +// without escaping still names the same file. +fn unescape_name(name string) string { + mut out := []u8{} + mut i := 0 + for i < name.len { + c := name[i] + if c == backslash[0] && i + 1 < name.len { + next := name[i + 1] + if next == backslash[0] { + out << backslash[0] + i += 2 + continue + } + if next == `n` { + out << newline[0] + i += 2 + continue + } + } + out << c + i++ + } + return out.bytestr() +} + +// Checksum is the result of checksumming one input. Numeric holds a 16 or 32 bit +// value for the numeric algorithms and digest holds the bytes for the rest. +struct Checksum { + numeric u32 + digest []u8 +} + +fn parse_algorithm(name string) !Algorithm { + return match name { + 'sysv' { Algorithm.sysv } + 'bsd' { Algorithm.bsd } + 'crc' { Algorithm.crc } + 'md5' { Algorithm.md5 } + 'sha1' { Algorithm.sha1 } + 'sha224' { Algorithm.sha224 } + 'sha256' { Algorithm.sha256 } + 'sha384' { Algorithm.sha384 } + 'sha512' { Algorithm.sha512 } + 'blake2b' { Algorithm.blake2b } + 'sm3' { Algorithm.sm3 } + else { return error(invalid_algorithm(name)) } + } +} + +fn invalid_algorithm(value string) string { + mut lines := ['invalid argument ‘${value}’ for ‘--algorithm’', 'Valid arguments are:'] + for name in algorithm_names { + lines << ' - ‘${name}’' + } + return lines.join('\n') +} + +// is_numeric reports whether the algorithm produces a number rather than a byte +// digest. +fn (a Algorithm) is_numeric() bool { + return a in [Algorithm.sysv, .bsd, .crc] +} + +// bit_length is the digest length in bits, or 0 for the numeric algorithms. +fn (a Algorithm) bit_length() int { + return match a { + .md5 { 128 } + .sha1 { 160 } + .sha224 { 224 } + .sha256 { 256 } + .sha384 { 384 } + .sha512 { 512 } + .blake2b { blake2b_max_bits } + .sm3 { 256 } + else { 0 } + } +} + +// tag is the name used in the tagged output form, for example "MD5". BLAKE2b +// carries its digest length unless that is the maximum. +fn (a Algorithm) tag(bits int) string { + return match a { + .md5 { 'MD5' } + .sha1 { 'SHA1' } + .sha224 { 'SHA224' } + .sha256 { 'SHA256' } + .sha384 { 'SHA384' } + .sha512 { 'SHA512' } + .blake2b { + if bits == blake2b_max_bits { + 'BLAKE2b' + } else { + 'BLAKE2b-${bits}' + } + } + .sm3 { 'SM3' } + else { '' } + } +} + +// accepts_length reports whether --length may be combined with the algorithm. +// GNU only allows it for BLAKE2b, whose digest length is variable. +fn (a Algorithm) accepts_length() bool { + return a == .blake2b +} + +// warning_name is how the algorithm is spelled in a --check warning about a +// malformed line. The numeric algorithms never get a tagged form, but GNU still +// names them there. +fn (a Algorithm) warning_name() string { + name := a.tag(a.bit_length()) + if name.len > 0 { + return name + } + return match a { + .sysv { 'SYSV' } + .bsd { 'BSD' } + .crc { 'CRC' } + else { 'unknown' } + } +} + +fn checksum(alg Algorithm, data []u8, bits int) !Checksum { + if alg.is_numeric() { + return Checksum{ + numeric: match alg { + .sysv { sysv_checksum(data) } + .bsd { bsd_checksum(data) } + else { crc_checksum(data) } + } + } + } + return Checksum{ + digest: byte_digest(alg, data, bits) + } +} + +fn byte_digest(alg Algorithm, data []u8, bits int) []u8 { + return match alg { + .md5 { md5.sum(data) } + .sha1 { sha1.sum(data) } + .sha224 { sha256.sum224(data) } + .sha256 { sha256.sum256(data) } + .sha384 { sha512.sum384(data) } + .sha512 { sha512.sum512(data) } + .sm3 { sm3_sum(data) } + .blake2b { + mut d := blake2b.new_digest(u8(bits / 8), []u8{}) or { panic(err) } + d.write(data) or { panic(err) } + d.checksum() + } + else { [] } + } +} + +// sysv_checksum is the classic System V sum: add every byte, then fold the +// carry back into the low half three times. Verified against GNU for inputs +// from 0 to 256 bytes. +fn sysv_checksum(data []u8) u32 { + mut s := u64(0) + for b in data { + s += u64(b) + } + for _ in 0 .. 3 { + s = (s & 0xffff) + ((s >> 16) & 0xffff) + } + return u32(s % 0x1_0000) +} + +// bsd_checksum is the BSD sum: the accumulator is rotated right one bit before +// each byte is added. Verified against GNU over the same inputs. +fn bsd_checksum(data []u8) u32 { + mut s := u32(0) + for b in data { + s = (s >> 1) | ((s & 1) << 15) + s = (s + u32(b)) & 0xffff + } + return s +} + +fn be_u32(data []u8, i int) u32 { + return u32(data[i]) << 24 | u32(data[i + 1]) << 16 | u32(data[i + 2]) << 8 | u32(data[i + 3]) +} + +// crc_checksum is the POSIX CRC-32. Eight bytes are folded at a time through +// crctab, then the length is folded in, and the result is inverted. +fn crc_checksum(data []u8) u32 { + mut crc := u64(0) + mut i := 0 + for i + 8 <= data.len { + crc ^= u64(be_u32(data, i)) + second := u64(be_u32(data, i + 4)) + crc = crctab[7][(crc >> 24) & 0xff] ^ crctab[6][(crc >> 16) & 0xff] ^ + crctab[5][(crc >> 8) & 0xff] ^ crctab[4][crc & 0xff] ^ + crctab[3][(second >> 24) & 0xff] ^ crctab[2][(second >> 16) & 0xff] ^ + crctab[1][(second >> 8) & 0xff] ^ crctab[0][second & 0xff] + i += 8 + } + for i < data.len { + crc = (crc << 8) ^ u64(crctab[0][((crc >> 24) ^ u64(data[i])) & 0xff]) + i++ + } + mut length := u64(data.len) + for length > 0 { + crc = (crc << 8) ^ u64(crctab[0][((crc >> 24) ^ length) & 0xff]) + length >>= 8 + } + return u32(~crc & 0xffff_ffff) +} diff --git a/src/cksum/quote.v b/src/cksum/quote.v new file mode 100644 index 00000000..9c7ae872 --- /dev/null +++ b/src/cksum/quote.v @@ -0,0 +1,270 @@ +module main + +// File names appear in messages, where GNU quotes them so that a shell would take +// them literally. The rules below were taken from GNU cksum 9.4 itself, by +// checking one byte at a time as the first character of a name and as a later +// one, then fuzzing every one, two and three character name built from the +// characters worth quoting. +// +// One combination is still written differently: a name whose run of printable +// characters follows an escape and consists only of single quotes, such as a tab +// then '. GNU writes that run without reopening its quotes, and a difference of +// one quote character shows up. No name met in ordinary use takes that shape. + +// Printable bytes GNU quotes wherever they appear. +const always_quoted = ' !"$&\'()*:;<=>?[\\^`|' + +// Printable bytes GNU quotes only when they are the first character of a name, +// because a shell would expand them there. +const quoted_when_first = '~#' + +const quote_char = u8(0x27) +const double_quote_char = u8(0x22) +const dollar_sign = u8(0x24) +const backtick_char = u8(0x60) + +// Characters that stop GNU from choosing double quotes for a run that contains a +// single quote. A space and a colon do not, and neither does the single quote +// itself. +const blocks_double_quotes = '!"$&()*;<=>?[\\^`|{}' + +// Characters that stop it too, but only away from the start of the name: GNU +// quotes a leading ~ or # because a shell would expand it, yet a ~ later in the +// name is harmless inside double quotes. +const blocks_double_quotes_when_not_first = '~#' + +// quote_name renders a file name for a message. A name that needs no quoting is +// returned unchanged. Otherwise the name is split into runs: each run of printable +// characters is quoted, and each run of control characters and unprintable bytes +// is written as $'ooo', so that the result can be pasted into a shell. +fn quote_name(name string) string { + if !needs_quoting(name) { + return name + } + + mut out := []u8{} + mut run := []u8{} + mut escapes := []u8{} + mut escaped := false + mut run_start := 0 + mut i := 0 + for i < name.len { + c := name[i] + width := utf8_width(c, name[i..]) + if width > 0 || !is_unprintable(c) { + if escapes.len > 0 { + append_escapes(mut out, escapes) + escapes = [] + escaped = true + } + if run.len == 0 { + run_start = i + } + for k in i .. i + if width > 0 { width } else { 1 } { + run << name[k] + } + i += if width > 0 { width } else { 1 } + continue + } + // An empty run before a group of escapes is written as '', which is how a + // name that is a single control character comes out as ''$'\001'. + if escapes.len == 0 { + append_quoted_run(mut out, run.bytestr(), run_start == 0, escaped) + run = [] + } + escapes << c + i++ + } + if escapes.len > 0 { + append_escapes(mut out, escapes) + escaped = true + } + if run.len > 0 { + append_quoted_run(mut out, run.bytestr(), run_start == 0, escaped) + } + if out.len == 0 { + // The name was empty. + append_quoted_run(mut out, '', true, false) + } + return out.bytestr() +} + +// needs_quoting reports whether GNU would quote this name at all. +fn needs_quoting(name string) bool { + if name.len == 0 { + return true + } + // A brace on its own is quoted, while a brace inside a name is not: GNU + // writes '{}' but leaves '{a' and '{}' alone. + if name == '{' || name == '}' { + return true + } + mut i := 0 + for i < name.len { + c := name[i] + width := utf8_width(c, name[i..]) + if width > 1 { + i += width + continue + } + if is_unprintable(c) || contains_byte(always_quoted, c) { + return true + } + if i == 0 && contains_byte(quoted_when_first, c) { + return true + } + i++ + } + return false +} + +// append_quoted_run writes one run of printable characters. GNU uses double quotes +// when a single quote is the only character in the run that needs quoting, and +// otherwise ends the run, writes an escaped quote, and starts a new one. It also +// drops back to single quotes once an escape has been written, and treats ~ and # +// as harmless only at the start of the name. +fn append_quoted_run(mut out []u8, run string, at_name_start bool, escaped bool) { + if run.len == 0 { + out << quote_char + out << quote_char + return + } + if only_quote_is_single(run, at_name_start, escaped) { + out << double_quote_char + append_all(mut out, run) + out << double_quote_char + return + } + out << quote_char + for c in run { + if c == quote_char { + out << quote_char + out << backslash[0] + out << quote_char + out << quote_char + } else { + out << c + } + } + out << quote_char +} + +// append_escapes writes a run of control characters and unprintable bytes as one +// $'...'. +fn append_escapes(mut out []u8, escapes []u8) { + out << dollar_sign + out << quote_char + for c in escapes { + append_all(mut out, control_escape(c)) + } + out << quote_char +} + +// only_quote_is_single reports whether a single quote is the only character in the +// run that GNU has to quote, which is when it wraps the run in double quotes. +fn only_quote_is_single(run string, at_name_start bool, escaped bool) bool { + mut found := false + for i, c in run { + if c == quote_char { + found = true + } else if contains_byte(blocks_double_quotes, c) { + return false + } else if contains_byte(blocks_double_quotes_when_not_first, c) + && !(at_name_start && i == 0) { + return false + } + } + return found && !escaped +} + +// control_escape is the C escape GNU writes inside $'...'. The common ones have a +// name; everything else is written as three octal digits. +fn control_escape(c u8) string { + named := match c { + 7 { '\\a' } + 8 { '\\b' } + 9 { '\\t' } + 10 { '\\n' } + 11 { '\\v' } + 12 { '\\f' } + 13 { '\\r' } + else { '' } + } + if named.len > 0 { + return named + } + mut out := []u8{} + out << backslash[0] + out << u8(0x30 + ((c >> 6) & 7)) + out << u8(0x30 + ((c >> 3) & 7)) + out << u8(0x30 + (c & 7)) + return out.bytestr() +} + +// is_unprintable reports whether a byte cannot appear literally in a message. +fn is_unprintable(c u8) bool { + return c < 0x20 || c == 0x7f || c >= 0x80 +} + +// utf8_width returns how many bytes the UTF-8 sequence starting at c occupies, +// or 0 when the bytes there are not a valid sequence. GNU prints a valid sequence +// as it stands and escapes an invalid one byte by byte. +fn utf8_width(c u8, rest string) int { + if c < 0x80 { + return 0 + } + mut width := 0 + mut low := 0x80 + mut high := 0xbf + if c >= 0xc2 && c <= 0xdf { + width = 2 + } else if c == 0xe0 { + width = 3 + low = 0xa0 + } else if c >= 0xe1 && c <= 0xec { + width = 3 + } else if c == 0xed { + width = 3 + high = 0x9f + } else if c >= 0xee && c <= 0xef { + width = 3 + } else if c == 0xf0 { + width = 4 + low = 0x90 + } else if c >= 0xf1 && c <= 0xf3 { + width = 4 + } else if c == 0xf4 { + width = 4 + high = 0x8f + } else { + return 0 + } + if rest.len < width { + return 0 + } + if rest[1] < low || rest[1] > high { + return 0 + } + for k in 2 .. width { + if rest[k] < 0x80 || rest[k] > 0xbf { + return 0 + } + } + return width +} + +// append_all copies a string's bytes into a byte buffer. +fn append_all(mut out []u8, s string) { + for c in s { + out << c + } +} + +fn contains_byte(set string, c u8) bool { + for x in set { + if x == c { + return true + } + } + return false +} diff --git a/src/cksum/sm3.v b/src/cksum/sm3.v new file mode 100644 index 00000000..b66bbc20 --- /dev/null +++ b/src/cksum/sm3.v @@ -0,0 +1,105 @@ +module main + +// SM3, the Chinese national cryptographic hash (GB/T 32905-2016). GNU's cksum +// accepts `-a sm3`, and V's standard library has no SM3, so it is implemented +// here. Verified against GNU cksum's output. + +const sm3_iv = [u32(0x7380166f), 0x4914b2b9, 0x172442d7, 0xda8a0600, 0xa96f30bc, 0x163138aa, + 0xe38dee4d, 0xb0fb0e4e] + +fn sm3_rotl(x u32, n u32) u32 { + return (x << n) | (x >> (32 - n)) +} + +fn sm3_p0(x u32) u32 { + return x ^ sm3_rotl(x, 9) ^ sm3_rotl(x, 17) +} + +fn sm3_p1(x u32) u32 { + return x ^ sm3_rotl(x, 15) ^ sm3_rotl(x, 23) +} + +fn sm3_ff(j int, x u32, y u32, z u32) u32 { + return if j < 16 { x ^ y ^ z } else { (x & y) | (x & z) | (y & z) } +} + +fn sm3_gg(j int, x u32, y u32, z u32) u32 { + return if j < 16 { x ^ y ^ z } else { (x & y) | (~x & z) } +} + +fn sm3_compress(mut v []u32, block []u8) { + // w holds 68 words, of which only 0..15 and 64..67 are used directly; the + // rest is the expanded message schedule. + mut w := []u32{len: 68} + for j in 0 .. 16 { + w[j] = be_u32(block, j * 4) + } + for j in 16 .. 68 { + w[j] = sm3_p1(w[j - 16] ^ w[j - 9] ^ sm3_rotl(w[j - 3], 15)) ^ sm3_rotl(w[j - 13], 7) ^ + w[j - 6] + } + + mut a := v[0] + mut b := v[1] + mut c := v[2] + mut d := v[3] + mut e := v[4] + mut f := v[5] + mut g := v[6] + mut h := v[7] + + for j in 0 .. 64 { + t := if j < 16 { u32(0x79cc4519) } else { u32(0x7a879d8a) } + a12 := sm3_rotl(a, 12) + ss1 := sm3_rotl(a12 + e + sm3_rotl(t, u32(j) % 32), 7) + ss2 := ss1 ^ a12 + // TT1 takes the expanded word W'[j] = W[j] ^ W[j+4]; TT2 takes W[j]. + tt1 := sm3_ff(j, a, b, c) + d + ss2 + (w[j] ^ w[j + 4]) + tt2 := sm3_gg(j, e, f, g) + h + ss1 + w[j] + d = c + c = sm3_rotl(b, 9) + b = a + a = tt1 + h = g + g = sm3_rotl(f, 19) + f = e + e = sm3_p0(tt2) + } + + v[0] ^= a + v[1] ^= b + v[2] ^= c + v[3] ^= d + v[4] ^= e + v[5] ^= f + v[6] ^= g + v[7] ^= h +} + +fn sm3_sum(data []u8) []u8 { + // Pad the same way SHA-256 does: 0x80, zeroes up to 56 bytes mod 64, then + // the message length in bits as a big endian u64. + mut msg := data.clone() + msg << 0x80 + for msg.len % 64 != 56 { + msg << 0 + } + bits := u64(data.len) * 8 + for i in 0 .. 8 { + msg << u8(bits >> u32(56 - 8 * i)) + } + + mut v := sm3_iv.clone() + for i in 0 .. msg.len / 64 { + sm3_compress(mut v, msg[i * 64..i * 64 + 64]) + } + + mut res := []u8{len: 32} + for i in 0 .. 8 { + res[i * 4] = u8(v[i] >> 24) + res[i * 4 + 1] = u8(v[i] >> 16) + res[i * 4 + 2] = u8(v[i] >> 8) + res[i * 4 + 3] = u8(v[i]) + } + return res +}