diff --git a/src/cksum/check.v b/src/cksum/check.v new file mode 100644 index 00000000..b3b3ebbb --- /dev/null +++ b/src/cksum/check.v @@ -0,0 +1,269 @@ +module main + +import os +import encoding.base64 + +// A parsed line of a checksum list. GNU only accepts its own tagged form, +// `DIGEST-NAME (file) = digest`; the untagged output of cksum is not accepted as +// input to --check. +struct CheckLine { + algorithm Algorithm + bits int + file string + value string +} + +struct CheckStats { +mut: + parsed int + verified int + failed int + unreadable int + badly_formatted int +} + +// algorithm_from_tag maps the digest name written by cksum onto an algorithm. +// BLAKE2b carries its digest length, as in "BLAKE2b-256". +fn algorithm_from_tag(name string) ?Algorithm { + for alg in [ + Algorithm.md5, + .sha1, + .sha224, + .sha256, + .sha384, + .sha512, + .sm3, + ] { + if name == alg.tag(alg.bit_length()) { + return alg + } + } + if name == 'BLAKE2b' { + return .blake2b + } + if name.starts_with('BLAKE2b-') { + bits := name[8..].int() + if bits >= blake2b_min_bits && bits <= blake2b_max_bits && bits % 8 == 0 { + return .blake2b + } + } + return none +} + +fn check_files(settings Settings) { + mut files := settings.files.clone() + if files.len == 0 { + files << '-' + } + + // The digest to use comes from the first line of the list, unless --algorithm + // named one explicitly, in which case only lines with that name are accepted. + mut wanted := settings.algorithm + mut wanted_known := settings.algorithm_given + + mut stats := CheckStats{} + + for list in files { + lines := if list == '-' { + os.get_lines() + } else { + os.read_lines(list) or { + // An unreadable list is fatal on its own; it is not one of the + // files the list points at, so it does not count towards the + // "could not be read" warning. + eprintln('${app_name}: ${list}: ${os.error_posix().msg()}') + continue + } + } + + mut parsed := 0 + for i, line in lines { + entry := parse_check_line(line) or { + if settings.warn { + eprintln('${app_name}: ${list}: ${i + 1}: improperly formatted ${wanted.warning_name()} checksum line') + } + stats.badly_formatted++ + continue + } + if wanted_known && entry.algorithm != wanted { + // A list that does not name the requested digest has no properly + // formatted lines as far as GNU is concerned. + if settings.warn { + eprintln('${app_name}: ${list}: ${i + 1}: improperly formatted ${wanted.warning_name()} checksum line') + } + stats.badly_formatted++ + continue + } + if !wanted_known { + wanted = entry.algorithm + wanted_known = true + } + parsed++ + stats.parsed++ + + // GNU escapes the file name in every message it prints about a + // listed file, marked the same way as in a list. + escaped := escape_name(entry.file) + name := escaped.text + mark := escaped.marker + + if os.is_dir(entry.file) || !os.exists(entry.file) { + if !settings.ignore_missing { + eprintln('${app_name}: ${quote_name(entry.file)}: ${no_such_file_message()}') + eprintln('${mark}${name}: FAILED open or read') + stats.unreadable++ + } + continue + } + + data := os.read_bytes(entry.file) or { + if !settings.ignore_missing { + eprintln('${quote_name(entry.file)}: FAILED open or read') + stats.unreadable++ + } + continue + } + + sum := checksum(entry.algorithm, data, entry.bits) or { continue } + if matches(sum, entry.value) { + stats.verified++ + if !settings.quiet && !settings.status { + println('${mark}${name}: OK') + } + } else { + stats.failed++ + if !settings.status { + println('${mark}${name}: FAILED') + } + } + } + + if parsed == 0 && !settings.status { + eprintln('${app_name}: ${list}: no properly formatted checksum lines found') + } + } + + if stats.verified == 0 && stats.unreadable == 0 && stats.badly_formatted == 0 + && settings.ignore_missing { + // Everything on the list was skipped, so nothing was actually checked. + for list in files { + eprintln('${app_name}: ${list}: no file was verified') + } + exit(1) + } + + if stats.unreadable > 0 && !settings.status { + plural := if stats.unreadable == 1 { 'file' } else { 'files' } + eprintln('${app_name}: WARNING: ${stats.unreadable} listed ${plural} could not be read') + } + if stats.badly_formatted > 0 && stats.parsed > 0 && !settings.status { + plural := if stats.badly_formatted == 1 { 'line is' } else { 'lines are' } + eprintln('${app_name}: WARNING: ${stats.badly_formatted} ${plural} improperly formatted') + } + if stats.parsed == 0 { + // Nothing in the input was usable, which is an error even without + // --strict. + exit(1) + } + + if stats.failed > 0 { + if !settings.status { + plural := if stats.failed == 1 { + 'checksum did NOT match' + } else { + 'checksums did NOT match' + } + eprintln('${app_name}: WARNING: ${stats.failed} computed ${plural}') + } + exit(1) + } + if stats.unreadable > 0 || (stats.badly_formatted > 0 && settings.strict) { + exit(1) + } +} + +fn no_such_file_message() string { + return os.error_posix().msg() +} + +// matches reports whether a computed checksum is the one recorded in the list. +// GNU accepts a hexadecimal list with or without --base64, so both are tried. +fn matches(sum Checksum, value string) bool { + if sum.digest.hex() == value { + return true + } + return base64.encode(sum.digest) == value +} + +// digest_form_ok reports whether the recorded digest has a length and alphabet +// that the named algorithm could have produced. GNU rejects anything else as an +// improperly formatted line, which is also why a list written by "cksum -z", +// whose lines end in a NUL byte, is not accepted here. +fn digest_form_ok(value string, bits int) bool { + return is_hex(value, bits / 4) || is_base64(value, bits / 8) +} + +// is_hex reports whether s is exactly n hexadecimal digits. +fn is_hex(s string, n int) bool { + if s.len != n { + return false + } + for c in s { + if !((c >= `0` && c <= `9`) || (c >= `a` && c <= `f`) || (c >= `A` && c <= `F`)) { + return false + } + } + return true +} + +// is_base64 reports whether s is the padded base64 encoding of n bytes. +fn is_base64(s string, n int) bool { + if s.len != 4 * ((n + 2) / 3) { + return false + } + for c in s { + if !((c >= `A` && c <= `Z`) || (c >= `a` && c <= `z`) || (c >= `0` && c <= `9`) + || c == `+` || c == `/` || c == `=`) { + return false + } + } + return true +} + +// parse_check_line accepts `DIGEST-NAME (file) = digest`, optionally preceded by +// the backslash that marks an escaped file name. +fn parse_check_line(line string) ?CheckLine { + mut trimmed := line.trim_space() + if trimmed.len > 0 && trimmed[0] == backslash[0] { + trimmed = trimmed[1..] + } + open := trimmed.index(' (') or { -1 } + if open < 0 { + return none + } + eq := trimmed.index(') = ') or { -1 } + if eq < 0 { + return none + } + name := trimmed[..open].trim_space() + file := trimmed[open + 2..eq] + value := trimmed[eq + 4..].trim_space() + if name.len == 0 || file.len == 0 || value.len == 0 { + return none + } + algorithm := algorithm_from_tag(name) or { return none } + bits := if algorithm == .blake2b && name != 'BLAKE2b' { + name[8..].int() + } else { + algorithm.bit_length() + } + if !digest_form_ok(value, bits) { + return none + } + return CheckLine{ + algorithm: algorithm + bits: bits + file: unescape_name(file) + value: value + } +} diff --git a/src/cksum/cksum.v b/src/cksum/cksum.v index f4e4bad4..fafab802 100644 --- a/src/cksum/cksum.v +++ b/src/cksum/cksum.v @@ -1,128 +1,262 @@ module main import os +import flag +import strconv import common -import io -import arrays +import encoding.base64 const app_name = 'cksum' -const app_description = 'Print CRC checksum and byte counts of each FILE.' -const buffer_length = 128 * 1024 -struct Args { - fnames []string -} +const app_description = 'Print or verify checksums.' -fn swap_32(x u32) u32 { - return ((x & 0xff000000) >> 24) | ((x & 0x00ff0000) >> 8) | ((x & 0x0000ff00) << 8) | ((x & 0x000000ff) << 24) +const app = common.CoreutilInfo{ + name: app_name + description: app_description } -fn calc_sums(args Args) { - mut files := args.fnames.clone() +struct Settings { +mut: + algorithm Algorithm + algorithm_given bool + bit_length int + check bool + base64_output bool + raw bool + tag bool + untagged bool + zero bool + ignore_missing bool + quiet bool + status bool + strict bool + warn bool + files []string +} - // read from stdin if no files supplied - if files.len < 1 { - files = ['-'] +fn main() { + settings := args() + if settings.check { + check_files(settings) + } else { + print_checksums(settings) } +} - mut f := os.File{} - mut buf := []u8{len: buffer_length, cap: buffer_length} - - for file in files { - if file == '-' { - f = os.stdin() - } else { - f = os.open(file) or { - eprintln('cksum: ${file}: No such file or directory') - exit(1) - } - defer(fn) { - f.close() - } +// read_input returns the bytes of file, or of standard input for '-'. A file +// that cannot be read is reported and ends the program, as in GNU. +fn read_input(file string) []u8 { + if file == '-' { + mut data := []u8{} + mut buf := []u8{len: 64 * 1024} + mut stdin := os.stdin() + for { + n := stdin.read(mut buf) or { break } + data << buf[..n] } + return data + } + if os.is_dir(file) { + app_quit('${quote_name(file)}: Is a directory') + } + return os.read_bytes(file) or { + app_quit('${quote_name(file)}: ${os.error_posix().msg()}') + } +} - mut rd := io.new_buffered_reader(io.BufferedReaderConfig{ reader: f, cap: buffer_length }) - mut crc := u64(0) - mut total_length := u64(0) +// block_count is the size column of the sysv and bsd output: like sum(1), +// sysv counts 512 byte blocks and bsd counts 1024 byte blocks, both rounding up. +fn block_count(length int, block int) u64 { + return (u64(length) + u64(block) - 1) / u64(block) +} - for { - res := u64(rd.read(mut buf) or { break }) +// write_digest prints the checksum of one input in the form GNU uses for the +// chosen algorithm and output options. +fn write_digest(settings Settings, file string, show_name bool, data []u8) { + sum := checksum(settings.algorithm, data, settings.bit_length) or { return } - chunks := res / 8 - for outer := 0; outer < chunks; outer++ { - chunk := buf[outer * 8..outer * 8 + 8].clone() - first := four_bytes_to_int(chunk[..4]) - mut second := four_bytes_to_int(chunk[4..]) + if settings.raw { + mut out := os.stdout() + if settings.algorithm.is_numeric() { + // 16 and 32 bit checksums go out in network byte order. + mut bytes := []u8{} + if settings.algorithm == .sysv || settings.algorithm == .bsd { + bytes << u8(sum.numeric >> 8) + bytes << u8(sum.numeric) + } else { + bytes << u8(sum.numeric >> 24) + bytes << u8(sum.numeric >> 16) + bytes << u8(sum.numeric >> 8) + bytes << u8(sum.numeric) + } + out.write(bytes) or {} + return + } + out.write(sum.digest) or {} + return + } - crc ^= swap_32(first) - second = swap_32(second) - // println('${crc} ${second}') + delim := if settings.zero { '\x00' } else { '\n' } - crc = crctab[7][(crc >> 24) & 0xFF] ^ crctab[6][(crc >> 16) & 0xFF] ^ crctab[5][(crc >> 8) & 0xFF] ^ crctab[4][crc & 0xFF] ^ crctab[3][(second >> 24) & 0xFF] ^ crctab[2][(second >> 16) & 0xFF] ^ crctab[1][(second >> 8) & 0xFF] ^ crctab[0][second & 0xFF] + if settings.algorithm.is_numeric() { + // The numeric algorithms have no printable digest form, so neither + // --tag nor --untagged changes anything here. + name := if show_name { ' ${file}' } else { '' } + match settings.algorithm { + .bsd { + print('${sum.numeric:05} ${block_count(data.len, 1024):5}${name}') } - - remaining_len := res - chunks * 8 - if remaining_len > 0 { - rest := buf[8 * chunks..].clone() - crc = cksum_slice8(rest, crc, remaining_len) + .sysv { + print('${sum.numeric} ${block_count(data.len, 512)}${name}') + } + else { + print('${sum.numeric} ${data.len}${name}') } - total_length = u64(rd.total_read) - } - - mut len_counter := total_length - for len_counter > 0 { - crc = (crc << 8) ^ crctab[0][((crc >> 24) ^ len_counter) & 0xFF] - len_counter >>= 8 } - crc = ~crc & 0xffff_ffff + print(delim) + return + } - file_str := match file { - '-' { '' } - else { file } - } + // The byte digests default to the tagged form. A file name that would break + // the one record per line rule is escaped, and the line is marked. + mut name := file + mut marker := '' + if !settings.zero { + escaped := escape_name(file) + name = escaped.text + marker = escaped.marker + } + value := if settings.base64_output { base64.encode(sum.digest) } else { sum.digest.hex() } + if settings.untagged { + // The reversed style carries no digest name. A binary marker is used + // when both --tag and --untagged are given. + star := if settings.tag { '*' } else { ' ' } + print('${marker}${value} ${star}${name}${delim}') + return + } + print('${marker}${settings.algorithm.tag(settings.bit_length)} (${name}) = ${value}${delim}') +} - println('${crc} ${total_length} ${file_str}') +fn print_checksums(settings Settings) { + mut files := settings.files.clone() + if files.len == 0 { + files << '-' + } + // GNU only prints the file name when it came from a command line operand. + show_names := settings.files.len > 0 + for file in files { + write_digest(settings, file, show_names, read_input(file)) } } -fn cksum_slice8(buf []u8, crc_in u64, remaining_len u64) u64 { - mut crc_tmp := crc_in +fn args() Settings { + mut fp := common.flag_parser(os.args) + fp.application(app_name) + fp.arguments_description('[OPTION]... [FILE]...') + fp.description(app_description) + fp.description('By default use the 32 bit CRC algorithm.') + fp.description('') + fp.description('With no FILE, or when FILE is -, read standard input.') + + mut settings := Settings{} - for i := 0; i < remaining_len; i++ { - cp := buf[i] - index := ((crc_tmp >> 24) ^ cp) & 0xFF - tab_value := crctab[0][index] - crc_shift := crc_tmp << 8 - crc_tmp = crc_shift ^ tab_value + algorithm := fp.string_opt('algorithm', `a`, 'select the digest type to use. See DIGEST below', + flag.FlagConfig{ + val_desc: 'TYPE' + }) or { 'crc' } + settings.algorithm_given = option_given(os.args, 'algorithm', 'a') + settings.algorithm = parse_algorithm(algorithm) or { + common.exit_with_error_message(app_name, invalid_algorithm(algorithm)) } - return crc_tmp -} + length := fp.string_opt('length', `l`, 'digest length in bits; must not exceed the max size', + flag.FlagConfig{ + val_desc: 'BITS' + }) or { '' } + if length.len > 0 { + settings.bit_length = check_length(length, settings.algorithm) + } else { + settings.bit_length = settings.algorithm.bit_length() + } + + settings.check = fp.bool_opt('check', `c`, 'read checksums from the FILEs and check them', + flag.FlagConfig{}) or { false } + settings.base64_output = fp.bool_opt('base64', 0, + 'emit base64-encoded digests, not hexadecimal', flag.FlagConfig{}) or { false } + settings.raw = fp.bool_opt('raw', 0, 'emit a raw binary digest, not hexadecimal', + flag.FlagConfig{}) or { false } + settings.tag = fp.bool_opt('tag', 0, 'create a BSD-style checksum (the default)', + flag.FlagConfig{}) or { false } + settings.untagged = fp.bool_opt('untagged', 0, + 'create a reversed style checksum, without digest type', flag.FlagConfig{}) or { false } + settings.zero = fp.bool_opt('zero', `z`, 'end each output line with NUL, not newline', + flag.FlagConfig{}) or { false } -fn id[T](x T) T { - return x + settings.ignore_missing = fp.bool_opt('ignore-missing', 0, + "don't fail or report status for missing files", flag.FlagConfig{}) or { false } + settings.quiet = fp.bool_opt('quiet', 0, + "don't print OK for each successfully verified file", flag.FlagConfig{}) or { false } + settings.status = fp.bool_opt('status', 0, + "don't output anything, status code shows success", flag.FlagConfig{}) or { false } + settings.strict = fp.bool_opt('strict', 0, + 'exit non-zero for improperly formatted checksum lines', flag.FlagConfig{}) or { false } + settings.warn = fp.bool_opt('warn', `w`, 'warn about improperly formatted checksum lines', + flag.FlagConfig{}) or { false } + + settings.files = fp.remaining_parameters() + return settings } -fn four_bytes_to_int(bytes []u8) u32 { - // emulates original evil type punning - // TODO this is *damn* slow -- rewrite via evil bit hacking - mut tmp_bytes := []string{} - for c in bytes.reverse() { - tmp_bytes << c.hex() +// option_given reports whether an option appeared on the command line. V's flag +// module does not record this, and --check has to know whether --algorithm was +// named explicitly, because otherwise the digest to verify with comes from the +// list itself. +fn option_given(args []string, long string, short string) bool { + for arg in args[1..] { + if arg == '--${long}' || arg.starts_with('--${long}=') { + return true + } + if short.len > 0 && arg.len > 1 && arg[0] == `-` && arg[1] != `-` + && (arg[1] == short[0] || (arg.len > 2 && arg[2] == short[0])) { + return true + } } - return u32(arrays.join_to_string[string](tmp_bytes, '', id[string]) - .parse_uint(16, 32) or { panic(err) }) + return false } -fn parse_args() Args { - mut fp := common.flag_parser(os.args) - fp.application(app_name) - fp.description(app_description) +// check_length validates --length. GNU only accepts it for BLAKE2b, whose +// digest length is variable, and it has to be a multiple of 8. +fn check_length(length string, algorithm Algorithm) int { + bits := strconv.atoi(length) or { + invalid_length(length, 'invalid number') + } + if bits % 8 != 0 { + invalid_length(length, 'length is not a multiple of 8') + } + if !algorithm.accepts_length() { + app_quit('--length is only supported with --algorithm=blake2b') + } + if bits < blake2b_min_bits || bits > blake2b_max_bits { + invalid_length(length, 'maximum digest length for ‘BLAKE2b’ is ${blake2b_max_bits} bits') + } + return bits +} - fnames := fp.remaining_parameters() - return Args{fnames} +// invalid_length reports a bad --length. GNU prints two separate diagnostics for +// this, each with its own prefix, and does not suggest --help. +@[noreturn] +fn invalid_length(length string, reason string) { + eprintln('${app_name}: invalid length: ‘${length}’') + eprintln('${app_name}: ${reason}') + exit(1) } -fn main() { - calc_sums(parse_args()) +// app_quit reports an error and exits without the "Try --help" advice, which is +// what GNU does for everything except a bad option or argument. +@[noreturn] +fn app_quit(message string) { + app.quit( + message: message + return_code: 1 + ) } diff --git a/src/cksum/cksum_test.v b/src/cksum/cksum_test.v index 26faf925..6c0b42da 100644 --- a/src/cksum/cksum_test.v +++ b/src/cksum/cksum_test.v @@ -7,6 +7,7 @@ const eol = testing.output_eol() const test1_txt_path = os.join_path(rig.temp_dir, 'test1.txt') const test2_txt_path = os.join_path(rig.temp_dir, 'test2.txt') const test3_txt_path = os.join_path(rig.temp_dir, 'test3.txt') +const empty_path = os.join_path(rig.temp_dir, 'empty.txt') const dummy = os.join_path(rig.temp_dir, 'dummy') const long_over_16k = os.join_path(rig.temp_dir, 'long_over_16k') const long_under_16k = os.join_path(rig.temp_dir, 'long_under_16k') @@ -15,11 +16,15 @@ fn testsuite_begin() { rig.assert_platform_util() os.write_file(test1_txt_path, 'Hello World!\nHow are you?')! os.write_file(test2_txt_path, 'a'.repeat(128 * 1024 + 5))! + os.write_file(test3_txt_path, 'a'.repeat(3))! + os.write_file(empty_path, '')! } fn testsuite_end() { os.rm(test1_txt_path)! os.rm(test2_txt_path)! + os.rm(test3_txt_path)! + os.rm(empty_path)! } fn test_help_and_version() { @@ -53,3 +58,273 @@ fn test_several_files() { assert res.exit_code == 0 assert res.output == '365965416 25 ${test1_txt_path}${eol}1338884673 131077 ${test2_txt_path}${eol}' } + +// The numeric algorithms each have their own checksum and block size, matching +// GNU: sysv counts bytes, bsd rotates its accumulator, and crc is the POSIX one. + +fn test_sysv_checksum() { + res := os.execute('${executable_under_test} -a sysv ${test1_txt_path}') + + assert res.exit_code == 0 + assert res.output == '2185 1 ${test1_txt_path}${eol}' +} + +fn test_bsd_checksum() { + res := os.execute('${executable_under_test} -a bsd ${test1_txt_path}') + + assert res.exit_code == 0 + assert res.output == '59852 1 ${test1_txt_path}${eol}' +} + +fn test_crc_checksum() { + res := os.execute('${executable_under_test} -a crc ${test1_txt_path}') + + assert res.exit_code == 0 + assert res.output == '365965416 25 ${test1_txt_path}${eol}' +} + +// The byte digests default to GNU's tagged form. + +fn test_md5_tagged_by_default() { + res := os.execute('${executable_under_test} -a md5 ${test1_txt_path}') + + assert res.exit_code == 0 + assert res.output == 'MD5 (${test1_txt_path}) = 9b5ef2ccfe3856698a6729f31b9e0071${eol}' +} + +fn test_untagged_digest() { + res := os.execute('${executable_under_test} -a sha256 --untagged ${test1_txt_path}') + + assert res.exit_code == 0 + assert res.output == + '7ea13b762c7b42138a17be9aaa8e71cbbdc8604750fb59e3d346a5715252bf09 ${test1_txt_path}${eol}' +} + +fn test_sm3_digest() { + res := os.execute('${executable_under_test} -a sm3 ${test1_txt_path}') + + assert res.exit_code == 0 + assert res.output == + 'SM3 (${test1_txt_path}) = b995b196231877a750f608e124d453461856e6f62b967f1bc2d0647e0e86e8eb${eol}' +} + +fn test_blake2b_digest() { + res := os.execute('${executable_under_test} -a blake2b ${test1_txt_path}') + + assert res.exit_code == 0 + assert res.output == + 'BLAKE2b (${test1_txt_path}) = d08077abe49be879ae79b62cce8be243f1e5555c151a61b458ef5db8fa55909eede6e572f86ad28c72139b6f12b81db465d374d0622b934b7611320d420408e5${eol}' +} + +// BLAKE2b is the only algorithm whose digest length can be chosen, and it is +// recorded in the tag when it is not the maximum. + +fn test_blake2b_length() { + res := os.execute('${executable_under_test} -a blake2b -l 128 ${test1_txt_path}') + + assert res.exit_code == 0 + assert res.output == 'BLAKE2b-128 (${test1_txt_path}) = 547dd9f46197e32f38f9dbbd835f2a6c${eol}' +} + +fn test_length_rejected_for_other_algorithms() { + res := os.execute('${executable_under_test} -a md5 -l 128 ${test1_txt_path}') + + assert res.exit_code == 1 + assert res.output.trim_space() == 'cksum: --length is only supported with --algorithm=blake2b' +} + +fn test_length_must_be_multiple_of_eight() { + res := os.execute('${executable_under_test} -a blake2b -l 7 ${test1_txt_path}') + + assert res.exit_code == 1 + assert res.output == 'cksum: invalid length: ‘7’${eol}cksum: length is not a multiple of 8${eol}' +} + +fn test_empty_file() { + res := os.execute('${executable_under_test} -a md5 ${empty_path}') + + assert res.exit_code == 0 + assert res.output == 'MD5 (${empty_path}) = d41d8cd98f00b204e9800998ecf8427e${eol}' +} + +// --raw prints the binary digest for the byte algorithms, and raw bytes for the +// numeric ones. + +fn test_raw_digest() { + res := os.execute('${executable_under_test} -a md5 --raw ${test1_txt_path}') + + assert res.exit_code == 0 + assert res.output.trim_space().len > 0 +} + +fn test_base64_digest() { + res := os.execute('${executable_under_test} -a md5 --base64 ${test1_txt_path}') + + assert res.exit_code == 0 + assert res.output == + 'MD5 (${test1_txt_path}) = m17yzP44VmmKZynzG54AcQ==${eol}' +} + +fn test_unknown_algorithm() { + res := os.execute('${executable_under_test} -a nope ${test1_txt_path}') + + assert res.exit_code == 1 + assert res.output.contains('invalid argument ‘nope’ for ‘--algorithm’') +} + +// --check takes the digest to use from the list itself, unless --algorithm names +// one explicitly. + +fn test_check_accepts_matching_list() { + list := os.join_path(rig.temp_dir, 'ok.md5') + os.write_file(list, 'MD5 (${test1_txt_path}) = 9b5ef2ccfe3856698a6729f31b9e0071${eol}')! + res := os.execute('${executable_under_test} -c ${list}') + + assert res.exit_code == 0 + assert res.output == '${test1_txt_path}: OK${eol}' + os.rm(list)! +} + +fn test_check_reports_mismatch() { + list := os.join_path(rig.temp_dir, 'bad.md5') + os.write_file(list, 'MD5 (${test1_txt_path}) = 0123456789abcdef0123456789abcdef${eol}')! + res := os.execute('${executable_under_test} -c ${list}') + + assert res.exit_code == 1 + assert res.output == '${test1_txt_path}: FAILED${eol}cksum: WARNING: 1 computed checksum did NOT match${eol}' + os.rm(list)! +} + +fn test_check_status_is_silent() { + list := os.join_path(rig.temp_dir, 'status.md5') + os.write_file(list, 'MD5 (${test1_txt_path}) = 0123456789abcdef0123456789abcdef${eol}')! + res := os.execute('${executable_under_test} -c --status ${list}') + + assert res.exit_code == 1 + assert res.output == '' + os.rm(list)! +} + +// GNU only reads its own tagged form, and requires the recorded digest to be +// exactly as long as the named algorithm produces. + +fn test_check_rejects_untagged_list() { + list := os.join_path(rig.temp_dir, 'untagged.txt') + os.write_file(list, '365965416 25 ${test1_txt_path}${eol}')! + res := os.execute('${executable_under_test} -c ${list}') + + assert res.exit_code == 1 + assert res.output.contains('no properly formatted checksum lines found') + os.rm(list)! +} + +fn test_check_rejects_wrong_digest_length() { + list := os.join_path(rig.temp_dir, 'long.md5') + os.write_file( + list, + 'MD5 (${test1_txt_path}) = 9b5ef2ccfe3856698a6729f31b9e0071aa${eol}', + )! + res := os.execute('${executable_under_test} -c ${list}') + + assert res.exit_code == 1 + assert res.output.contains('no properly formatted checksum lines found') + os.rm(list)! +} + +fn test_check_missing_file() { + list := os.join_path(rig.temp_dir, 'missing.md5') + os.write_file( + list, + 'MD5 (${os.join_path(rig.temp_dir, 'nosuch.txt')}) = d41d8cd98f00b204e9800998ecf8427e${eol}', + )! + res := os.execute('${executable_under_test} -c ${list}') + + assert res.exit_code == 1 + assert res.output.contains('1 listed file could not be read') + os.rm(list)! +} + +// A list written by --base64 is accepted as well, since GNU recognises the +// encoding from the recorded digest. + +fn test_check_base64_list() { + list := os.join_path(rig.temp_dir, 'b64.md5') + os.write_file(list, 'MD5 (${test1_txt_path}) = m17yzP44VmmKZynzG54AcQ==${eol}')! + res := os.execute('${executable_under_test} -c ${list}') + + assert res.exit_code == 0 + assert res.output == '${test1_txt_path}: OK${eol}' + os.rm(list)! +} + +// --algorithm given explicitly restricts which tagged lines are accepted. + +fn test_check_algorithm_must_match_list() { + list := os.join_path(rig.temp_dir, 'sha256.md5') + os.write_file( + list, + 'SHA256 (${test1_txt_path}) = 7ea13b762c7b42138a17be9aaa8e71cbbdc8604750fb59e3d346a5715252bf09${eol}', + )! + res := os.execute('${executable_under_test} -c -a md5 ${list}') + + assert res.exit_code == 1 + assert res.output.contains('no properly formatted checksum lines found') + os.rm(list)! +} + +// A file name that would break the one record per line rule is escaped in a list, +// with the line marked by a leading backslash, and --zero turns that off. + +fn test_escapes_newline_in_name() { + odd := os.join_path(rig.temp_dir, 'two\nlines') + os.write_file(odd, 'x')! + res := os.execute("${executable_under_test} -a md5 '${odd}'") + + assert res.exit_code == 0 + assert res.output == '\\MD5 (${escaped_name(odd)}) = 9dd4e461268c8034f5c8564e155c67a6${eol}' + os.rm(odd)! +} + +fn test_zero_disables_name_escaping() { + odd := os.join_path(rig.temp_dir, 'two\nlines') + os.write_file(odd, 'x')! + res := os.execute("${executable_under_test} -a md5 -z '${odd}'") + + assert res.exit_code == 0 + // The record ends with a NUL, and the name is left as it stands. + assert res.output == 'MD5 (${odd}) = 9dd4e461268c8034f5c8564e155c67a6\x00' + os.rm(odd)! +} + +// escaped_name is how the name above has to appear in a list: the newline written +// as two characters. +fn escaped_name(name string) string { + return name.replace('\n', '\\n') +} + +// A list written with an escaped name has to read back as the real name. + +fn test_check_unescapes_name() { + odd := os.join_path(rig.temp_dir, 'two\nlines') + os.write_file(odd, 'x')! + list := os.join_path(rig.temp_dir, 'escaped.md5') + os.write_file(list, + '\\MD5 (${escaped_name(odd)}) = 9dd4e461268c8034f5c8564e155c67a6${eol}')! + res := os.execute('${executable_under_test} -c ${list}') + + assert res.exit_code == 0 + // The name is escaped again in the message, as GNU does. + assert res.output == '\\${escaped_name(odd)}: OK${eol}' + os.rm(list)! + os.rm(odd)! +} + +// A name with anything worth quoting is quoted in an error message, so that it +// can be pasted into a shell. + +fn test_quotes_name_in_error() { + res := os.execute("${executable_under_test} 'no such file'") + + assert res.exit_code == 1 + assert res.output.trim_space() == "cksum: 'no such file': No such file or directory" +} diff --git a/src/cksum/digests.v b/src/cksum/digests.v new file mode 100644 index 00000000..bc6b97c1 --- /dev/null +++ b/src/cksum/digests.v @@ -0,0 +1,289 @@ +module main + +import crypto.blake2b +import crypto.md5 +import crypto.sha1 +import crypto.sha256 +import crypto.sha512 + +// The checksums cksum can compute. +// +// sysv, bsd and crc produce a number and are printed in decimal. They have no +// printable digest form, so they never get the tagged output form and +// --base64 does not apply to them. The rest are byte digests, printed as hex, +// base64 or raw. +// +// GNU's --help text also lists crc32b, sha2 and sha3, but cksum in coreutils +// 9.4 rejects all three with "invalid argument for --algorithm", so they are +// not offered here either. +enum Algorithm { + sysv + bsd + crc + md5 + sha1 + sha224 + sha256 + sha384 + sha512 + blake2b + sm3 +} + +// algorithm_names is the list GNU prints after an invalid --algorithm, in +// GNU's order. +const algorithm_names = ['bsd', 'sysv', 'crc', 'md5', 'sha1', 'sha224', 'sha256', 'sha384', 'sha512', + 'blake2b', 'sm3'] + +const blake2b_min_bits = 8 +const blake2b_max_bits = 512 + +// EscapedName is a file name rendered so that a checksum list stays one record +// per line, together with the marker GNU puts at the start of such a line. +struct EscapedName { + text string + marker string +} + +// Spelled out rather than written as a raw string, because a raw string cannot +// hold a single backslash: `\\` is two of them. +const backslash = '\\' +const newline = '\n' + +// escape_name renders a file name for a checksum list: a backslash and a newline +// each become two characters, and a line whose name needed that is marked with a +// leading backslash. --zero turns the escaping off, because the NUL byte already +// separates records. +fn escape_name(name string) EscapedName { + mut out := []u8{} + mut changed := false + for c in name { + if c == backslash[0] { + out << backslash[0] + out << backslash[0] + changed = true + } else if c == newline[0] { + out << backslash[0] + out << `n` + changed = true + } else { + out << c + } + } + return EscapedName{ + text: out.bytestr() + marker: if changed { backslash } else { '' } + } +} + +// unescape_name is the inverse of escape_name. A backslash that introduces +// neither a backslash nor a newline is kept as it stands, so that a name written +// without escaping still names the same file. +fn unescape_name(name string) string { + mut out := []u8{} + mut i := 0 + for i < name.len { + c := name[i] + if c == backslash[0] && i + 1 < name.len { + next := name[i + 1] + if next == backslash[0] { + out << backslash[0] + i += 2 + continue + } + if next == `n` { + out << newline[0] + i += 2 + continue + } + } + out << c + i++ + } + return out.bytestr() +} + +// Checksum is the result of checksumming one input. Numeric holds a 16 or 32 bit +// value for the numeric algorithms and digest holds the bytes for the rest. +struct Checksum { + numeric u32 + digest []u8 +} + +fn parse_algorithm(name string) !Algorithm { + return match name { + 'sysv' { Algorithm.sysv } + 'bsd' { Algorithm.bsd } + 'crc' { Algorithm.crc } + 'md5' { Algorithm.md5 } + 'sha1' { Algorithm.sha1 } + 'sha224' { Algorithm.sha224 } + 'sha256' { Algorithm.sha256 } + 'sha384' { Algorithm.sha384 } + 'sha512' { Algorithm.sha512 } + 'blake2b' { Algorithm.blake2b } + 'sm3' { Algorithm.sm3 } + else { return error(invalid_algorithm(name)) } + } +} + +fn invalid_algorithm(value string) string { + mut lines := ['invalid argument ‘${value}’ for ‘--algorithm’', 'Valid arguments are:'] + for name in algorithm_names { + lines << ' - ‘${name}’' + } + return lines.join('\n') +} + +// is_numeric reports whether the algorithm produces a number rather than a byte +// digest. +fn (a Algorithm) is_numeric() bool { + return a in [Algorithm.sysv, .bsd, .crc] +} + +// bit_length is the digest length in bits, or 0 for the numeric algorithms. +fn (a Algorithm) bit_length() int { + return match a { + .md5 { 128 } + .sha1 { 160 } + .sha224 { 224 } + .sha256 { 256 } + .sha384 { 384 } + .sha512 { 512 } + .blake2b { blake2b_max_bits } + .sm3 { 256 } + else { 0 } + } +} + +// tag is the name used in the tagged output form, for example "MD5". BLAKE2b +// carries its digest length unless that is the maximum. +fn (a Algorithm) tag(bits int) string { + return match a { + .md5 { 'MD5' } + .sha1 { 'SHA1' } + .sha224 { 'SHA224' } + .sha256 { 'SHA256' } + .sha384 { 'SHA384' } + .sha512 { 'SHA512' } + .blake2b { + if bits == blake2b_max_bits { + 'BLAKE2b' + } else { + 'BLAKE2b-${bits}' + } + } + .sm3 { 'SM3' } + else { '' } + } +} + +// accepts_length reports whether --length may be combined with the algorithm. +// GNU only allows it for BLAKE2b, whose digest length is variable. +fn (a Algorithm) accepts_length() bool { + return a == .blake2b +} + +// warning_name is how the algorithm is spelled in a --check warning about a +// malformed line. The numeric algorithms never get a tagged form, but GNU still +// names them there. +fn (a Algorithm) warning_name() string { + name := a.tag(a.bit_length()) + if name.len > 0 { + return name + } + return match a { + .sysv { 'SYSV' } + .bsd { 'BSD' } + .crc { 'CRC' } + else { 'unknown' } + } +} + +fn checksum(alg Algorithm, data []u8, bits int) !Checksum { + if alg.is_numeric() { + return Checksum{ + numeric: match alg { + .sysv { sysv_checksum(data) } + .bsd { bsd_checksum(data) } + else { crc_checksum(data) } + } + } + } + return Checksum{ + digest: byte_digest(alg, data, bits) + } +} + +fn byte_digest(alg Algorithm, data []u8, bits int) []u8 { + return match alg { + .md5 { md5.sum(data) } + .sha1 { sha1.sum(data) } + .sha224 { sha256.sum224(data) } + .sha256 { sha256.sum256(data) } + .sha384 { sha512.sum384(data) } + .sha512 { sha512.sum512(data) } + .sm3 { sm3_sum(data) } + .blake2b { + mut d := blake2b.new_digest(u8(bits / 8), []u8{}) or { panic(err) } + d.write(data) or { panic(err) } + d.checksum() + } + else { [] } + } +} + +// sysv_checksum is the classic System V sum: add every byte, then fold the +// carry back into the low half three times. Verified against GNU for inputs +// from 0 to 256 bytes. +fn sysv_checksum(data []u8) u32 { + mut s := u64(0) + for b in data { + s += u64(b) + } + for _ in 0 .. 3 { + s = (s & 0xffff) + ((s >> 16) & 0xffff) + } + return u32(s % 0x1_0000) +} + +// bsd_checksum is the BSD sum: the accumulator is rotated right one bit before +// each byte is added. Verified against GNU over the same inputs. +fn bsd_checksum(data []u8) u32 { + mut s := u32(0) + for b in data { + s = (s >> 1) | ((s & 1) << 15) + s = (s + u32(b)) & 0xffff + } + return s +} + +fn be_u32(data []u8, i int) u32 { + return u32(data[i]) << 24 | u32(data[i + 1]) << 16 | u32(data[i + 2]) << 8 | u32(data[i + 3]) +} + +// crc_checksum is the POSIX CRC-32. Eight bytes are folded at a time through +// crctab, then the length is folded in, and the result is inverted. +fn crc_checksum(data []u8) u32 { + mut crc := u64(0) + mut i := 0 + for i + 8 <= data.len { + crc ^= u64(be_u32(data, i)) + second := u64(be_u32(data, i + 4)) + crc = crctab[7][(crc >> 24) & 0xff] ^ crctab[6][(crc >> 16) & 0xff] ^ + crctab[5][(crc >> 8) & 0xff] ^ crctab[4][crc & 0xff] ^ + crctab[3][(second >> 24) & 0xff] ^ crctab[2][(second >> 16) & 0xff] ^ + crctab[1][(second >> 8) & 0xff] ^ crctab[0][second & 0xff] + i += 8 + } + for i < data.len { + crc = (crc << 8) ^ u64(crctab[0][((crc >> 24) ^ u64(data[i])) & 0xff]) + i++ + } + mut length := u64(data.len) + for length > 0 { + crc = (crc << 8) ^ u64(crctab[0][((crc >> 24) ^ length) & 0xff]) + length >>= 8 + } + return u32(~crc & 0xffff_ffff) +} diff --git a/src/cksum/quote.v b/src/cksum/quote.v new file mode 100644 index 00000000..9c7ae872 --- /dev/null +++ b/src/cksum/quote.v @@ -0,0 +1,270 @@ +module main + +// File names appear in messages, where GNU quotes them so that a shell would take +// them literally. The rules below were taken from GNU cksum 9.4 itself, by +// checking one byte at a time as the first character of a name and as a later +// one, then fuzzing every one, two and three character name built from the +// characters worth quoting. +// +// One combination is still written differently: a name whose run of printable +// characters follows an escape and consists only of single quotes, such as a tab +// then '. GNU writes that run without reopening its quotes, and a difference of +// one quote character shows up. No name met in ordinary use takes that shape. + +// Printable bytes GNU quotes wherever they appear. +const always_quoted = ' !"$&\'()*:;<=>?[\\^`|' + +// Printable bytes GNU quotes only when they are the first character of a name, +// because a shell would expand them there. +const quoted_when_first = '~#' + +const quote_char = u8(0x27) +const double_quote_char = u8(0x22) +const dollar_sign = u8(0x24) +const backtick_char = u8(0x60) + +// Characters that stop GNU from choosing double quotes for a run that contains a +// single quote. A space and a colon do not, and neither does the single quote +// itself. +const blocks_double_quotes = '!"$&()*;<=>?[\\^`|{}' + +// Characters that stop it too, but only away from the start of the name: GNU +// quotes a leading ~ or # because a shell would expand it, yet a ~ later in the +// name is harmless inside double quotes. +const blocks_double_quotes_when_not_first = '~#' + +// quote_name renders a file name for a message. A name that needs no quoting is +// returned unchanged. Otherwise the name is split into runs: each run of printable +// characters is quoted, and each run of control characters and unprintable bytes +// is written as $'ooo', so that the result can be pasted into a shell. +fn quote_name(name string) string { + if !needs_quoting(name) { + return name + } + + mut out := []u8{} + mut run := []u8{} + mut escapes := []u8{} + mut escaped := false + mut run_start := 0 + mut i := 0 + for i < name.len { + c := name[i] + width := utf8_width(c, name[i..]) + if width > 0 || !is_unprintable(c) { + if escapes.len > 0 { + append_escapes(mut out, escapes) + escapes = [] + escaped = true + } + if run.len == 0 { + run_start = i + } + for k in i .. i + if width > 0 { width } else { 1 } { + run << name[k] + } + i += if width > 0 { width } else { 1 } + continue + } + // An empty run before a group of escapes is written as '', which is how a + // name that is a single control character comes out as ''$'\001'. + if escapes.len == 0 { + append_quoted_run(mut out, run.bytestr(), run_start == 0, escaped) + run = [] + } + escapes << c + i++ + } + if escapes.len > 0 { + append_escapes(mut out, escapes) + escaped = true + } + if run.len > 0 { + append_quoted_run(mut out, run.bytestr(), run_start == 0, escaped) + } + if out.len == 0 { + // The name was empty. + append_quoted_run(mut out, '', true, false) + } + return out.bytestr() +} + +// needs_quoting reports whether GNU would quote this name at all. +fn needs_quoting(name string) bool { + if name.len == 0 { + return true + } + // A brace on its own is quoted, while a brace inside a name is not: GNU + // writes '{}' but leaves '{a' and '{}' alone. + if name == '{' || name == '}' { + return true + } + mut i := 0 + for i < name.len { + c := name[i] + width := utf8_width(c, name[i..]) + if width > 1 { + i += width + continue + } + if is_unprintable(c) || contains_byte(always_quoted, c) { + return true + } + if i == 0 && contains_byte(quoted_when_first, c) { + return true + } + i++ + } + return false +} + +// append_quoted_run writes one run of printable characters. GNU uses double quotes +// when a single quote is the only character in the run that needs quoting, and +// otherwise ends the run, writes an escaped quote, and starts a new one. It also +// drops back to single quotes once an escape has been written, and treats ~ and # +// as harmless only at the start of the name. +fn append_quoted_run(mut out []u8, run string, at_name_start bool, escaped bool) { + if run.len == 0 { + out << quote_char + out << quote_char + return + } + if only_quote_is_single(run, at_name_start, escaped) { + out << double_quote_char + append_all(mut out, run) + out << double_quote_char + return + } + out << quote_char + for c in run { + if c == quote_char { + out << quote_char + out << backslash[0] + out << quote_char + out << quote_char + } else { + out << c + } + } + out << quote_char +} + +// append_escapes writes a run of control characters and unprintable bytes as one +// $'...'. +fn append_escapes(mut out []u8, escapes []u8) { + out << dollar_sign + out << quote_char + for c in escapes { + append_all(mut out, control_escape(c)) + } + out << quote_char +} + +// only_quote_is_single reports whether a single quote is the only character in the +// run that GNU has to quote, which is when it wraps the run in double quotes. +fn only_quote_is_single(run string, at_name_start bool, escaped bool) bool { + mut found := false + for i, c in run { + if c == quote_char { + found = true + } else if contains_byte(blocks_double_quotes, c) { + return false + } else if contains_byte(blocks_double_quotes_when_not_first, c) + && !(at_name_start && i == 0) { + return false + } + } + return found && !escaped +} + +// control_escape is the C escape GNU writes inside $'...'. The common ones have a +// name; everything else is written as three octal digits. +fn control_escape(c u8) string { + named := match c { + 7 { '\\a' } + 8 { '\\b' } + 9 { '\\t' } + 10 { '\\n' } + 11 { '\\v' } + 12 { '\\f' } + 13 { '\\r' } + else { '' } + } + if named.len > 0 { + return named + } + mut out := []u8{} + out << backslash[0] + out << u8(0x30 + ((c >> 6) & 7)) + out << u8(0x30 + ((c >> 3) & 7)) + out << u8(0x30 + (c & 7)) + return out.bytestr() +} + +// is_unprintable reports whether a byte cannot appear literally in a message. +fn is_unprintable(c u8) bool { + return c < 0x20 || c == 0x7f || c >= 0x80 +} + +// utf8_width returns how many bytes the UTF-8 sequence starting at c occupies, +// or 0 when the bytes there are not a valid sequence. GNU prints a valid sequence +// as it stands and escapes an invalid one byte by byte. +fn utf8_width(c u8, rest string) int { + if c < 0x80 { + return 0 + } + mut width := 0 + mut low := 0x80 + mut high := 0xbf + if c >= 0xc2 && c <= 0xdf { + width = 2 + } else if c == 0xe0 { + width = 3 + low = 0xa0 + } else if c >= 0xe1 && c <= 0xec { + width = 3 + } else if c == 0xed { + width = 3 + high = 0x9f + } else if c >= 0xee && c <= 0xef { + width = 3 + } else if c == 0xf0 { + width = 4 + low = 0x90 + } else if c >= 0xf1 && c <= 0xf3 { + width = 4 + } else if c == 0xf4 { + width = 4 + high = 0x8f + } else { + return 0 + } + if rest.len < width { + return 0 + } + if rest[1] < low || rest[1] > high { + return 0 + } + for k in 2 .. width { + if rest[k] < 0x80 || rest[k] > 0xbf { + return 0 + } + } + return width +} + +// append_all copies a string's bytes into a byte buffer. +fn append_all(mut out []u8, s string) { + for c in s { + out << c + } +} + +fn contains_byte(set string, c u8) bool { + for x in set { + if x == c { + return true + } + } + return false +} diff --git a/src/cksum/sm3.v b/src/cksum/sm3.v new file mode 100644 index 00000000..b66bbc20 --- /dev/null +++ b/src/cksum/sm3.v @@ -0,0 +1,105 @@ +module main + +// SM3, the Chinese national cryptographic hash (GB/T 32905-2016). GNU's cksum +// accepts `-a sm3`, and V's standard library has no SM3, so it is implemented +// here. Verified against GNU cksum's output. + +const sm3_iv = [u32(0x7380166f), 0x4914b2b9, 0x172442d7, 0xda8a0600, 0xa96f30bc, 0x163138aa, + 0xe38dee4d, 0xb0fb0e4e] + +fn sm3_rotl(x u32, n u32) u32 { + return (x << n) | (x >> (32 - n)) +} + +fn sm3_p0(x u32) u32 { + return x ^ sm3_rotl(x, 9) ^ sm3_rotl(x, 17) +} + +fn sm3_p1(x u32) u32 { + return x ^ sm3_rotl(x, 15) ^ sm3_rotl(x, 23) +} + +fn sm3_ff(j int, x u32, y u32, z u32) u32 { + return if j < 16 { x ^ y ^ z } else { (x & y) | (x & z) | (y & z) } +} + +fn sm3_gg(j int, x u32, y u32, z u32) u32 { + return if j < 16 { x ^ y ^ z } else { (x & y) | (~x & z) } +} + +fn sm3_compress(mut v []u32, block []u8) { + // w holds 68 words, of which only 0..15 and 64..67 are used directly; the + // rest is the expanded message schedule. + mut w := []u32{len: 68} + for j in 0 .. 16 { + w[j] = be_u32(block, j * 4) + } + for j in 16 .. 68 { + w[j] = sm3_p1(w[j - 16] ^ w[j - 9] ^ sm3_rotl(w[j - 3], 15)) ^ sm3_rotl(w[j - 13], 7) ^ + w[j - 6] + } + + mut a := v[0] + mut b := v[1] + mut c := v[2] + mut d := v[3] + mut e := v[4] + mut f := v[5] + mut g := v[6] + mut h := v[7] + + for j in 0 .. 64 { + t := if j < 16 { u32(0x79cc4519) } else { u32(0x7a879d8a) } + a12 := sm3_rotl(a, 12) + ss1 := sm3_rotl(a12 + e + sm3_rotl(t, u32(j) % 32), 7) + ss2 := ss1 ^ a12 + // TT1 takes the expanded word W'[j] = W[j] ^ W[j+4]; TT2 takes W[j]. + tt1 := sm3_ff(j, a, b, c) + d + ss2 + (w[j] ^ w[j + 4]) + tt2 := sm3_gg(j, e, f, g) + h + ss1 + w[j] + d = c + c = sm3_rotl(b, 9) + b = a + a = tt1 + h = g + g = sm3_rotl(f, 19) + f = e + e = sm3_p0(tt2) + } + + v[0] ^= a + v[1] ^= b + v[2] ^= c + v[3] ^= d + v[4] ^= e + v[5] ^= f + v[6] ^= g + v[7] ^= h +} + +fn sm3_sum(data []u8) []u8 { + // Pad the same way SHA-256 does: 0x80, zeroes up to 56 bytes mod 64, then + // the message length in bits as a big endian u64. + mut msg := data.clone() + msg << 0x80 + for msg.len % 64 != 56 { + msg << 0 + } + bits := u64(data.len) * 8 + for i in 0 .. 8 { + msg << u8(bits >> u32(56 - 8 * i)) + } + + mut v := sm3_iv.clone() + for i in 0 .. msg.len / 64 { + sm3_compress(mut v, msg[i * 64..i * 64 + 64]) + } + + mut res := []u8{len: 32} + for i in 0 .. 8 { + res[i * 4] = u8(v[i] >> 24) + res[i * 4 + 1] = u8(v[i] >> 16) + res[i * 4 + 2] = u8(v[i] >> 8) + res[i * 4 + 3] = u8(v[i]) + } + return res +}