// The byte helpers: hexadecimal, comparison, UTF-8 as strict as Go's, Go's // utf8.DecodeRune and Go's %q, as bytes.test.ts of datekeys-ts. The texts of // %q are those printed by Go 1.26. import 'dart:convert'; import 'dart:typed_data'; import 'package:crypto/crypto.dart'; import 'package:datekeys/src/bytes.dart'; import 'package:test/test.dart'; // One backslash, for the texts of Go's %q. const bs = '\\'; String cp(List runes) => String.fromCharCodes(runes); void main() { test('converts hexadecimal', () { expect(toHex([0, 1, 0xab, 0xff]), '0001abff'); expect(toHex([]), ''); expect(fromHex('0001ABff'), [0, 1, 0xab, 0xff]); expect(fromHex(''), isEmpty); expect(() => fromHex('abc'), throwsFormatException); expect(() => fromHex('zz'), throwsFormatException); expect(() => fromHex('0g'), throwsFormatException); expect(() => fromHex('${cp([0xff10])}0'), throwsFormatException); }); test('compares and joins byte strings', () { final a = [1, 2]; expect(equalBytes(a, [1, 2]), isTrue); expect(equalBytes(a, [1, 3]), isFalse); expect(equalBytes(a, [1]), isFalse); expect(compareBytes(a, [1, 2]), 0); expect(compareBytes(a, [1, 3]), -1); expect(compareBytes([2], a), 1); expect(compareBytes([1], a), -1); expect(compareBytes(a, [1]), 1); expect(compareBytes([0xff], [0x00, 0x00]), 1); expect( concatBytes([ a, [], Uint8List.fromList([3]), ]), [1, 2, 3], ); expect(concatBytes([]), isEmpty); }); group('UTF-8', () { // Each byte string that UTF-8 forbids, and what Go's utf8.Valid refuses. final invalid = >{ 'an invalid byte': [0x61, 0xff], 'a byte above F4': [0xf5, 0x80, 0x80, 0x80], 'an overlong NUL': [0xc0, 0x80], 'an overlong three-byte form': [0xe0, 0x80, 0x80], 'an overlong four-byte form': [0xf0, 0x80, 0x80, 0x80], 'an encoded high surrogate': [0xed, 0xa0, 0x80], 'an encoded low surrogate': [0xed, 0xbf, 0xbf], 'a code point above U+10FFFF': [0xf4, 0x90, 0x80, 0x80], 'a truncated sequence': [0xe2, 0x82], 'a lone continuation byte': [0x80], }; test('dart:convert refuses what Go refuses, and so does decodeUtf8', () { for (final MapEntry(key: name, value: b) in invalid.entries) { // Short inputs and long ones: dart2js decodes the long ones with the // TextDecoder of the platform. for (final input in [ b, [...b, ...List.filled(40, 0x61)], ]) { expect( () => const Utf8Decoder().convert(input), throwsFormatException, reason: name, ); expect(decodeUtf8(input), isNull, reason: name); expect(isValidUtf8(input), isFalse, reason: name); } } }); test('dart:convert drops a leading U+FEFF, and decodeUtf8 keeps it', () { // The reason decodeUtf8 does not use Utf8Decoder: a text that starts // with U+FEFF would decode as one that does not, on the VM and on the // web alike. for (final rest in [ [0x61], List.filled(40, 0x61), ]) { final input = [0xef, 0xbb, 0xbf, ...rest]; expect(const Utf8Decoder().convert(input), cp(rest)); expect(decodeUtf8(input), cp([0xfeff, ...rest])); } // Further ones are characters for both. expect( decodeUtf8([0x61, 0xef, 0xbb, 0xbf, 0xef, 0xbb, 0xbf]), cp([0x61, 0xfeff, 0xfeff]), ); expect( decodeUtf8([0xef, 0xbb, 0xbf, 0xef, 0xbb, 0xbf]), cp([0xfeff, 0xfeff]), ); }); test('decodes every valid sequence, U+FFFD and U+10FFFF included', () { final valid = , String>{ []: '', [0x00]: cp([0]), [0x7f]: cp([0x7f]), [0xc2, 0x80]: cp([0x80]), [0xdf, 0xbf]: cp([0x7ff]), [0xe0, 0xa0, 0x80]: cp([0x800]), [0xed, 0x9f, 0xbf]: cp([0xd7ff]), [0xee, 0x80, 0x80]: cp([0xe000]), [0xef, 0xbf, 0xbd]: cp([0xfffd]), [0xf0, 0x90, 0x80, 0x80]: cp([0x10000]), [0xf4, 0x8f, 0xbf, 0xbf]: cp([0x10ffff]), }; for (final MapEntry(key: b, value: s) in valid.entries) { expect(decodeUtf8(b), s, reason: toHex(b)); expect(isValidUtf8(b), isTrue, reason: toHex(b)); expect(utf8Bytes(s), b, reason: toHex(b)); } }); test('writes a well-formed string as dart:convert does', () { final s = cp([0xfeff, 0x61, 0xe9, 0x20ac, 0x10000, 0x1f600, 0x10ffff]); expect(utf8Bytes(s), utf8.encode(s)); expect(isWellFormedUtf16(s), isTrue); }); test( 'writes a lone surrogate in generalized UTF-8, which is not UTF-8', () { // dart:convert writes U+FFFD in its place: the string would encode as // another. expect(utf8.encode(cp([0xd800])), [0xef, 0xbf, 0xbd]); final cases = , String>{ [0xd800]: 'eda080', [0xdfff]: 'edbfbf', [0x61, 0xdc00, 0xd800]: '61edb080eda080', [0xd800, 0xd800, 0xdc00]: 'eda080f0908080', [0xdbff, 0x61]: 'edafbf61', }; for (final MapEntry(key: units, value: hex) in cases.entries) { final s = cp(units); expect(toHex(utf8Bytes(s)), hex); expect(isWellFormedUtf16(s), isFalse); expect(isValidUtf8(utf8Bytes(s)), isFalse); } }, ); test('decodes runes as Go utf8.DecodeRune', () { final b = [ 0x41, 0xc3, 0xa9, 0xe2, 0x82, 0xac, 0xf0, 0x9f, // 0x98, 0x80, 0xc0, 0x80, 0xe0, 0x80, 0xf5, ]; expect(decodeRune(b, 0), (0x41, 1)); expect(decodeRune(b, 1), (0xe9, 2)); expect(decodeRune(b, 3), (0x20ac, 3)); expect(decodeRune(b, 6), (0x1f600, 4)); for (final i in [7, 10, 11, 12, 13, 14]) { expect(decodeRune(b, i), (0xfffd, 1), reason: '$i'); } expect(decodeRune(b, 15), (0xfffd, 0)); expect(decodeRune([0xf0, 0x9f, 0x98], 0), (0xfffd, 1)); expect(decodeRune([0xc3], 0), (0xfffd, 1)); expect(decodeRune([0xef, 0xbf, 0xbd], 0), (0xfffd, 3)); expect(decodeRune([], 0), (0xfffd, 0)); }); }); group("Go's %q", () { test('quotes as strconv.Quote', () { expect( goQuote([0x61, 0x00, 0x62, 0xff, 0xc3, 0xa9, 0xe2, 0x80, 0xa8]), '"a${bs}x00b${bs}xff${cp([0xe9])}${bs}u2028"', ); expect( goQuote(utf8Bytes(cp([7, 8, 12, 10, 13, 9, 11]))), '"${bs}a${bs}b${bs}f${bs}n${bs}r${bs}t${bs}v"', ); expect(goQuote(utf8Bytes('"$bs')), '"$bs"$bs$bs"'); expect( goQuote(utf8Bytes(cp([0x10000, 0x1f600, 0x301]))), '"${cp([0x10000, 0x1f600, 0x301])}"', ); expect( goQuote(utf8Bytes(cp([0xa0, 0xfeff, 0xe000, 0x10ffff, 0x7f]))), '"${bs}u00a0${bs}ufeff${bs}ue000${bs}U0010ffff${bs}x7f"', ); expect( goQuote([0xed, 0xa0, 0x80, 0xf4, 0x90, 0x80, 0x80, 0xe2, 0x82]), '"${bs}xed${bs}xa0${bs}x80${bs}xf4${bs}x90${bs}x80${bs}x80${bs}xe2' '${bs}x82"', ); expect(goQuote(utf8Bytes(cp([0xfffd]))), '"${cp([0xfffd])}"'); expect(goQuote(utf8Bytes(cp([0x378]))), '"${bs}u0378"'); expect(goQuote([]), '""'); // New in Unicode 16, which the platform may know and Go 1.26 does not. expect( goQuote(utf8Bytes(cp([0x31e4, 0x31e3, 0x1fbcb, 0x1fbca]))), '"${bs}u31e4${cp([0x31e3])}${bs}U0001fbcb${cp([0x1fbca])}"', ); // A lone surrogate, as its generalized UTF-8. expect( goQuote(utf8Bytes(cp([0x61, 0xd800]))), '"a${bs}xed${bs}xa0${bs}x80"', ); }); test("classifies every rune as Go 1.26's strconv.IsPrint", () { // The IsPrint bit set of all runes (bit r at byte r >> 3, mask // 1 << (r & 7)) and its SHA-256, computed by Go 1.26.8 (Unicode // 15.0.0), as bytes.test.ts pins it. final bits = Uint8List(0x110000 ~/ 8); var runes = 0; var ranges = 0; var previous = false; for (var r = 0; r < 0x110000; r++) { final p = isPrint(r); if (p) { bits[r >> 3] |= 1 << (r & 7); if (r >= 0x80) { runes++; if (!previous) ranges++; } } previous = p && r >= 0x80; } expect((runes, ranges), (148903, 710)); expect( sha256.convert(bits).toString(), 'dbe4beecc0027b38c16c32d304958e611e27baa40412efea4cf616346eb631a2', ); for (final r in [ 0x80, 0x9f, 0xa0, 0xad, 0x31e4, 0xe000, 0xfeff, 0x10ffff, ]) { expect(isPrint(r), isFalse, reason: r.toRadixString(16)); } for (final r in [ 0x20, 0x7e, 0xa1, 0xe9, 0x31e3, 0xfffd, 0x1f600, 0xe0100, ]) { expect(isPrint(r), isTrue, reason: r.toRadixString(16)); } for (final r in [0x00, 0x1f, 0x7f]) { expect(isPrint(r), isFalse, reason: r.toRadixString(16)); } }); }); }