You can not select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
dateKeys-dart/test/bytes_test.dart

270 lines
8.9 KiB

// The byte helpers: hexadecimal, comparison, UTF-8 as strict as Go's, Go's
// utf8.DecodeRune and Go's %q, as bytes.test.ts of datekeys-ts. The texts of
// %q are those printed by Go 1.26.
import 'dart:convert';
import 'dart:typed_data';
import 'package:crypto/crypto.dart';
import 'package:datekeys/src/bytes.dart';
import 'package:test/test.dart';
// One backslash, for the texts of Go's %q.
const bs = '\\';
String cp(List<int> runes) => String.fromCharCodes(runes);
void main() {
test('converts hexadecimal', () {
expect(toHex([0, 1, 0xab, 0xff]), '0001abff');
expect(toHex([]), '');
expect(fromHex('0001ABff'), [0, 1, 0xab, 0xff]);
expect(fromHex(''), isEmpty);
expect(() => fromHex('abc'), throwsFormatException);
expect(() => fromHex('zz'), throwsFormatException);
expect(() => fromHex('0g'), throwsFormatException);
expect(() => fromHex('${cp([0xff10])}0'), throwsFormatException);
});
test('compares and joins byte strings', () {
final a = [1, 2];
expect(equalBytes(a, [1, 2]), isTrue);
expect(equalBytes(a, [1, 3]), isFalse);
expect(equalBytes(a, [1]), isFalse);
expect(compareBytes(a, [1, 2]), 0);
expect(compareBytes(a, [1, 3]), -1);
expect(compareBytes([2], a), 1);
expect(compareBytes([1], a), -1);
expect(compareBytes(a, [1]), 1);
expect(compareBytes([0xff], [0x00, 0x00]), 1);
expect(
concatBytes([
a,
<int>[],
Uint8List.fromList([3]),
]),
[1, 2, 3],
);
expect(concatBytes([]), isEmpty);
});
group('UTF-8', () {
// Each byte string that UTF-8 forbids, and what Go's utf8.Valid refuses.
final invalid = <String, List<int>>{
'an invalid byte': [0x61, 0xff],
'a byte above F4': [0xf5, 0x80, 0x80, 0x80],
'an overlong NUL': [0xc0, 0x80],
'an overlong three-byte form': [0xe0, 0x80, 0x80],
'an overlong four-byte form': [0xf0, 0x80, 0x80, 0x80],
'an encoded high surrogate': [0xed, 0xa0, 0x80],
'an encoded low surrogate': [0xed, 0xbf, 0xbf],
'a code point above U+10FFFF': [0xf4, 0x90, 0x80, 0x80],
'a truncated sequence': [0xe2, 0x82],
'a lone continuation byte': [0x80],
};
test('dart:convert refuses what Go refuses, and so does decodeUtf8', () {
for (final MapEntry(key: name, value: b) in invalid.entries) {
// Short inputs and long ones: dart2js decodes the long ones with the
// TextDecoder of the platform.
for (final input in [
b,
[...b, ...List.filled(40, 0x61)],
]) {
expect(
() => const Utf8Decoder().convert(input),
throwsFormatException,
reason: name,
);
expect(decodeUtf8(input), isNull, reason: name);
expect(isValidUtf8(input), isFalse, reason: name);
}
}
});
test('dart:convert drops a leading U+FEFF, and decodeUtf8 keeps it', () {
// The reason decodeUtf8 does not use Utf8Decoder: a text that starts
// with U+FEFF would decode as one that does not, on the VM and on the
// web alike.
for (final rest in [
[0x61],
List.filled(40, 0x61),
]) {
final input = [0xef, 0xbb, 0xbf, ...rest];
expect(const Utf8Decoder().convert(input), cp(rest));
expect(decodeUtf8(input), cp([0xfeff, ...rest]));
}
// Further ones are characters for both.
expect(
decodeUtf8([0x61, 0xef, 0xbb, 0xbf, 0xef, 0xbb, 0xbf]),
cp([0x61, 0xfeff, 0xfeff]),
);
expect(
decodeUtf8([0xef, 0xbb, 0xbf, 0xef, 0xbb, 0xbf]),
cp([0xfeff, 0xfeff]),
);
});
test('decodes every valid sequence, U+FFFD and U+10FFFF included', () {
final valid = <List<int>, String>{
[]: '',
[0x00]: cp([0]),
[0x7f]: cp([0x7f]),
[0xc2, 0x80]: cp([0x80]),
[0xdf, 0xbf]: cp([0x7ff]),
[0xe0, 0xa0, 0x80]: cp([0x800]),
[0xed, 0x9f, 0xbf]: cp([0xd7ff]),
[0xee, 0x80, 0x80]: cp([0xe000]),
[0xef, 0xbf, 0xbd]: cp([0xfffd]),
[0xf0, 0x90, 0x80, 0x80]: cp([0x10000]),
[0xf4, 0x8f, 0xbf, 0xbf]: cp([0x10ffff]),
};
for (final MapEntry(key: b, value: s) in valid.entries) {
expect(decodeUtf8(b), s, reason: toHex(b));
expect(isValidUtf8(b), isTrue, reason: toHex(b));
expect(utf8Bytes(s), b, reason: toHex(b));
}
});
test('writes a well-formed string as dart:convert does', () {
final s = cp([0xfeff, 0x61, 0xe9, 0x20ac, 0x10000, 0x1f600, 0x10ffff]);
expect(utf8Bytes(s), utf8.encode(s));
expect(isWellFormedUtf16(s), isTrue);
});
test(
'writes a lone surrogate in generalized UTF-8, which is not UTF-8',
() {
// dart:convert writes U+FFFD in its place: the string would encode as
// another.
expect(utf8.encode(cp([0xd800])), [0xef, 0xbf, 0xbd]);
final cases = <List<int>, String>{
[0xd800]: 'eda080',
[0xdfff]: 'edbfbf',
[0x61, 0xdc00, 0xd800]: '61edb080eda080',
[0xd800, 0xd800, 0xdc00]: 'eda080f0908080',
[0xdbff, 0x61]: 'edafbf61',
};
for (final MapEntry(key: units, value: hex) in cases.entries) {
final s = cp(units);
expect(toHex(utf8Bytes(s)), hex);
expect(isWellFormedUtf16(s), isFalse);
expect(isValidUtf8(utf8Bytes(s)), isFalse);
}
},
);
test('decodes runes as Go utf8.DecodeRune', () {
final b = [
0x41, 0xc3, 0xa9, 0xe2, 0x82, 0xac, 0xf0, 0x9f, //
0x98, 0x80, 0xc0, 0x80, 0xe0, 0x80, 0xf5,
];
expect(decodeRune(b, 0), (0x41, 1));
expect(decodeRune(b, 1), (0xe9, 2));
expect(decodeRune(b, 3), (0x20ac, 3));
expect(decodeRune(b, 6), (0x1f600, 4));
for (final i in [7, 10, 11, 12, 13, 14]) {
expect(decodeRune(b, i), (0xfffd, 1), reason: '$i');
}
expect(decodeRune(b, 15), (0xfffd, 0));
expect(decodeRune([0xf0, 0x9f, 0x98], 0), (0xfffd, 1));
expect(decodeRune([0xc3], 0), (0xfffd, 1));
expect(decodeRune([0xef, 0xbf, 0xbd], 0), (0xfffd, 3));
expect(decodeRune([], 0), (0xfffd, 0));
});
});
group("Go's %q", () {
test('quotes as strconv.Quote', () {
expect(
goQuote([0x61, 0x00, 0x62, 0xff, 0xc3, 0xa9, 0xe2, 0x80, 0xa8]),
'"a${bs}x00b${bs}xff${cp([0xe9])}${bs}u2028"',
);
expect(
goQuote(utf8Bytes(cp([7, 8, 12, 10, 13, 9, 11]))),
'"${bs}a${bs}b${bs}f${bs}n${bs}r${bs}t${bs}v"',
);
expect(goQuote(utf8Bytes('"$bs')), '"$bs"$bs$bs"');
expect(
goQuote(utf8Bytes(cp([0x10000, 0x1f600, 0x301]))),
'"${cp([0x10000, 0x1f600, 0x301])}"',
);
expect(
goQuote(utf8Bytes(cp([0xa0, 0xfeff, 0xe000, 0x10ffff, 0x7f]))),
'"${bs}u00a0${bs}ufeff${bs}ue000${bs}U0010ffff${bs}x7f"',
);
expect(
goQuote([0xed, 0xa0, 0x80, 0xf4, 0x90, 0x80, 0x80, 0xe2, 0x82]),
'"${bs}xed${bs}xa0${bs}x80${bs}xf4${bs}x90${bs}x80${bs}x80${bs}xe2'
'${bs}x82"',
);
expect(goQuote(utf8Bytes(cp([0xfffd]))), '"${cp([0xfffd])}"');
expect(goQuote(utf8Bytes(cp([0x378]))), '"${bs}u0378"');
expect(goQuote([]), '""');
// New in Unicode 16, which the platform may know and Go 1.26 does not.
expect(
goQuote(utf8Bytes(cp([0x31e4, 0x31e3, 0x1fbcb, 0x1fbca]))),
'"${bs}u31e4${cp([0x31e3])}${bs}U0001fbcb${cp([0x1fbca])}"',
);
// A lone surrogate, as its generalized UTF-8.
expect(
goQuote(utf8Bytes(cp([0x61, 0xd800]))),
'"a${bs}xed${bs}xa0${bs}x80"',
);
});
test("classifies every rune as Go 1.26's strconv.IsPrint", () {
// The IsPrint bit set of all runes (bit r at byte r >> 3, mask
// 1 << (r & 7)) and its SHA-256, computed by Go 1.26.8 (Unicode
// 15.0.0), as bytes.test.ts pins it.
final bits = Uint8List(0x110000 ~/ 8);
var runes = 0;
var ranges = 0;
var previous = false;
for (var r = 0; r < 0x110000; r++) {
final p = isPrint(r);
if (p) {
bits[r >> 3] |= 1 << (r & 7);
if (r >= 0x80) {
runes++;
if (!previous) ranges++;
}
}
previous = p && r >= 0x80;
}
expect((runes, ranges), (148903, 710));
expect(
sha256.convert(bits).toString(),
'dbe4beecc0027b38c16c32d304958e611e27baa40412efea4cf616346eb631a2',
);
for (final r in [
0x80,
0x9f,
0xa0,
0xad,
0x31e4,
0xe000,
0xfeff,
0x10ffff,
]) {
expect(isPrint(r), isFalse, reason: r.toRadixString(16));
}
for (final r in [
0x20,
0x7e,
0xa1,
0xe9,
0x31e3,
0xfffd,
0x1f600,
0xe0100,
]) {
expect(isPrint(r), isTrue, reason: r.toRadixString(16));
}
for (final r in [0x00, 0x1f, 0x7f]) {
expect(isPrint(r), isFalse, reason: r.toRadixString(16));
}
});
});
}

Powered by TurnKey Linux.