You can not select more than 25 topics
Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
270 lines
8.9 KiB
270 lines
8.9 KiB
// The byte helpers: hexadecimal, comparison, UTF-8 as strict as Go's, Go's
|
|
// utf8.DecodeRune and Go's %q, as bytes.test.ts of datekeys-ts. The texts of
|
|
// %q are those printed by Go 1.26.
|
|
|
|
import 'dart:convert';
|
|
import 'dart:typed_data';
|
|
|
|
import 'package:crypto/crypto.dart';
|
|
import 'package:datekeys/src/bytes.dart';
|
|
import 'package:test/test.dart';
|
|
|
|
// One backslash, for the texts of Go's %q.
|
|
const bs = '\\';
|
|
|
|
String cp(List<int> runes) => String.fromCharCodes(runes);
|
|
|
|
void main() {
|
|
test('converts hexadecimal', () {
|
|
expect(toHex([0, 1, 0xab, 0xff]), '0001abff');
|
|
expect(toHex([]), '');
|
|
expect(fromHex('0001ABff'), [0, 1, 0xab, 0xff]);
|
|
expect(fromHex(''), isEmpty);
|
|
expect(() => fromHex('abc'), throwsFormatException);
|
|
expect(() => fromHex('zz'), throwsFormatException);
|
|
expect(() => fromHex('0g'), throwsFormatException);
|
|
expect(() => fromHex('${cp([0xff10])}0'), throwsFormatException);
|
|
});
|
|
|
|
test('compares and joins byte strings', () {
|
|
final a = [1, 2];
|
|
expect(equalBytes(a, [1, 2]), isTrue);
|
|
expect(equalBytes(a, [1, 3]), isFalse);
|
|
expect(equalBytes(a, [1]), isFalse);
|
|
expect(compareBytes(a, [1, 2]), 0);
|
|
expect(compareBytes(a, [1, 3]), -1);
|
|
expect(compareBytes([2], a), 1);
|
|
expect(compareBytes([1], a), -1);
|
|
expect(compareBytes(a, [1]), 1);
|
|
expect(compareBytes([0xff], [0x00, 0x00]), 1);
|
|
expect(
|
|
concatBytes([
|
|
a,
|
|
<int>[],
|
|
Uint8List.fromList([3]),
|
|
]),
|
|
[1, 2, 3],
|
|
);
|
|
expect(concatBytes([]), isEmpty);
|
|
});
|
|
|
|
group('UTF-8', () {
|
|
// Each byte string that UTF-8 forbids, and what Go's utf8.Valid refuses.
|
|
final invalid = <String, List<int>>{
|
|
'an invalid byte': [0x61, 0xff],
|
|
'a byte above F4': [0xf5, 0x80, 0x80, 0x80],
|
|
'an overlong NUL': [0xc0, 0x80],
|
|
'an overlong three-byte form': [0xe0, 0x80, 0x80],
|
|
'an overlong four-byte form': [0xf0, 0x80, 0x80, 0x80],
|
|
'an encoded high surrogate': [0xed, 0xa0, 0x80],
|
|
'an encoded low surrogate': [0xed, 0xbf, 0xbf],
|
|
'a code point above U+10FFFF': [0xf4, 0x90, 0x80, 0x80],
|
|
'a truncated sequence': [0xe2, 0x82],
|
|
'a lone continuation byte': [0x80],
|
|
};
|
|
|
|
test('dart:convert refuses what Go refuses, and so does decodeUtf8', () {
|
|
for (final MapEntry(key: name, value: b) in invalid.entries) {
|
|
// Short inputs and long ones: dart2js decodes the long ones with the
|
|
// TextDecoder of the platform.
|
|
for (final input in [
|
|
b,
|
|
[...b, ...List.filled(40, 0x61)],
|
|
]) {
|
|
expect(
|
|
() => const Utf8Decoder().convert(input),
|
|
throwsFormatException,
|
|
reason: name,
|
|
);
|
|
expect(decodeUtf8(input), isNull, reason: name);
|
|
expect(isValidUtf8(input), isFalse, reason: name);
|
|
}
|
|
}
|
|
});
|
|
|
|
test('dart:convert drops a leading U+FEFF, and decodeUtf8 keeps it', () {
|
|
// The reason decodeUtf8 does not use Utf8Decoder: a text that starts
|
|
// with U+FEFF would decode as one that does not, on the VM and on the
|
|
// web alike.
|
|
for (final rest in [
|
|
[0x61],
|
|
List.filled(40, 0x61),
|
|
]) {
|
|
final input = [0xef, 0xbb, 0xbf, ...rest];
|
|
expect(const Utf8Decoder().convert(input), cp(rest));
|
|
expect(decodeUtf8(input), cp([0xfeff, ...rest]));
|
|
}
|
|
// Further ones are characters for both.
|
|
expect(
|
|
decodeUtf8([0x61, 0xef, 0xbb, 0xbf, 0xef, 0xbb, 0xbf]),
|
|
cp([0x61, 0xfeff, 0xfeff]),
|
|
);
|
|
expect(
|
|
decodeUtf8([0xef, 0xbb, 0xbf, 0xef, 0xbb, 0xbf]),
|
|
cp([0xfeff, 0xfeff]),
|
|
);
|
|
});
|
|
|
|
test('decodes every valid sequence, U+FFFD and U+10FFFF included', () {
|
|
final valid = <List<int>, String>{
|
|
[]: '',
|
|
[0x00]: cp([0]),
|
|
[0x7f]: cp([0x7f]),
|
|
[0xc2, 0x80]: cp([0x80]),
|
|
[0xdf, 0xbf]: cp([0x7ff]),
|
|
[0xe0, 0xa0, 0x80]: cp([0x800]),
|
|
[0xed, 0x9f, 0xbf]: cp([0xd7ff]),
|
|
[0xee, 0x80, 0x80]: cp([0xe000]),
|
|
[0xef, 0xbf, 0xbd]: cp([0xfffd]),
|
|
[0xf0, 0x90, 0x80, 0x80]: cp([0x10000]),
|
|
[0xf4, 0x8f, 0xbf, 0xbf]: cp([0x10ffff]),
|
|
};
|
|
for (final MapEntry(key: b, value: s) in valid.entries) {
|
|
expect(decodeUtf8(b), s, reason: toHex(b));
|
|
expect(isValidUtf8(b), isTrue, reason: toHex(b));
|
|
expect(utf8Bytes(s), b, reason: toHex(b));
|
|
}
|
|
});
|
|
|
|
test('writes a well-formed string as dart:convert does', () {
|
|
final s = cp([0xfeff, 0x61, 0xe9, 0x20ac, 0x10000, 0x1f600, 0x10ffff]);
|
|
expect(utf8Bytes(s), utf8.encode(s));
|
|
expect(isWellFormedUtf16(s), isTrue);
|
|
});
|
|
|
|
test(
|
|
'writes a lone surrogate in generalized UTF-8, which is not UTF-8',
|
|
() {
|
|
// dart:convert writes U+FFFD in its place: the string would encode as
|
|
// another.
|
|
expect(utf8.encode(cp([0xd800])), [0xef, 0xbf, 0xbd]);
|
|
final cases = <List<int>, String>{
|
|
[0xd800]: 'eda080',
|
|
[0xdfff]: 'edbfbf',
|
|
[0x61, 0xdc00, 0xd800]: '61edb080eda080',
|
|
[0xd800, 0xd800, 0xdc00]: 'eda080f0908080',
|
|
[0xdbff, 0x61]: 'edafbf61',
|
|
};
|
|
for (final MapEntry(key: units, value: hex) in cases.entries) {
|
|
final s = cp(units);
|
|
expect(toHex(utf8Bytes(s)), hex);
|
|
expect(isWellFormedUtf16(s), isFalse);
|
|
expect(isValidUtf8(utf8Bytes(s)), isFalse);
|
|
}
|
|
},
|
|
);
|
|
|
|
test('decodes runes as Go utf8.DecodeRune', () {
|
|
final b = [
|
|
0x41, 0xc3, 0xa9, 0xe2, 0x82, 0xac, 0xf0, 0x9f, //
|
|
0x98, 0x80, 0xc0, 0x80, 0xe0, 0x80, 0xf5,
|
|
];
|
|
expect(decodeRune(b, 0), (0x41, 1));
|
|
expect(decodeRune(b, 1), (0xe9, 2));
|
|
expect(decodeRune(b, 3), (0x20ac, 3));
|
|
expect(decodeRune(b, 6), (0x1f600, 4));
|
|
for (final i in [7, 10, 11, 12, 13, 14]) {
|
|
expect(decodeRune(b, i), (0xfffd, 1), reason: '$i');
|
|
}
|
|
expect(decodeRune(b, 15), (0xfffd, 0));
|
|
expect(decodeRune([0xf0, 0x9f, 0x98], 0), (0xfffd, 1));
|
|
expect(decodeRune([0xc3], 0), (0xfffd, 1));
|
|
expect(decodeRune([0xef, 0xbf, 0xbd], 0), (0xfffd, 3));
|
|
expect(decodeRune([], 0), (0xfffd, 0));
|
|
});
|
|
});
|
|
|
|
group("Go's %q", () {
|
|
test('quotes as strconv.Quote', () {
|
|
expect(
|
|
goQuote([0x61, 0x00, 0x62, 0xff, 0xc3, 0xa9, 0xe2, 0x80, 0xa8]),
|
|
'"a${bs}x00b${bs}xff${cp([0xe9])}${bs}u2028"',
|
|
);
|
|
expect(
|
|
goQuote(utf8Bytes(cp([7, 8, 12, 10, 13, 9, 11]))),
|
|
'"${bs}a${bs}b${bs}f${bs}n${bs}r${bs}t${bs}v"',
|
|
);
|
|
expect(goQuote(utf8Bytes('"$bs')), '"$bs"$bs$bs"');
|
|
expect(
|
|
goQuote(utf8Bytes(cp([0x10000, 0x1f600, 0x301]))),
|
|
'"${cp([0x10000, 0x1f600, 0x301])}"',
|
|
);
|
|
expect(
|
|
goQuote(utf8Bytes(cp([0xa0, 0xfeff, 0xe000, 0x10ffff, 0x7f]))),
|
|
'"${bs}u00a0${bs}ufeff${bs}ue000${bs}U0010ffff${bs}x7f"',
|
|
);
|
|
expect(
|
|
goQuote([0xed, 0xa0, 0x80, 0xf4, 0x90, 0x80, 0x80, 0xe2, 0x82]),
|
|
'"${bs}xed${bs}xa0${bs}x80${bs}xf4${bs}x90${bs}x80${bs}x80${bs}xe2'
|
|
'${bs}x82"',
|
|
);
|
|
expect(goQuote(utf8Bytes(cp([0xfffd]))), '"${cp([0xfffd])}"');
|
|
expect(goQuote(utf8Bytes(cp([0x378]))), '"${bs}u0378"');
|
|
expect(goQuote([]), '""');
|
|
// New in Unicode 16, which the platform may know and Go 1.26 does not.
|
|
expect(
|
|
goQuote(utf8Bytes(cp([0x31e4, 0x31e3, 0x1fbcb, 0x1fbca]))),
|
|
'"${bs}u31e4${cp([0x31e3])}${bs}U0001fbcb${cp([0x1fbca])}"',
|
|
);
|
|
// A lone surrogate, as its generalized UTF-8.
|
|
expect(
|
|
goQuote(utf8Bytes(cp([0x61, 0xd800]))),
|
|
'"a${bs}xed${bs}xa0${bs}x80"',
|
|
);
|
|
});
|
|
|
|
test("classifies every rune as Go 1.26's strconv.IsPrint", () {
|
|
// The IsPrint bit set of all runes (bit r at byte r >> 3, mask
|
|
// 1 << (r & 7)) and its SHA-256, computed by Go 1.26.8 (Unicode
|
|
// 15.0.0), as bytes.test.ts pins it.
|
|
final bits = Uint8List(0x110000 ~/ 8);
|
|
var runes = 0;
|
|
var ranges = 0;
|
|
var previous = false;
|
|
for (var r = 0; r < 0x110000; r++) {
|
|
final p = isPrint(r);
|
|
if (p) {
|
|
bits[r >> 3] |= 1 << (r & 7);
|
|
if (r >= 0x80) {
|
|
runes++;
|
|
if (!previous) ranges++;
|
|
}
|
|
}
|
|
previous = p && r >= 0x80;
|
|
}
|
|
expect((runes, ranges), (148903, 710));
|
|
expect(
|
|
sha256.convert(bits).toString(),
|
|
'dbe4beecc0027b38c16c32d304958e611e27baa40412efea4cf616346eb631a2',
|
|
);
|
|
for (final r in [
|
|
0x80,
|
|
0x9f,
|
|
0xa0,
|
|
0xad,
|
|
0x31e4,
|
|
0xe000,
|
|
0xfeff,
|
|
0x10ffff,
|
|
]) {
|
|
expect(isPrint(r), isFalse, reason: r.toRadixString(16));
|
|
}
|
|
for (final r in [
|
|
0x20,
|
|
0x7e,
|
|
0xa1,
|
|
0xe9,
|
|
0x31e3,
|
|
0xfffd,
|
|
0x1f600,
|
|
0xe0100,
|
|
]) {
|
|
expect(isPrint(r), isTrue, reason: r.toRadixString(16));
|
|
}
|
|
for (final r in [0x00, 0x1f, 0x7f]) {
|
|
expect(isPrint(r), isFalse, reason: r.toRadixString(16));
|
|
}
|
|
});
|
|
});
|
|
}
|