You can not select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
dateKeys-dart/tool/pathrule_go_vectors.go

1112 lines
36 KiB

//go:build ignore
// Writes test/vectors/pathrule_vectors.json and the same JSON as a Dart
// constant, pathrule_vectors.g.dart: the results of internal/pathrule of the
// Go reference (spec §29.5, §29.5.1, §29.6) for lib/src/pathrule.dart. Every
// expected value is computed here by the Go reference; none is written by
// hand.
//
// - strings: the cases of the tests of internal/pathrule and of
// pathrule.test.ts of datekeys-ts, edges of this port and both sides of
// each limit of R2, R3 and R6b; then strings drawn from a fixed seed,
// random and adversarial: combining marks in and out of canonical order,
// Hangul, the ignorables and the whitelist of R4 in every place, emoji
// with and without their selectors, the best-fit look-alikes of ASCII,
// device names and 8.3 aliases, segments near the limits of R2 and R3,
// texts of several lines, and bytes that are not valid UTF-8. Each with
// the result of CheckPath, CheckComment and CheckAuthor, and its NFD and
// Key. Fold maps each code point alone: the planes check it.
// - trees: sets of paths, those of TestCheckTree and of paths.json and
// others drawn from the seed among names that collide, each with the
// result of CheckTree and, for R7, the two paths it names; folders: the
// cases of R9, as a pattern of names, since they take 65 536 paths.
// - code_points: Assigned, DefaultIgnorable and Lower of code points
// drawn from the seed; and planes: for each plane and each function, the
// SHA-256 of one line per code point, the exhaustive comparison.
//
// Binary values are lower-case hexadecimal, and the JSON is ASCII, so that
// the Dart constant is too. The seed is fixed: every run writes the same
// bytes.
//
// internal/pathrule can be imported only from inside the tree of
// datekeys-go, so this program runs in an export of it, which it does not
// change, and never in the repository itself. From the root of datekeys-dart:
//
// commit=$(git -C ../datekeys-go rev-parse v0.12)
// out=$PWD/test/vectors
// tmp=$(mktemp -d)
// git -C ../datekeys-go archive "$commit" | tar -x -C "$tmp"
// cp tool/pathrule_go_vectors.go "$tmp"
// (cd "$tmp" && go run ./pathrule_go_vectors.go -source "$commit" -out "$out")
// rm -rf "$tmp"
package main
import (
"bytes"
"crypto/sha256"
"encoding/hex"
"encoding/json"
"errors"
"flag"
"fmt"
"hash"
"log"
"math/rand/v2"
"os"
"path/filepath"
"runtime"
"slices"
"strconv"
"strings"
"unicode/utf8"
"g.activething.com/go/DateKeys/internal/pathrule"
)
var rng = rand.New(rand.NewChaCha8(sha256.Sum256([]byte("datekeys-dart stage 4a: pathrule"))))
func h(s string) string { return hex.EncodeToString([]byte(s)) }
// result is "ok" or the text of err.
func result(err error) string {
if err == nil {
return "ok"
}
return err.Error()
}
// ---------------------------------------------------------------------------
// The tables, read back from pathrule.Canonical, the only exported view of
// them: they give the pools of adversarial code points.
type tables struct {
assigned, ignorable [][2]rune
marks []rune // a non-zero combining class
byClass map[int][]rune
classes []int
decomp, fold, lower []rune
vs15, vs16 []rune
bestFit []rune // every code point that some table maps to ASCII
bestFitTo map[byte][]rune
}
func hexRune(s string) rune {
v, err := strconv.ParseUint(s, 16, 32)
if err != nil {
log.Fatal(err)
}
return rune(v)
}
func readTables() *tables {
t := &tables{byClass: map[int][]rune{}, bestFitTo: map[byte][]rune{}}
seen := map[rune]bool{}
for _, line := range strings.Split(strings.TrimSuffix(pathrule.Canonical(), "\n"), "\n") {
f := strings.Fields(line)
switch f[0] {
case "assigned":
t.assigned = append(t.assigned, [2]rune{hexRune(f[1]), hexRune(f[2])})
case "ignorable":
t.ignorable = append(t.ignorable, [2]rune{hexRune(f[1]), hexRune(f[2])})
case "ccc":
r := hexRune(f[1])
c, _ := strconv.Atoi(f[2])
t.marks = append(t.marks, r)
if t.byClass[c] == nil {
t.classes = append(t.classes, c)
}
t.byClass[c] = append(t.byClass[c], r)
case "decomp":
t.decomp = append(t.decomp, hexRune(f[1]))
case "fold":
t.fold = append(t.fold, hexRune(f[1]))
case "lower":
t.lower = append(t.lower, hexRune(f[1]))
case "vs15":
t.vs15 = append(t.vs15, hexRune(f[1]))
case "vs16":
t.vs16 = append(t.vs16, hexRune(f[1]))
case "bestfit":
r := hexRune(f[2])
b, _ := strconv.ParseUint(f[3], 16, 8)
if !seen[r] {
seen[r] = true
t.bestFit = append(t.bestFit, r)
}
t.bestFitTo[byte(b)] = append(t.bestFitTo[byte(b)], r)
}
}
slices.Sort(t.classes)
slices.Sort(t.bestFit)
return t
}
var tb = readTables()
// ---------------------------------------------------------------------------
// Pools
func pick[T any](xs []T) T { return xs[rng.IntN(len(xs))] }
func chance(p float64) bool { return rng.Float64() < p }
func inRange(rs [][2]rune) rune {
x := pick(rs)
return x[0] + rune(rng.IntN(int(x[1]-x[0])+1))
}
func runesOf(lo, hi rune) []rune {
var out []rune
for r := lo; r <= hi; r++ {
out = append(out, r)
}
return out
}
const (
zwnj = "\u200c"
zwj = "\u200d"
vs15 = "\ufe0e"
vs16 = "\ufe0f"
)
var (
letters = strings.Split("abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ", "")
digits = strings.Split("0123456789", "")
punct = strings.Split(" .~-_$#'()+,;=@[]^`{}!%&", "")
forbidden = strings.Split(`"*:<>?\|`, "")
controls = []string{"\x00", "\x01", "\t", "\n", "\r", "\x1b", "\x1f", "\x7f", "\u0080", "\u0085", "\u009f"}
bidi = []string{"\u202a", "\u202b", "\u202c", "\u202d", "\u202e", "\u2066", "\u2067", "\u2068", "\u2069", "\u061c", "\u200e", "\u200f"}
spaces = []string{"\u00a0", "\u1680", "\u2000", "\u2001", "\u2002", "\u2007", "\u200a", "\u2028", "\u2029", "\u202f", "\u205f", "\u3000", "\ufeff"}
whitelist = []string{zwnj, zwj, vs15, vs16}
emoji = []string{"\U0001F600", "\U0001F3F3", "\U0001F308", "\U0001F468", "\U0001F469", "\U0001F467", "\U0001F3F4", "\u2764", "\u263a", "\U0001F44D", "\U0001F3FB", "\u00a9", "\u2122", "#", "*", "1"}
special = []string{"\u00df", "\u1e9e", "\u0131", "\u0130", "\u03c2", "\u03c3", "\u03a3", "\u212a", "\u212b", "\u00c5", "\ufb00", "\u01c5", "\u01c4", "\u01c6", "\uab70", "\u13a0", "\ua7cb", "\u0390", "\u01d6", "\u1f82", "\U0001D160", "\u00e9", "e\u0301", "\u00b9", "\u00b2", "\u00b3", "\u00b5", "\u017f", "\u2126", "\u03a9"}
invalid = []string{"\x80", "\xbf", "\xc0\x80", "\xc1\xbf", "\xc2", "\xdf", "\xe0\x80\x80", "\xe0\xa0", "\xe2\x82", "\xed\xa0\x80", "\xed\xbf\xbf", "\xef\xbf", "\xf0\x80\x80\x80", "\xf0\x9f\x98", "\xf4\x90\x80\x80", "\xf5\x80\x80\x80", "\xf8\x88\x80\x80\x80", "\xfe", "\xff"}
reserved = []string{"CON", "PRN", "AUX", "NUL", "CONIN$", "CONOUT$", "COM0", "COM5", "COM9", "LPT0", "LPT1", "LPT9", "COM\u00b9", "COM\u00b2", "COM\u00b3", "LPT\u00b9", "LPT\u00b2", "LPT\u00b3"}
)
func hangul() string {
switch rng.IntN(4) {
case 0:
return string(rune(0x1100 + rng.IntN(19)))
case 1:
return string(rune(0x1161 + rng.IntN(21)))
case 2:
return string(rune(0x11a8 + rng.IntN(27)))
}
return string(rune(0xac00 + rng.IntN(11172)))
}
func assignedRune() rune {
for {
r := inRange(tb.assigned)
if r < 0xd800 || r > 0xdfff {
return r
}
}
}
func unassignedRune() rune {
switch rng.IntN(4) {
case 0:
return rune(0xfdd0 + rng.IntN(32))
case 1:
return rune(rng.IntN(17)<<16 | 0xfffe | rng.IntN(2))
}
for {
r := rune(rng.IntN(0x110000))
if !pathrule.Assigned(r) && (r < 0xd800 || r > 0xdfff) {
return r
}
}
}
func privateUse() rune {
switch rng.IntN(3) {
case 0:
return rune(0xf000 + rng.IntN(0x100))
case 1:
return rune(0xe000 + rng.IntN(0x1900))
}
return rune(0xf0000 + rng.IntN(0x1fffe))
}
// unit is one piece of a string, from a pool chosen by weight: mostly code
// points that pass R4, so that the later rules are reached too.
func unit() string {
switch n := rng.IntN(100); {
case n < 22:
return pick(letters)
case n < 27:
return pick(digits)
case n < 32:
return pick(punct)
case n < 34:
return pick(forbidden)
case n < 36:
return pick(controls)
case n < 44:
return string(pick(tb.marks))
case n < 49:
return string(pick(tb.decomp))
case n < 54:
return string(pick(tb.fold))
case n < 56:
return string(pick(tb.lower))
case n < 60:
return hangul()
case n < 66:
return pick(whitelist)
case n < 68:
return string(inRange(tb.ignorable))
case n < 71:
return pick(emoji)
case n < 74:
return string(pick(append(tb.vs15, tb.vs16...)))
case n < 82:
return string(pick(tb.bestFit))
case n < 83:
return pick(bidi)
case n < 85:
return pick(spaces)
case n < 86:
return string(privateUse())
case n < 87:
return string(unassignedRune())
case n < 91:
return string(assignedRune())
case n < 93:
return pick(invalid)
}
return pick(special)
}
func segment(min, max int) string {
var b strings.Builder
for n := min + rng.IntN(max-min+1); n > 0; n-- {
b.WriteString(unit())
}
return b.String()
}
func join(segs []string) string { return strings.Join(segs, "/") }
// mixed is a path of one to four segments of mixed units.
func mixed() string {
segs := make([]string, 1+rng.IntN(4)*rng.IntN(2))
for i := range segs {
segs[i] = segment(1, 8)
}
return join(segs)
}
// marks is a run of a base and combining marks of several classes, equal
// classes and starters among them, and ZWJ or ZWNJ between them.
func marks() string {
var b strings.Builder
for k := 1 + rng.IntN(3); k > 0; k-- {
switch rng.IntN(5) {
case 0:
b.WriteString(hangul())
case 1:
b.WriteRune(pick(tb.decomp))
case 2:
b.WriteString(pick(special))
default:
b.WriteString(pick(letters))
}
for m := rng.IntN(6); m > 0; m-- {
switch rng.IntN(10) {
case 0:
b.WriteString(pick(whitelist[:2]))
case 1:
b.WriteString(pick(letters))
case 2:
// Two marks of the same class, in either order.
c := tb.byClass[pick(tb.classes)]
b.WriteRune(pick(c))
b.WriteRune(pick(c))
case 3:
b.WriteRune(pick(tb.decomp))
default:
b.WriteRune(pick(tb.marks))
}
}
}
return b.String()
}
// emojiSeq is an emoji sequence with its selectors and joiners right or
// wrong.
func emojiSeq() string {
var b strings.Builder
for k := 1 + rng.IntN(3); k > 0; k-- {
switch rng.IntN(4) {
case 0:
b.WriteRune(pick(tb.vs15))
b.WriteString(vs15)
case 1:
b.WriteRune(pick(tb.vs16))
b.WriteString(vs16)
case 2:
b.WriteString(pick(emoji))
default:
b.WriteString(unit())
}
switch rng.IntN(6) {
case 0:
b.WriteString(zwj)
case 1:
b.WriteString(pick(whitelist))
case 2:
b.WriteString(pick(whitelist) + pick(whitelist))
case 3:
b.WriteRune(rune(0xe0020 + rng.IntN(0x60)))
}
}
if chance(0.2) {
return pick(whitelist) + b.String()
}
return b.String()
}
// lookalike maps each ASCII byte of s, at random, to a code point that some
// best-fit table maps to it.
func lookalike(s string) string {
var b strings.Builder
for i := 0; i < len(s); i++ {
if rs := tb.bestFitTo[s[i]]; len(rs) > 0 && chance(0.6) {
b.WriteRune(pick(rs))
} else {
b.WriteByte(s[i])
}
}
return b.String()
}
func randCase(s string) string {
var b strings.Builder
for _, r := range s {
if r >= 'A' && r <= 'Z' && chance(0.5) {
r += 'a' - 'A'
}
b.WriteRune(r)
}
return b.String()
}
// name is a segment of the shape of a device name, an 8.3 alias or the
// reserved prefix of R10, as it is or projected by the best-fit tables.
func name() string {
var s string
switch rng.IntN(5) {
case 0, 1:
s = randCase(pick(reserved))
if chance(0.4) {
s += strings.Repeat(" ", 1+rng.IntN(2))
}
if chance(0.6) {
s += "." + segment(0, 3)
}
case 2, 3:
base := segment(0, 7)
s = base + "~" + strings.Repeat("1", 1+rng.IntN(7))
if chance(0.3) {
s = base + "~" + strconv.Itoa(rng.IntN(10000000))
}
if chance(0.6) {
s += "." + segment(0, 4)
}
if chance(0.1) {
s += ".x"
}
default:
s = randCase(".datekeys-") + segment(0, 3)
if chance(0.3) {
i := 1 + rng.IntN(9)
s = s[:i] + pick(whitelist[:2]) + s[i:]
}
}
if chance(0.4) {
s = lookalike(s)
}
if chance(0.3) {
s += "/" + segment(1, 4)
}
return s
}
// long is a segment at a limit of R3, in bytes or in UTF-16 code units of
// its NFD, or a path at the limit of R2.
func long() string {
switch rng.IntN(7) {
case 0:
return strings.Repeat("a", 250+rng.IntN(10))
case 1:
return strings.Repeat("\u00e9", 125+rng.IntN(5)) + strings.Repeat("a", rng.IntN(3))
case 2:
return strings.Repeat("\u0390", 83+rng.IntN(5))
case 3:
return strings.Repeat("\U0001D160", 40+rng.IntN(5))
case 4:
return strings.Repeat(hangul(), 80+rng.IntN(10))
case 5:
return strings.Repeat("\U0001F600", 62+rng.IntN(4))
}
segs := make([]string, 30+rng.IntN(5))
for i := range segs {
segs[i] = segment(1, 2)
}
return join(segs)
}
// broken puts bytes that are not valid UTF-8 among valid ones, also before a
// joiner or a selector, where the Go reference counts each such byte as the
// three bytes of U+FFFD.
func broken() string {
var b strings.Builder
for k := 1 + rng.IntN(5); k > 0; k-- {
switch rng.IntN(5) {
case 0, 1:
b.WriteString(pick(invalid))
case 2:
b.WriteString(pick(whitelist))
case 3:
b.WriteString(pick(letters))
default:
b.WriteString(unit())
}
}
return b.String()
}
// text is a text of several lines, as a comment or a declared author holds.
func text() string {
lines := make([]string, 1+rng.IntN(4))
for i := range lines {
var b strings.Builder
for n := rng.IntN(8); n > 0; n-- {
switch rng.IntN(8) {
case 0:
b.WriteString(" ")
case 1:
b.WriteString(pick([]string{"\t", "\r", "\r\n", "\v"}))
case 2:
b.WriteString(emojiSeq())
default:
b.WriteString(unit())
}
}
lines[i] = b.String()
}
s := strings.Join(lines, "\n")
if chance(0.2) {
s = " " + s
}
if chance(0.2) {
s += " "
}
return s
}
// ---------------------------------------------------------------------------
// Cases
// stringCase is a string with the results of the Go reference. Author is
// left out when it is Comment, as it is for most strings. Fold is not here:
// it maps each code point alone, which the planes check for every one.
type stringCase struct {
Name string `json:"name,omitempty"`
In string `json:"in"`
Path string `json:"path"`
Comment string `json:"comment"`
Author string `json:"author,omitempty"`
NFD string `json:"nfd"`
Key string `json:"key"`
}
func stringOf(name, s string) stringCase {
c := stringCase{
Name: name,
In: h(s),
Path: result(pathrule.CheckPath(s)),
Comment: result(pathrule.CheckComment(s)),
NFD: h(pathrule.NFD(s)),
Key: h(pathrule.Key(s)),
}
if a := result(pathrule.CheckAuthor(s)); a != c.Comment {
c.Author = a
}
return c
}
type treeCase struct {
Name string `json:"name,omitempty"`
In []string `json:"in"`
Result string `json:"result"`
Paths [2]int `json:"paths"`
}
func treeOf(name string, paths []string) treeCase {
err := pathrule.CheckTree(paths)
c := treeCase{Name: name, Result: result(err)}
for _, p := range paths {
c.In = append(c.In, h(p))
}
var e *pathrule.Error
if errors.As(err, &e) {
c.Paths = e.Paths
} else if err != nil {
log.Fatalf("CheckTree: an error that is not a pathrule.Error: %v", err)
}
return c
}
// The named cases: those of pathrule_test.go of datekeys-go and of
// pathrule.test.ts of datekeys-ts, and the edges this port must keep.
func namedStrings() []stringCase {
scotland := "\U0001F3F4\U000E0067\U000E0062\U000E0073\U000E0063\U000E0074\U000E007F"
named := [][2]string{
// TestCheckPath of datekeys-go.
{"a file", "nota.txt"},
{"a file in a folder", "fotos/playa.jpg"},
{"U+00BF, which bestfit1250 maps to '?'", "\u00bfQu\u00e9 es esto.jpg"},
{"U+2665, which bestfit874 maps to 0x03", "Para ti \u2665.jpg"},
{"U+00A7, which bestfit874 maps to 0x15", "\u00a7 3 contrato.pdf"},
{"U+2192, which bestfit1253 maps to '>'", "Madrid \u2192 Lisboa"},
{"VS16 after U+2764", "\u2764" + vs16 + ".txt"},
{"the rainbow flag", "\U0001F3F3" + vs16 + zwj + "\U0001F308"},
{"a family", "\U0001F468" + zwj + "\U0001F469" + zwj + "\U0001F467"},
{"ZWNJ between letters", "ab" + zwnj + "c"},
{"the reserved prefix below the first level", "a/.datekeys-x"},
{"nine before the tilde", "ABCDEFGHI~1"},
{"a tilde inside", "a~1b"},
{"a tilde and a year", "report~2023.txt"},
{"255 bytes", strings.Repeat("a", 255)},
{"85 times U+0390: 255 UTF-16 units after NFD", strings.Repeat("\u0390", 85)},
{"U+00A0 at both ends", "\u00a0a\u00a0"},
{"a leading slash", "/a"},
{"two slashes", "a//b"},
{"a trailing slash", "a/"},
{"33 segments", strings.Repeat("a/", 32) + "a"},
{"two dots", ".."},
{"a dot segment", "a/./b"},
{"ZWJ alone", zwj},
{"256 bytes", strings.Repeat("a", 256)},
{"127 times U+0390: 381 UTF-16 units after NFD", strings.Repeat("\u0390", 127)},
{"a colon", "a:b"},
{"a backslash", "a\\b"},
{"a question mark", "a?b"},
{"TAB", "a\tb"},
{"U+2028", "a\u2028b"},
{"U+00AD", "a\u00adb"},
{"U+206A", "a\u206ab"},
{"U+202E", "a\u202eb"},
{"U+F03A", "informe\uf03aanexo"},
{"U+0378", "a\u0378b"},
{"U+FFFE", "a\ufffeb"},
{"the flag of Scotland", scotland},
{"VS17", "\U0001F600\U000E0100"},
{"VS16 after a letter", "a" + vs16},
{"ZWJ first", zwj + "a"},
{"ZWJ last", "a" + zwj},
{"ZWJ twice", "a" + zwj + zwj + "b"},
{"ZWJ after ZWNJ", "a" + zwnj + zwj + "b"},
{"a leading space", " a"},
{"a trailing space", "a "},
{"a trailing dot", "a."},
{"CON.txt", "CON.txt"},
{"con", "con"},
{"Aux .log", "Aux .log"},
{"COM\u00b9", "COM\u00b9"},
{"lpt9.doc", "lpt9.doc"},
{"CONIN$", "CONIN$"},
{"an 8.3 alias", "ABCDEF~1"},
{"an 8.3 alias with an extension", "ABCDEF~1.TXT"},
{"~1", "~1"},
{"an 8.3 alias with \u00c4", "\u00c4~1.txt"},
{"CON.txt in full-width forms", "\uff23\uff2f\uff2e.txt"},
{"U+2216 SET MINUS", "a\u2216b"},
{"U+2236 RATIO", "a\u2236b"},
{"U+3000 first", "\u3000a"},
{"a full-width solidus", "a\uff0fb"},
{"two full-width dots", "\uff0e\uff0e"},
{".datekeys-x", ".datekeys-x"},
{".DateKeys-abc/b", ".DateKeys-abc/b"},
// TestNFD and TestKey of datekeys-go.
{"precomposed e acute", "\u00e9"},
{"the Angstrom sign", "\u212b"},
{"the Kelvin sign", "\u212a"},
{"Hangul with a final jamo", "\ud55c"},
{"Hangul without", "\ud558"},
{"two marks out of canonical order", "a\u0301\u0323"},
{"U+0390", "\u0390"},
{"U+01D6, a recursive decomposition", "\u01d6"},
{"ZWJ between marks, a starter", "a\u0301\u200d\u0323"},
{"A.txt", "A.txt"},
{"Stra\u00dfe", "Stra\u00dfe"},
{"STRASSE", "STRASSE"},
{"dotless i", "\u0131"},
{"decomposed e acute", "e\u0301"},
{"ab with ZWNJ", "a\u200cb"},
{"marks in canonical order", "a\u0323\u0301"},
{".DateKeys-x", ".DateKeys-x"},
// TestTexts of datekeys-go.
{"a comment of two lines", "Hola\tmundo\nsegunda l\u00ednea"},
{"VS16 in a text", "\u2764\ufe0f para ti"},
{"CR LF", "Hola\r\n"},
{"tag characters", "Hola\U000E0049\U000E0047"},
{"ZWJ at the end of a line", "a\u200d\nb"},
{"U+FFFE in a text", "a\ufffe"},
{"U+FDD0", "a\ufdd0"},
{"an author", "Ana L\u00f3pez"},
{"TAB in an author", "Ana\tL\u00f3pez"},
{"LF in an author", "Ana\nL\u00f3pez"},
{"a leading space in an author", " Ana"},
{"a trailing space in an author", "Ana "},
{"U+200E", "a\u200eb"},
// pathrule.test.ts of datekeys-ts.
{"RLO in the second segment", "a/\u202e"},
{"U+00A5, which cp932 maps to '\\'", "a\u00a5b"},
{"a dot and ZWJ", ".\u200d"},
{"two dots and ZWNJ", "a/..\u200c"},
{"VS15 after U+2764", "\u2764\ufe0e.txt"},
{"two dots, no alias", "ABC~1.tar.gz"},
{"VS16 on the second line", "ok\na\ufe0f"},
{"LF at the end of an author", "Ana\n"},
{"BOM first", "\ufeffa"},
// The edges of this port.
{"empty", ""},
{"a slash", "/"},
{"a dot", "."},
{"an emoji of four bytes", "\U0001F600"},
{"U+FF5E", "\uff5e"},
{"the U+FFFD of an invalid byte before a final ZWJ: no error", "\xffa\u200d"},
{"the U+FFFD of invalid bytes before a ZWJ that is not last", "\xff\xffa\u200dbcde"},
{"an invalid byte before VS16", "\xff\ufe0f"},
{"a surrogate in UTF-8 bytes", "\xed\xa0\x80"},
{"a truncated emoji", "a\xf0\x9f\x98"},
{"an invalid byte projected to three bytes", strings.Repeat("\xff", 250) + "\uff21"},
{"invalid bytes and a full-width dot", "\xff\uff0e"},
{"a dot and an invalid byte", ".\xff"},
{"a device name and an invalid byte", "CON\xff"},
{"a device name and a broken superscript", "COM\xb9"},
{"an alias with an invalid byte", "\xff~1"},
{"an alias of nine runes with an invalid byte", "\xffABCDEFG~1"},
{"an alias of eight runes, a truncated sequence", "\xe2\x82ABC~12"},
{"an alias with three invalid bytes after the dot", "A~1.\xff\xff\xff"},
{"an alias with four invalid bytes after the dot", "A~1.\xff\xff\xff\xff"},
{"a reserved prefix after an invalid byte", "\xff.datekeys-x"},
{"a full-width reserved prefix", "\uff0edatekeys-x"},
{"a reserved prefix with ZWJ", ".date\u200dkeys-x"},
{"a reserved prefix with a Kelvin sign", ".datE\u212aeys-x"},
{"seven digits after the tilde", "A~1234567"},
{"six digits after the tilde", "~123456"},
{"full-width digits after a tilde", "ABC~\uff11"},
{"a full-width tilde and digit", "ABC\uff5e\uff11"},
{"U+25D9, which bestfit874 maps to LF", "a\u25d9b"},
{"U+FF0E at the end", "a\uff0e"},
{"U+3000 at both ends of a text", "\u3000a\u3000"},
{"a trailing space before the dot of a device name", "NUL .txt"},
{"a device name with a superscript of another table", "LPT\u00b9.x"},
{"a mark after a starter after a mark", "a\u0301b\u0300"},
{"a mark of class 0 between marks", "a\u0301\u034f\u0323"},
{"two marks of the same class", "a\u0301\u0300"},
{"Hangul jamo after a syllable", "\uac00\u11a8"},
{"an astral decomposition", "\U0001D160"},
{"a combining mark first", "\u0301a"},
{"VS16 after a keycap base", "#\ufe0f"},
{"VS15 after a digit", "1\ufe0e"},
{"ZWJ between emoji, then VS16", "\U0001F468\u200d\ufe0f"},
{"ZWNJ alone in a line", "a\n\u200c"},
{"an empty line", "a\n\nb"},
{"a text of bidirectional controls only", "\u2066\u2069"},
// The limits of R2, R3 and R6b, on both sides.
{"32 segments", strings.Repeat("a/", 31) + "a"},
{"85 times U+0390 and a letter: 256 UTF-16 units after NFD", strings.Repeat("\u0390", 85) + "a"},
{"42 times U+1D160: 168 bytes, 252 UTF-16 units after NFD", strings.Repeat("\U0001d160", 42)},
{"43 times U+1D160: 172 bytes, 258 UTF-16 units after NFD", strings.Repeat("\U0001d160", 43)},
{"127 times U+00E9 and a letter: 255 bytes", strings.Repeat("\u00e9", 127) + "a"},
{"128 times U+00E9: 256 bytes", strings.Repeat("\u00e9", 128)},
{"63 emoji: 252 bytes, 126 UTF-16 units", strings.Repeat("\U0001f600", 63)},
{"64 emoji: 256 bytes", strings.Repeat("\U0001f600", 64)},
{"a base of nine runes before the tilde", "ABCDEFG~1"},
{"an 8.3 alias with an extension of four", "ABCDEF~1.TEXT"},
{"an 8.3 alias with an astral letter", "\U00010400BCDEF~1"},
{"an 8.3 alias with an astral extension", "ABCDEF~1.\U00010400\U00010401\U00010402"},
{"an astral extension of four", "ABCDEF~1.\U00010400\U00010401\U00010402\U00010403"},
}
var out []stringCase
for _, n := range named {
out = append(out, stringOf(n[0], n[1]))
}
return out
}
func randomStrings(n int, seen map[string]bool) []stringCase {
var out []stringCase
gens := []struct {
weight int
gen func() string
}{
{26, mixed}, {18, marks}, {10, emojiSeq}, {14, name}, {1, long},
{10, broken}, {8, text}, {8, func() string { return lookalike(mixed()) }},
}
total := 0
for _, g := range gens {
total += g.weight
}
for len(out) < n {
k := rng.IntN(total)
var s string
for _, g := range gens {
if k < g.weight {
s = g.gen()
break
}
k -= g.weight
}
if seen[s] {
continue
}
seen[s] = true
out = append(out, stringOf("", s))
}
return out
}
func namedTrees() []treeCase {
named := []struct {
name string
paths []string
}{
// TestCheckTree of datekeys-go.
{"three files in two folders", []string{"a/b", "a/c", "d"}},
{"A.txt and a.txt", []string{"A.txt", "a.txt"}},
{"A and a/b", []string{"A", "a/b"}},
{"a/b and a/b/c", []string{"a/b", "a/b/c"}},
{"Fotos/b and fotos/a", []string{"Fotos/b", "fotos/a"}},
{"STRASSE and Stra\u00dfe", []string{"STRASSE", "Stra\u00dfe"}},
{"ab with and without ZWNJ", []string{"ab", "a\u200cb"}},
{"k and the Kelvin sign", []string{"k", "\u212a"}},
// The trees of testdata/vectors/paths.json that reach CheckTree.
{"U+FF5E and U+1F600", []string{"\uff5e", "\U0001F600"}},
{"the NFD and the NFC of a name", []string{"e\u0301", "\u00e9"}},
// The edges of this port.
{"none", nil},
{"one", []string{"a"}},
{"the same path twice", []string{"a", "a"}},
{"the same folder twice", []string{"a/b", "a/c"}},
{"a folder, then a file of its name", []string{"a/b", "a"}},
{"a file below two folders that collide", []string{"x/a/b", "X/a/c"}},
{"two invalid bytes of the same key", []string{"\xff", "\xfe"}},
{"the same invalid byte twice", []string{"\xff/a", "\xff/b"}},
{"a segment whose key is empty", []string{"\u200d/a", "a"}},
{"an empty segment", []string{"a//b", "a/b"}},
{"dotless i and i", []string{"\u0131", "i"}},
{"I with a dot and i with a combining dot", []string{"\u0130", "i\u0307"}},
{"a ligature and two letters", []string{"\ufb00", "ff"}},
{"final sigma and sigma", []string{"\u03c2", "\u03c3"}},
{"Cherokee small and capital", []string{"\uab70", "\u13a0"}},
{"DZ with caron in three cases", []string{"\u01c4/a", "\u01c5/b", "\u01c6/c"}},
{"Hangul composed and decomposed", []string{"\ud55c", "\u1112\u1161\u11ab"}},
{"the capital sharp s and ss", []string{"\u1e9e", "ss"}},
{"full-width CON in two cases", []string{"\uff23\uff2f\uff2e", "\uff43\uff4f\uff4e"}},
{"a file, then a deeper file of the same name", []string{"a/b", "a/b/c/d"}},
}
var out []treeCase
for _, t := range named {
out = append(out, treeOf(t.name, t.paths))
}
return out
}
// The names of random trees: names that collide by their key, and others.
var treeNames = []string{
"A", "a", "B", "b", "A.txt", "a.txt", "Stra\u00dfe", "STRASSE", "strasse", "k", "\u212a", "K",
"\u00e9", "e\u0301", "E\u0301", "ab", "a\u200cb", "\u0131", "i", "I", "\u0130", "i\u0307", "\ufb00", "ff", "FF",
"\u03c2", "\u03c3", "\u03a3", "\xff", "\xfe", "\xff\xfe", "\u200d", "", "\u01c5", "\u01c6", "\u01c4", "\uab70", "\u13a0", "\ud55c",
"\u1112\u1161\u11ab", "\u1e9e", "ss", "SS", "\uff23\uff2f\uff2e", "\uff43\uff4f\uff4e", "x", "y", "z", "\uff5e", "\U0001F600", "a\u0301\u0323",
"a\u0323\u0301", "a\u0301\u200d\u0323",
}
func randomTrees(n int) []treeCase {
var out []treeCase
for len(out) < n {
paths := make([]string, 1+rng.IntN(6))
for i := range paths {
segs := make([]string, 1+rng.IntN(3))
for j := range segs {
segs[j] = pick(treeNames)
}
paths[i] = join(segs)
}
if chance(0.3) {
// Distinct names, sorted, as a head holds them.
paths = nil
used := map[string]bool{}
for k := 1 + rng.IntN(8); k > 0; k-- {
p := join([]string{pick(letters[:6]), pick(letters[:6]) + pick(digits)})
if chance(0.3) {
p = pick(letters[:6]) + pick(digits)
}
if !used[p] {
used[p] = true
paths = append(paths, p)
}
}
slices.Sort(paths)
}
out = append(out, treeOf("", paths))
}
return out
}
// folderCase is a tree of Count paths, each prefix + i in Digits decimal
// digits + suffix, for i from 0.
type folderCase struct {
Name string `json:"name"`
Prefix string `json:"prefix"`
Digits int `json:"digits"`
Suffix string `json:"suffix"`
Count int `json:"count"`
Result string `json:"result"`
}
func folders() []folderCase {
cases := []folderCase{
{Name: "65 536 folders", Prefix: "d", Digits: 5, Suffix: "/f", Count: 65536},
{Name: "65 535 folders", Prefix: "d", Digits: 5, Suffix: "/f", Count: 65535},
{Name: "two folders for each of 32 768 paths", Prefix: "a", Digits: 5, Suffix: "/b/c", Count: 32768},
{Name: "two folders for each of 32 767 paths, and one more", Prefix: "x/a", Digits: 5, Suffix: "/b", Count: 32767},
{Name: "one shared folder and 65 535 below it", Prefix: "x/y", Digits: 5, Suffix: "/f", Count: 65535},
}
for i := range cases {
c := &cases[i]
paths := make([]string, c.Count)
for j := range paths {
paths[j] = fmt.Sprintf("%s%0*d%s", c.Prefix, c.Digits, j, c.Suffix)
}
c.Result = result(pathrule.CheckTree(paths))
}
return cases
}
// ---------------------------------------------------------------------------
// Code points
type planeDigests struct {
Plane int `json:"plane"`
Props string `json:"props"`
NFD string `json:"nfd"`
Fold string `json:"fold"`
Key string `json:"key"`
Path string `json:"path"`
Comment string `json:"comment"`
Author string `json:"author"`
}
// plane hashes one line per code point of the plane p for each function;
// the string of a code point is its UTF-8, U+FFFD for a surrogate, as Go's
// string(rune(r)) writes it.
func plane(p int) planeDigests {
hs := make([]hash.Hash, 7)
for i := range hs {
hs[i] = sha256.New()
}
for r := rune(p << 16); r <= rune(p<<16|0xffff); r++ {
s := string(r)
fmt.Fprintf(hs[0], "%X %t %t %X\n", r, pathrule.Assigned(r), pathrule.DefaultIgnorable(r), pathrule.Lower(r))
fmt.Fprintf(hs[1], "%X %x\n", r, pathrule.NFD(s))
fmt.Fprintf(hs[2], "%X %x\n", r, pathrule.Fold(s))
fmt.Fprintf(hs[3], "%X %x\n", r, pathrule.Key(s))
fmt.Fprintf(hs[4], "%X %s\n", r, result(pathrule.CheckPath(s)))
fmt.Fprintf(hs[5], "%X %s\n", r, result(pathrule.CheckComment(s)))
fmt.Fprintf(hs[6], "%X %s\n", r, result(pathrule.CheckAuthor(s)))
}
d := func(i int) string { return hex.EncodeToString(hs[i].Sum(nil)) }
return planeDigests{p, d(0), d(1), d(2), d(3), d(4), d(5), d(6)}
}
// codePoints are code points drawn from the seed and the edges of the
// tables, with Assigned, DefaultIgnorable and Lower: [r, assigned,
// ignorable, lower], booleans as 0 and 1.
func codePoints() [][4]int {
set := map[rune]bool{}
for _, r := range []rune{
'a', 'A', 0x0378, 0x00ad, 0x200d, 0xfe0f, 0x206a, 0xe0041, 0xe0000, 0x2028, 0xfffe,
0xf03a, 0x10ffff, 0xd800, 0xdfff, 0xfffd, 0x0130, 0x0131, 0x212a, 0x1e9e, 0xa7cb,
0x13a0, 0xab70, 0x01c5, 0x03a3, 0x0000, 0x007f, 0x0080, 0x009f, 0x00a0, 0x0e33,
} {
set[r] = true
}
for _, rs := range [][][2]rune{tb.assigned, tb.ignorable} {
for _, x := range rs {
for _, r := range []rune{x[0] - 1, x[0], x[1], x[1] + 1} {
if r >= 0 && r <= 0x10ffff && chance(0.15) {
set[r] = true
}
}
}
}
for len(set) < 1200 {
switch rng.IntN(4) {
case 0:
set[pick(tb.lower)] = true
case 1:
set[rune(rng.IntN(0x3000))] = true
default:
set[rune(rng.IntN(0x110000))] = true
}
}
rs := make([]rune, 0, len(set))
for r := range set {
rs = append(rs, r)
}
slices.Sort(rs)
out := make([][4]int, len(rs))
b := func(v bool) int {
if v {
return 1
}
return 0
}
for i, r := range rs {
out[i] = [4]int{int(r), b(pathrule.Assigned(r)), b(pathrule.DefaultIgnorable(r)), int(pathrule.Lower(r))}
}
return out
}
// ---------------------------------------------------------------------------
// Output
// ascii escapes every code point above U+007F of JSON as \uXXXX, in UTF-16
// for those above U+FFFF: the same JSON, in ASCII.
func ascii(b []byte) []byte {
var out bytes.Buffer
for _, r := range string(b) {
switch {
case r < 0x80:
out.WriteByte(byte(r))
case r < 0x10000:
fmt.Fprintf(&out, `\u%04x`, r)
default:
r -= 0x10000
fmt.Fprintf(&out, `\u%04x\u%04x`, 0xd800+(r>>10), 0xdc00+(r&0x3ff))
}
}
return out.Bytes()
}
func marshal(v any) []byte {
var b bytes.Buffer
enc := json.NewEncoder(&b)
enc.SetEscapeHTML(false)
if err := enc.Encode(v); err != nil {
log.Fatal(err)
}
return ascii(bytes.TrimSuffix(b.Bytes(), []byte("\n")))
}
// list writes the items of a JSON array one per line.
func list[T any](w *bytes.Buffer, key string, items []T, last bool) {
fmt.Fprintf(w, " %q: [\n", key)
for i, it := range items {
w.WriteString(" ")
w.Write(marshal(it))
if i < len(items)-1 {
w.WriteByte(',')
}
w.WriteByte('\n')
}
w.WriteString(" ]")
if !last {
w.WriteByte(',')
}
w.WriteByte('\n')
}
func main() {
source := flag.String("source", "", "the commit of datekeys-go that this tree exports")
outDir := flag.String("out", "", "the directory test/vectors of datekeys-dart")
flag.Parse()
if *source == "" || *outDir == "" {
log.Fatal("usage: go run ./pathrule_go_vectors.go -source COMMIT -out DIR")
}
if !utf8.ValidString(pathrule.Canonical()) {
log.Fatal("the canonical text is not UTF-8")
}
if sum := sha256.Sum256([]byte(pathrule.Canonical())); hex.EncodeToString(sum[:]) != pathrule.TablesDigest {
log.Fatal("the canonical text does not give TablesDigest")
}
seen := map[string]bool{}
strs := namedStrings()
for _, c := range strs {
in, _ := hex.DecodeString(c.In)
seen[string(in)] = true
}
strs = append(strs, randomStrings(1300, seen)...)
trees := append(namedTrees(), randomTrees(350)...)
var planes []planeDigests
for p := 0; p <= 16; p++ {
planes = append(planes, plane(p))
}
var w bytes.Buffer
w.WriteString("{\n")
head := []struct {
key string
v any
}{
{"description", "The results of internal/pathrule of the Go reference (spec \u00a729.5, \u00a729.5.1, \u00a729.6) for lib/src/pathrule.dart; see tool/pathrule_go_vectors.go. " +
"Binary values are lower-case hex. strings: in, and the result of CheckPath, CheckComment and CheckAuthor, ok or the text of the error, author left out when it is the comment, and its NFD and Key. " +
"trees: the paths of CheckTree, its result and, for R7, the two paths it names. folders: CheckTree of count paths, prefix, i in digits decimal digits and suffix. " +
"code_points: [r, Assigned, DefaultIgnorable, Lower]. planes: for each plane, the SHA-256 of one line per code point of each function, of its UTF-8 or of U+FFFD for a surrogate: " +
"props \"%X %t %t %X\" of r, Assigned, DefaultIgnorable and Lower; nfd, fold and key \"%X %x\" of r and the result; path, comment and author \"%X %s\" of r and the result; each line ends in LF."},
{"generator", "tool/pathrule_go_vectors.go, " + runtime.Version()},
{"source", "datekeys-go " + *source},
{"unicode_version", pathrule.UnicodeVersion},
{"tables_digest", pathrule.TablesDigest},
{"sources", len(pathrule.Sources)},
{"limits", map[string]int{
"max_path_len": pathrule.MaxPathLen, "max_segments": pathrule.MaxSegments,
"max_segment_len": pathrule.MaxSegmentLen, "max_segment_utf16": pathrule.MaxSegmentUTF16,
"max_implicit_dirs": pathrule.MaxImplicitDirs, "max_comment_len": pathrule.MaxCommentLen,
"max_author_len": pathrule.MaxAuthorLen,
}},
}
for _, kv := range head {
fmt.Fprintf(&w, " %q: %s,\n", kv.key, marshal(kv.v))
}
list(&w, "strings", strs, false)
list(&w, "trees", trees, false)
list(&w, "folders", folders(), false)
list(&w, "code_points", codePoints(), false)
list(&w, "planes", planes, true)
w.WriteString("}\n")
var check any
if err := json.Unmarshal(w.Bytes(), &check); err != nil {
log.Fatalf("the JSON does not parse: %v", err)
}
if bytes.Contains(w.Bytes(), []byte("'''")) {
log.Fatal("the JSON holds three quotes")
}
path := filepath.Join(*outDir, "pathrule_vectors.json")
if err := os.WriteFile(path, w.Bytes(), 0o644); err != nil {
log.Fatal(err)
}
fmt.Printf("wrote %s, %d bytes: %d strings, %d trees\n", path, w.Len(), len(strs), len(trees))
dart := "// Generated by tool/pathrule_go_vectors.go from pathrule_vectors.json, for\n" +
"// the tests that also run compiled to JavaScript, where no file can be\n" +
"// read. Do not edit.\n\n" +
"/// The text of test/vectors/pathrule_vectors.json.\n" +
"const pathruleVectorsJson = r'''\n" + w.String() + "''';\n"
dpath := filepath.Join(*outDir, "pathrule_vectors.g.dart")
if err := os.WriteFile(dpath, []byte(dart), 0o644); err != nil {
log.Fatal(err)
}
fmt.Printf("wrote %s\n", dpath)
}

Powered by TurnKey Linux.