Format 3, step 1: Unicode 18.0.0 tables and the path and text rules
internal/pathrule checks the paths and the texts of a format 3 head
(spec 29.5, 29.6) with its own tables, never with the Unicode functions
of the platform, whose version changes with each runtime.
- gen reads the 19 pinned data files (UnicodeData, DerivedCoreProperties,
CaseFolding and emoji-variation-sequences of Unicode 18.0.0, and the
15 WindowsBestFit tables), checks their SHA-256 and writes tables.go:
assigned code points, Default_Ignorable_Code_Point, full canonical
decompositions and combining classes, C and F folding, the bases of
the emoji variation sequences, and the non-ASCII code points each
code page maps to ASCII. The data files stay out of git, in .cache.
- NFD, Fold and the key of R7; CheckPath with R2 to R6c and R10,
CheckTree with R7 and then R9, and CheckComment and CheckAuthor with
the invisible-character rule. Errors name the rule and never echo the
creator's text, so that another implementation can match them.
- The canonical text of the tables has a SHA-256, TablesDigest, which
the tests recompute and a TypeScript implementation will share.
- Checked against golang.org/x/text (Unicode 15.0.0) outside this
module: NFD matches on every code point both know, and folding only
differs on the 86 Cherokee letters that CaseFolding.txt folds to upper
case, as these tables do.
- The spec pins the SHA-256 of the 19 files in 29.5.1.
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
1 week ago
|
|
|
|
package pathrule
|
|
|
|
|
|
|
|
|
|
|
|
import (
|
|
|
|
|
|
"errors"
|
|
|
|
|
|
"fmt"
|
|
|
|
|
|
"strings"
|
|
|
|
|
|
"testing"
|
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
// The generated tables match their canonical text: nobody edited them.
|
|
|
|
|
|
func TestTablesDigest(t *testing.T) {
|
|
|
|
|
|
if got := digest(); got != TablesDigest {
|
|
|
|
|
|
t.Fatalf("digest of the tables %s, TablesDigest %s", got, TablesDigest)
|
|
|
|
|
|
}
|
|
|
|
|
|
if UnicodeVersion != "18.0.0" {
|
|
|
|
|
|
t.Fatalf("UnicodeVersion %q", UnicodeVersion)
|
|
|
|
|
|
}
|
|
|
|
|
|
if len(Sources) != 19 || len(bestFitTables) != 15 {
|
|
|
|
|
|
t.Fatalf("%d sources, %d best-fit tables", len(Sources), len(bestFitTables))
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
func TestProperties(t *testing.T) {
|
|
|
|
|
|
for _, c := range []struct {
|
|
|
|
|
|
r rune
|
|
|
|
|
|
assigned, ignorable bool
|
|
|
|
|
|
}{
|
|
|
|
|
|
{'a', true, false},
|
|
|
|
|
|
{0x0378, false, false}, // unassigned
|
|
|
|
|
|
{0x00AD, true, true}, // soft hyphen
|
|
|
|
|
|
{0x200D, true, true}, // ZWJ
|
|
|
|
|
|
{0xFE0F, true, true}, // VS16
|
|
|
|
|
|
{0x206A, true, true}, // deprecated format character
|
|
|
|
|
|
{0xE0041, true, true}, // tag A
|
|
|
|
|
|
{0xE0000, false, true}, // unassigned, still ignorable
|
|
|
|
|
|
{0x2028, true, false}, // line separator
|
|
|
|
|
|
{0xFFFE, false, false}, // noncharacter
|
|
|
|
|
|
{0xF03A, true, false}, // private use
|
|
|
|
|
|
{0x10FFFF, false, false},
|
|
|
|
|
|
} {
|
|
|
|
|
|
if Assigned(c.r) != c.assigned || DefaultIgnorable(c.r) != c.ignorable {
|
|
|
|
|
|
t.Errorf("U+%04X: assigned %v ignorable %v, want %v %v", c.r, Assigned(c.r), DefaultIgnorable(c.r), c.assigned, c.ignorable)
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
func TestNFD(t *testing.T) {
|
|
|
|
|
|
for _, c := range []struct{ in, want string }{
|
|
|
|
|
|
{"é", "é"},
|
|
|
|
|
|
{"Å", "Å"}, // ANGSTROM SIGN, a singleton decomposition
|
|
|
|
|
|
{"K", "K"}, // KELVIN SIGN
|
|
|
|
|
|
{"한", "한"}, // Hangul, with a final jamo
|
|
|
|
|
|
{"하", "하"}, // Hangul, without
|
|
|
|
|
|
{"ạ́", "ạ́"}, // canonical order
|
|
|
|
|
|
{"ΐ", "ΐ"},
|
|
|
|
|
|
{"ǖ", "ǖ"}, // recursive decomposition
|
|
|
|
|
|
{"ạ́", "ạ́"}, // ZWJ is a starter
|
|
|
|
|
|
} {
|
|
|
|
|
|
if got := NFD(c.in); got != c.want {
|
|
|
|
|
|
t.Errorf("NFD(%+q) = %+q, want %+q", c.in, got, c.want)
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
func TestKey(t *testing.T) {
|
|
|
|
|
|
for _, same := range [][2]string{
|
|
|
|
|
|
{"A.txt", "a.txt"},
|
|
|
|
|
|
{"Straße", "STRASSE"},
|
|
|
|
|
|
{"K", "k"},
|
|
|
|
|
|
{"ı", "i"},
|
|
|
|
|
|
{"é", "é"},
|
|
|
|
|
|
{"ab", "ab"},
|
|
|
|
|
|
{"ạ́", "ạ́"},
|
|
|
|
|
|
{".DateKeys-x", ".datekeys-x"},
|
|
|
|
|
|
} {
|
|
|
|
|
|
if Key(same[0]) != Key(same[1]) {
|
|
|
|
|
|
t.Errorf("Key(%+q) = %+q, Key(%+q) = %+q", same[0], Key(same[0]), same[1], Key(same[1]))
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
if Key("a") == Key("b") {
|
|
|
|
|
|
t.Error("distinct keys expected")
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// rule is the rule of a pathrule error, "" for nil.
|
|
|
|
|
|
func rule(err error) string {
|
|
|
|
|
|
var e *Error
|
|
|
|
|
|
if errors.As(err, &e) {
|
|
|
|
|
|
return e.Rule
|
|
|
|
|
|
}
|
|
|
|
|
|
if err != nil {
|
|
|
|
|
|
return "?" + err.Error()
|
|
|
|
|
|
}
|
|
|
|
|
|
return ""
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
func TestCheckPath(t *testing.T) {
|
|
|
|
|
|
const (
|
|
|
|
|
|
zwj = ""
|
|
|
|
|
|
zwnj = ""
|
|
|
|
|
|
vs16 = "️"
|
|
|
|
|
|
)
|
|
|
|
|
|
scotland := "\U0001F3F4\U000E0067\U000E0062\U000E0073\U000E0063\U000E0074\U000E007F"
|
|
|
|
|
|
for _, c := range []struct {
|
|
|
|
|
|
path, rule string
|
|
|
|
|
|
}{
|
|
|
|
|
|
// Accepted.
|
|
|
|
|
|
{"nota.txt", ""},
|
|
|
|
|
|
{"fotos/playa.jpg", ""},
|
|
|
|
|
|
{"¿Qué es esto.jpg", ""}, // bestfit1250 maps ¿ to '?', which R6c no longer rejects
|
|
|
|
|
|
{"Para ti ♥.jpg", ""}, // bestfit874 maps ♥ to 0x03
|
|
|
|
|
|
{"§ 3 contrato.pdf", ""}, // bestfit874 maps § to 0x15
|
|
|
|
|
|
{"Madrid → Lisboa", ""}, // bestfit1253 maps → to '>'
|
|
|
|
|
|
{"❤" + vs16 + ".txt", ""}, // ❤️
|
|
|
|
|
|
{"\U0001F3F3" + vs16 + zwj + "\U0001F308", ""}, // rainbow flag
|
|
|
|
|
|
{"\U0001F468" + zwj + "\U0001F469" + zwj + "\U0001F467", ""},
|
|
|
|
|
|
{"ab" + zwnj + "c", ""},
|
|
|
|
|
|
{"a/.datekeys-x", ""}, // only the first segment is reserved
|
|
|
|
|
|
{"ABCDEFGHI~1", ""}, // longer than 8 before the dot
|
|
|
|
|
|
{"a~1b", ""},
|
|
|
|
|
|
{"report~2023.txt", ""},
|
|
|
|
|
|
{strings.Repeat("a", 255), ""},
|
|
|
|
|
|
{strings.Repeat("ΐ", 85), ""}, // 170 bytes, 255 UTF-16 units after NFD
|
|
|
|
|
|
{" a ", ""}, // no table maps U+00A0 to U+0020... checked below
|
|
|
|
|
|
|
|
|
|
|
|
// R2.
|
|
|
|
|
|
{"/a", "R2"},
|
|
|
|
|
|
{"a//b", "R2"},
|
|
|
|
|
|
{"a/", "R2"},
|
|
|
|
|
|
{strings.Repeat("a/", 32) + "a", "R2"},
|
|
|
|
|
|
// R3.
|
|
|
|
|
|
{"..", "R3"},
|
|
|
|
|
|
{"a/./b", "R3"},
|
|
|
|
|
|
{zwj, "R4b"}, // R3 sees an empty stripped segment? No: R3 first.
|
|
|
|
|
|
{strings.Repeat("a", 256), "R3"},
|
|
|
|
|
|
{strings.Repeat("ΐ", 127), "R3"}, // 254 bytes, 381 UTF-16 units after NFD
|
|
|
|
|
|
// R4.
|
|
|
|
|
|
{"a:b", "R4"},
|
|
|
|
|
|
{"a\\b", "R4"},
|
|
|
|
|
|
{"a?b", "R4"},
|
|
|
|
|
|
{"a\tb", "R4"},
|
|
|
|
|
|
{"a
b", "R4"},
|
|
|
|
|
|
{"ab", "R4"},
|
|
|
|
|
|
{"ab", "R4"},
|
|
|
|
|
|
{"ab", "R4"},
|
|
|
|
|
|
{"informeanexo", "R4"},
|
|
|
|
|
|
{"ab", "R4"},
|
|
|
|
|
|
{"ab", "R4"},
|
|
|
|
|
|
{scotland, "R4"},
|
|
|
|
|
|
{"\U0001F600\U000E0100", "R4"}, // VS17
|
|
|
|
|
|
// R4b.
|
|
|
|
|
|
{"a" + vs16, "R4b"},
|
|
|
|
|
|
{zwj + "a", "R4b"},
|
|
|
|
|
|
{"a" + zwj, "R4b"},
|
|
|
|
|
|
{"a" + zwj + zwj + "b", "R4b"},
|
|
|
|
|
|
{"a" + zwnj + zwj + "b", "R4b"},
|
|
|
|
|
|
// R5.
|
|
|
|
|
|
{" a", "R5"},
|
|
|
|
|
|
{"a ", "R5"},
|
|
|
|
|
|
{"a.", "R5"},
|
|
|
|
|
|
// R6.
|
|
|
|
|
|
{"CON.txt", "R6"},
|
|
|
|
|
|
{"con", "R6"},
|
|
|
|
|
|
{"Aux .log", "R6"},
|
|
|
|
|
|
{"COM¹", "R6"},
|
|
|
|
|
|
{"lpt9.doc", "R6"},
|
|
|
|
|
|
{"CONIN$", "R6"},
|
|
|
|
|
|
// R6b.
|
|
|
|
|
|
{"ABCDEF~1", "R6b"},
|
|
|
|
|
|
{"ABCDEF~1.TXT", "R6b"},
|
|
|
|
|
|
{"~1", "R6b"},
|
|
|
|
|
|
{"Ä~1.txt", "R6b"},
|
|
|
|
|
|
// R6c.
|
|
|
|
|
|
{"CON.txt", "R6c"}, // full-width CON
|
|
|
|
|
|
{"a∖b", "R6c"}, // SET MINUS
|
|
|
|
|
|
{"a∶b", "R6c"}, // RATIO
|
|
|
|
|
|
{" a", "R6c"}, // bestfit1252 maps U+3000 to U+0020, which breaks R5
|
|
|
|
|
|
{"a/b", "R6c"}, // full-width solidus
|
|
|
|
|
|
{"..", "R6c"}, // full-width dots project to ".."
|
|
|
|
|
|
// R10.
|
|
|
|
|
|
{".datekeys-x", "R10"},
|
|
|
|
|
|
{".DateKeys-abc/b", "R10"},
|
|
|
|
|
|
} {
|
|
|
|
|
|
got := rule(CheckPath(c.path))
|
|
|
|
|
|
if c.path == zwj {
|
|
|
|
|
|
// R3 comes before R4b in a segment: the stripped segment is empty.
|
|
|
|
|
|
if got != "R3" {
|
|
|
|
|
|
t.Errorf("CheckPath(%+q): %s, want R3", c.path, got)
|
|
|
|
|
|
}
|
|
|
|
|
|
continue
|
|
|
|
|
|
}
|
|
|
|
|
|
if c.path == " a " {
|
|
|
|
|
|
// Whatever the tables give, it must be R6c or nothing (R5 is
|
|
|
|
|
|
// about U+0020 only).
|
|
|
|
|
|
if got != "" && got != "R6c" {
|
|
|
|
|
|
t.Errorf("CheckPath(%+q): %s", c.path, got)
|
|
|
|
|
|
}
|
|
|
|
|
|
continue
|
|
|
|
|
|
}
|
|
|
|
|
|
if got != c.rule {
|
|
|
|
|
|
t.Errorf("CheckPath(%+q): %s (%v), want %s", c.path, got, CheckPath(c.path), c.rule)
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
func TestCheckTree(t *testing.T) {
|
|
|
|
|
|
for _, c := range []struct {
|
|
|
|
|
|
paths []string
|
|
|
|
|
|
rule string
|
|
|
|
|
|
}{
|
|
|
|
|
|
{[]string{"a/b", "a/c", "d"}, ""},
|
|
|
|
|
|
{[]string{"A.txt", "a.txt"}, "R7"},
|
|
|
|
|
|
{[]string{"A", "a/b"}, "R7"},
|
|
|
|
|
|
{[]string{"a/b", "a/b/c"}, "R7"},
|
|
|
|
|
|
{[]string{"Fotos/b", "fotos/a"}, "R7"},
|
|
|
|
|
|
{[]string{"STRASSE", "Straße"}, "R7"},
|
|
|
|
|
|
{[]string{"ab", "ab"}, "R7"},
|
|
|
|
|
|
{[]string{"k", "K"}, "R7"},
|
|
|
|
|
|
} {
|
|
|
|
|
|
if got := rule(CheckTree(c.paths)); got != c.rule {
|
|
|
|
|
|
t.Errorf("CheckTree(%+q): %s, want %s", c.paths, got, c.rule)
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
many := make([]string, MaxImplicitDirs+1)
|
|
|
|
|
|
for i := range many {
|
|
|
|
|
|
many[i] = fmt.Sprintf("d%05d/f", i)
|
|
|
|
|
|
}
|
|
|
|
|
|
if got := rule(CheckTree(many)); got != "R9" {
|
|
|
|
|
|
t.Errorf("%d folders: %s, want R9", len(many), got)
|
|
|
|
|
|
}
|
|
|
|
|
|
if got := rule(CheckTree(many[:MaxImplicitDirs])); got != "" {
|
|
|
|
|
|
t.Errorf("%d folders: %s", MaxImplicitDirs, got)
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
func TestTexts(t *testing.T) {
|
|
|
|
|
|
for _, c := range []struct {
|
|
|
|
|
|
text string
|
|
|
|
|
|
comment bool
|
|
|
|
|
|
rule string
|
|
|
|
|
|
}{
|
|
|
|
|
|
{"Hola\tmundo\nsegunda línea", true, ""},
|
|
|
|
|
|
{"❤️ para ti", true, ""},
|
|
|
|
|
|
{"Hola\r\n", true, "text"},
|
|
|
|
|
|
{"ab", true, "text"},
|
|
|
|
|
|
{"Hola\U000E0049\U000E0047", true, "text"}, // tag characters
|
|
|
|
|
|
{"\U0001F600\U000E0100", true, "text"}, // VS17
|
|
|
|
|
|
{"a️", true, "text"}, // VS16 after a letter
|
|
|
|
|
|
{"a\nb", true, "text"}, // ZWJ at the end of a line
|
|
|
|
|
|
{"a", true, "text"},
|
|
|
|
|
|
{"a", true, "text"},
|
|
|
|
|
|
{"ab", true, "text"},
|
|
|
|
|
|
{"Ana López", false, ""},
|
|
|
|
|
|
{"Ana\tLópez", false, "text"},
|
|
|
|
|
|
{"Ana\nLópez", false, "text"},
|
|
|
|
|
|
{" Ana", false, "text"},
|
|
|
|
|
|
{"Ana ", false, "text"},
|
|
|
|
|
|
{"ab", false, "text"},
|
|
|
|
|
|
} {
|
|
|
|
|
|
check := CheckAuthor
|
|
|
|
|
|
if c.comment {
|
|
|
|
|
|
check = CheckComment
|
|
|
|
|
|
}
|
|
|
|
|
|
if got := rule(check(c.text)); got != c.rule {
|
|
|
|
|
|
t.Errorf("%+q (comment %v): %s, want %s", c.text, c.comment, got, c.rule)
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|