// Package pathrule checks the paths and the texts of the head of a format 3 // capsule (spec §29.5, §29.6) with fixed Unicode 18.0.0 and WindowsBestFit // tables (spec §29.5.1), never with the Unicode functions of the platform, // whose version changes with each runtime: Go 1.26.8 carries Unicode 15.0.0, // and Node 24.9, 16.0. // // The tables are generated by internal/pathrule/gen; see tables.go. package pathrule //go:generate go run ./gen -data ../../.cache -go tables.go import ( "crypto/sha256" "encoding/hex" "fmt" "slices" "strings" ) // mapping is a sorted table from a code point to a sequence of code points: // the value of keys[i] is data[start[i]:start[i+1]]. type mapping struct { keys []rune start []uint16 data []rune } func (m *mapping) lookup(r rune) ([]rune, bool) { i, ok := slices.BinarySearch(m.keys, r) if !ok { return nil, false } return m.data[m.start[i]:m.start[i+1]], true } // bestFitTable is one code page of WindowsBestFit, reduced to the non-ASCII // code points it maps to an ASCII byte. type bestFitTable struct { name string from []rune to []byte } func (t *bestFitTable) lookup(r rune) (byte, bool) { i, ok := slices.BinarySearch(t.from, r) if !ok { return 0, false } return t.to[i], true } // inRanges reports whether r lies in one of the inclusive ranges lo, hi, ... func inRanges(ranges []rune, r rune) bool { // The first range whose hi is >= r. lo, hi := 0, len(ranges)/2 for lo < hi { m := int(uint(lo+hi) >> 1) if ranges[2*m+1] < r { lo = m + 1 } else { hi = m } } return lo < len(ranges)/2 && ranges[2*lo] <= r } // Assigned reports whether r is an assigned code point of Unicode 18.0.0: a // code point that is not of general category Cn. func Assigned(r rune) bool { return inRanges(assignedRanges, r) } // DefaultIgnorable reports whether r has the property // Default_Ignorable_Code_Point in Unicode 18.0.0. func DefaultIgnorable(r rune) bool { return inRanges(ignorableRanges, r) } // ccc is the canonical combining class of r. func ccc(r rune) uint8 { i, ok := slices.BinarySearchFunc(cccTable, r, func(e uint32, r rune) int { switch c := rune(e >> 8); { case c < r: return -1 case c > r: return 1 } return 0 }) if !ok { return 0 } return uint8(cccTable[i]) } // Hangul syllable decomposition (Unicode §3.12). const ( hangulS = 0xAC00 hangulL = 0x1100 hangulV = 0x1161 hangulT = 0x11A7 hangulN = 21 * 28 hangulC = 19 * hangulN ) // NFD returns the canonical decomposition of s, normalization form D of UAX // #15 with the data of Unicode 18.0.0: every code point replaced by its full // canonical decomposition, Hangul syllables decomposed algorithmically, and // each run of combining marks put in canonical order, a stable sort by // canonical combining class. func NFD(s string) string { var out []rune for _, r := range s { switch { case r >= hangulS && r < hangulS+hangulC: i := r - hangulS out = append(out, hangulL+i/hangulN, hangulV+(i%hangulN)/28) if t := i % 28; t != 0 { out = append(out, hangulT+t) } default: if d, ok := decompositions.lookup(r); ok { out = append(out, d...) } else { out = append(out, r) } } } // Canonical ordering: insertion sort within each run of non-starters, // stable, by combining class. for i := 1; i < len(out); i++ { c := ccc(out[i]) if c == 0 { continue } for j := i; j > 0; j-- { p := ccc(out[j-1]) if p <= c { break } out[j-1], out[j] = out[j], out[j-1] } } return string(out) } // Fold returns the case folding of s: the mappings of CaseFolding.txt with // status C or F, and U+0131 (ı) to U+0069 (i), as NTFS does (spec §29.5, R7). func Fold(s string) string { var b strings.Builder for _, r := range s { if r == 0x0131 { b.WriteRune(0x0069) } else if f, ok := foldings.lookup(r); ok { for _, x := range f { b.WriteRune(x) } } else { b.WriteRune(r) } } return b.String() } // Lower returns the simple lowercase mapping of r in Unicode 18.0.0, field // 13 of UnicodeData.txt, or r itself. The key of words uses it (spec §38.1). func Lower(r rune) rune { if l, ok := lowercases.lookup(r); ok { return l[0] } return r } // The whitelist of R4: the Default_Ignorable_Code_Point that a path may hold // (spec §29.5). const ( zwnj = 0x200C zwj = 0x200D vs15 = 0xFE0E vs16 = 0xFE0F ) func whitelisted(r rune) bool { return r == zwnj || r == zwj || r == vs15 || r == vs16 } // stripWhitelist removes the whitelisted code points of R4 from s. func stripWhitelist(s string) string { return strings.Map(func(r rune) rune { if whitelisted(r) { return -1 } return r }, s) } // Key is the key of a segment in R7: NFD(fold(NFD(s′))), where s′ is the // segment without the whitelisted code points of R4. They are removed // before normalizing because they have combining class 0: between two // combining marks, they would change their canonical order. func Key(segment string) string { return NFD(Fold(NFD(stripWhitelist(segment)))) } // allowsSelector reports whether base forms an emoji variation sequence with // the selector sel, U+FE0E or U+FE0F (R4b). func allowsSelector(base, sel rune) bool { bases := vs16Bases if sel == vs15 { bases = vs15Bases } _, ok := slices.BinarySearch(bases, base) return ok } // Canonical is the canonical text of the tables, whose SHA-256 is // TablesDigest; the generator writes the same text from the data files. Each // value is in upper-case hexadecimal of at least four digits, except the // combining class and the code page, in decimal, and the best-fit byte, in // two hexadecimal digits. func Canonical() string { var b strings.Builder fmt.Fprintf(&b, "unicode %s\n", UnicodeVersion) for i := 0; i < len(assignedRanges); i += 2 { fmt.Fprintf(&b, "assigned %04X %04X\n", assignedRanges[i], assignedRanges[i+1]) } for i := 0; i < len(ignorableRanges); i += 2 { fmt.Fprintf(&b, "ignorable %04X %04X\n", ignorableRanges[i], ignorableRanges[i+1]) } for _, e := range cccTable { fmt.Fprintf(&b, "ccc %04X %d\n", e>>8, e&0xFF) } for _, m := range []struct { name string t *mapping }{{"decomp", &decompositions}, {"fold", &foldings}, {"lower", &lowercases}} { for i, k := range m.t.keys { fmt.Fprintf(&b, "%s %04X", m.name, k) for _, r := range m.t.data[m.t.start[i]:m.t.start[i+1]] { fmt.Fprintf(&b, " %04X", r) } b.WriteByte('\n') } } for _, r := range vs15Bases { fmt.Fprintf(&b, "vs15 %04X\n", r) } for _, r := range vs16Bases { fmt.Fprintf(&b, "vs16 %04X\n", r) } for _, t := range bestFitTables { for i, r := range t.from { fmt.Fprintf(&b, "bestfit %s %04X %02X\n", t.name, r, t.to[i]) } } return b.String() } // digest is the SHA-256 of Canonical, in hexadecimal. func digest() string { sum := sha256.Sum256([]byte(Canonical())) return hex.EncodeToString(sum[:]) }