|
|
// Package pathrule checks the paths and the texts of the head of a format 3
|
|
|
// capsule (spec §29.5, §29.6) with fixed Unicode 18.0.0 and WindowsBestFit
|
|
|
// tables (spec §29.5.1), never with the Unicode functions of the platform,
|
|
|
// whose version changes with each runtime: Go 1.26.8 carries Unicode 15.0.0,
|
|
|
// and Node 24.9, 16.0.
|
|
|
//
|
|
|
// The tables are generated by internal/pathrule/gen; see tables.go.
|
|
|
package pathrule
|
|
|
|
|
|
//go:generate go run ./gen -data ../../.cache -go tables.go
|
|
|
|
|
|
import (
|
|
|
"crypto/sha256"
|
|
|
"encoding/hex"
|
|
|
"fmt"
|
|
|
"slices"
|
|
|
"strings"
|
|
|
)
|
|
|
|
|
|
// mapping is a sorted table from a code point to a sequence of code points:
|
|
|
// the value of keys[i] is data[start[i]:start[i+1]].
|
|
|
type mapping struct {
|
|
|
keys []rune
|
|
|
start []uint16
|
|
|
data []rune
|
|
|
}
|
|
|
|
|
|
func (m *mapping) lookup(r rune) ([]rune, bool) {
|
|
|
i, ok := slices.BinarySearch(m.keys, r)
|
|
|
if !ok {
|
|
|
return nil, false
|
|
|
}
|
|
|
return m.data[m.start[i]:m.start[i+1]], true
|
|
|
}
|
|
|
|
|
|
// bestFitTable is one code page of WindowsBestFit, reduced to the non-ASCII
|
|
|
// code points it maps to an ASCII byte.
|
|
|
type bestFitTable struct {
|
|
|
name string
|
|
|
from []rune
|
|
|
to []byte
|
|
|
}
|
|
|
|
|
|
func (t *bestFitTable) lookup(r rune) (byte, bool) {
|
|
|
i, ok := slices.BinarySearch(t.from, r)
|
|
|
if !ok {
|
|
|
return 0, false
|
|
|
}
|
|
|
return t.to[i], true
|
|
|
}
|
|
|
|
|
|
// inRanges reports whether r lies in one of the inclusive ranges lo, hi, ...
|
|
|
func inRanges(ranges []rune, r rune) bool {
|
|
|
// The first range whose hi is >= r.
|
|
|
lo, hi := 0, len(ranges)/2
|
|
|
for lo < hi {
|
|
|
m := int(uint(lo+hi) >> 1)
|
|
|
if ranges[2*m+1] < r {
|
|
|
lo = m + 1
|
|
|
} else {
|
|
|
hi = m
|
|
|
}
|
|
|
}
|
|
|
return lo < len(ranges)/2 && ranges[2*lo] <= r
|
|
|
}
|
|
|
|
|
|
// Assigned reports whether r is an assigned code point of Unicode 18.0.0: a
|
|
|
// code point that is not of general category Cn.
|
|
|
func Assigned(r rune) bool { return inRanges(assignedRanges, r) }
|
|
|
|
|
|
// DefaultIgnorable reports whether r has the property
|
|
|
// Default_Ignorable_Code_Point in Unicode 18.0.0.
|
|
|
func DefaultIgnorable(r rune) bool { return inRanges(ignorableRanges, r) }
|
|
|
|
|
|
// ccc is the canonical combining class of r.
|
|
|
func ccc(r rune) uint8 {
|
|
|
i, ok := slices.BinarySearchFunc(cccTable, r, func(e uint32, r rune) int {
|
|
|
switch c := rune(e >> 8); {
|
|
|
case c < r:
|
|
|
return -1
|
|
|
case c > r:
|
|
|
return 1
|
|
|
}
|
|
|
return 0
|
|
|
})
|
|
|
if !ok {
|
|
|
return 0
|
|
|
}
|
|
|
return uint8(cccTable[i])
|
|
|
}
|
|
|
|
|
|
// Hangul syllable decomposition (Unicode §3.12).
|
|
|
const (
|
|
|
hangulS = 0xAC00
|
|
|
hangulL = 0x1100
|
|
|
hangulV = 0x1161
|
|
|
hangulT = 0x11A7
|
|
|
hangulN = 21 * 28
|
|
|
hangulC = 19 * hangulN
|
|
|
)
|
|
|
|
|
|
// NFD returns the canonical decomposition of s, normalization form D of UAX
|
|
|
// #15 with the data of Unicode 18.0.0: every code point replaced by its full
|
|
|
// canonical decomposition, Hangul syllables decomposed algorithmically, and
|
|
|
// each run of combining marks put in canonical order, a stable sort by
|
|
|
// canonical combining class.
|
|
|
func NFD(s string) string {
|
|
|
var out []rune
|
|
|
for _, r := range s {
|
|
|
switch {
|
|
|
case r >= hangulS && r < hangulS+hangulC:
|
|
|
i := r - hangulS
|
|
|
out = append(out, hangulL+i/hangulN, hangulV+(i%hangulN)/28)
|
|
|
if t := i % 28; t != 0 {
|
|
|
out = append(out, hangulT+t)
|
|
|
}
|
|
|
default:
|
|
|
if d, ok := decompositions.lookup(r); ok {
|
|
|
out = append(out, d...)
|
|
|
} else {
|
|
|
out = append(out, r)
|
|
|
}
|
|
|
}
|
|
|
}
|
|
|
// Canonical ordering: insertion sort within each run of non-starters,
|
|
|
// stable, by combining class.
|
|
|
for i := 1; i < len(out); i++ {
|
|
|
c := ccc(out[i])
|
|
|
if c == 0 {
|
|
|
continue
|
|
|
}
|
|
|
for j := i; j > 0; j-- {
|
|
|
p := ccc(out[j-1])
|
|
|
if p <= c {
|
|
|
break
|
|
|
}
|
|
|
out[j-1], out[j] = out[j], out[j-1]
|
|
|
}
|
|
|
}
|
|
|
return string(out)
|
|
|
}
|
|
|
|
|
|
// Fold returns the case folding of s: the mappings of CaseFolding.txt with
|
|
|
// status C or F, and U+0131 (ı) to U+0069 (i), as NTFS does (spec §29.5, R7).
|
|
|
func Fold(s string) string {
|
|
|
var b strings.Builder
|
|
|
for _, r := range s {
|
|
|
if r == 0x0131 {
|
|
|
b.WriteRune(0x0069)
|
|
|
} else if f, ok := foldings.lookup(r); ok {
|
|
|
for _, x := range f {
|
|
|
b.WriteRune(x)
|
|
|
}
|
|
|
} else {
|
|
|
b.WriteRune(r)
|
|
|
}
|
|
|
}
|
|
|
return b.String()
|
|
|
}
|
|
|
|
|
|
// Lower returns the simple lowercase mapping of r in Unicode 18.0.0, field
|
|
|
// 13 of UnicodeData.txt, or r itself. The key of words uses it (spec §38.1).
|
|
|
func Lower(r rune) rune {
|
|
|
if l, ok := lowercases.lookup(r); ok {
|
|
|
return l[0]
|
|
|
}
|
|
|
return r
|
|
|
}
|
|
|
|
|
|
// The whitelist of R4: the Default_Ignorable_Code_Point that a path may hold
|
|
|
// (spec §29.5).
|
|
|
const (
|
|
|
zwnj = 0x200C
|
|
|
zwj = 0x200D
|
|
|
vs15 = 0xFE0E
|
|
|
vs16 = 0xFE0F
|
|
|
)
|
|
|
|
|
|
func whitelisted(r rune) bool { return r == zwnj || r == zwj || r == vs15 || r == vs16 }
|
|
|
|
|
|
// stripWhitelist removes the whitelisted code points of R4 from s.
|
|
|
func stripWhitelist(s string) string {
|
|
|
return strings.Map(func(r rune) rune {
|
|
|
if whitelisted(r) {
|
|
|
return -1
|
|
|
}
|
|
|
return r
|
|
|
}, s)
|
|
|
}
|
|
|
|
|
|
// Key is the key of a segment in R7: NFD(fold(NFD(s′))), where s′ is the
|
|
|
// segment without the whitelisted code points of R4. They are removed
|
|
|
// before normalizing because they have combining class 0: between two
|
|
|
// combining marks, they would change their canonical order.
|
|
|
func Key(segment string) string {
|
|
|
return NFD(Fold(NFD(stripWhitelist(segment))))
|
|
|
}
|
|
|
|
|
|
// allowsSelector reports whether base forms an emoji variation sequence with
|
|
|
// the selector sel, U+FE0E or U+FE0F (R4b).
|
|
|
func allowsSelector(base, sel rune) bool {
|
|
|
bases := vs16Bases
|
|
|
if sel == vs15 {
|
|
|
bases = vs15Bases
|
|
|
}
|
|
|
_, ok := slices.BinarySearch(bases, base)
|
|
|
return ok
|
|
|
}
|
|
|
|
|
|
// Canonical is the canonical text of the tables, whose SHA-256 is
|
|
|
// TablesDigest; the generator writes the same text from the data files. Each
|
|
|
// value is in upper-case hexadecimal of at least four digits, except the
|
|
|
// combining class and the code page, in decimal, and the best-fit byte, in
|
|
|
// two hexadecimal digits.
|
|
|
func Canonical() string {
|
|
|
var b strings.Builder
|
|
|
fmt.Fprintf(&b, "unicode %s\n", UnicodeVersion)
|
|
|
for i := 0; i < len(assignedRanges); i += 2 {
|
|
|
fmt.Fprintf(&b, "assigned %04X %04X\n", assignedRanges[i], assignedRanges[i+1])
|
|
|
}
|
|
|
for i := 0; i < len(ignorableRanges); i += 2 {
|
|
|
fmt.Fprintf(&b, "ignorable %04X %04X\n", ignorableRanges[i], ignorableRanges[i+1])
|
|
|
}
|
|
|
for _, e := range cccTable {
|
|
|
fmt.Fprintf(&b, "ccc %04X %d\n", e>>8, e&0xFF)
|
|
|
}
|
|
|
for _, m := range []struct {
|
|
|
name string
|
|
|
t *mapping
|
|
|
}{{"decomp", &decompositions}, {"fold", &foldings}, {"lower", &lowercases}} {
|
|
|
for i, k := range m.t.keys {
|
|
|
fmt.Fprintf(&b, "%s %04X", m.name, k)
|
|
|
for _, r := range m.t.data[m.t.start[i]:m.t.start[i+1]] {
|
|
|
fmt.Fprintf(&b, " %04X", r)
|
|
|
}
|
|
|
b.WriteByte('\n')
|
|
|
}
|
|
|
}
|
|
|
for _, r := range vs15Bases {
|
|
|
fmt.Fprintf(&b, "vs15 %04X\n", r)
|
|
|
}
|
|
|
for _, r := range vs16Bases {
|
|
|
fmt.Fprintf(&b, "vs16 %04X\n", r)
|
|
|
}
|
|
|
for _, t := range bestFitTables {
|
|
|
for i, r := range t.from {
|
|
|
fmt.Fprintf(&b, "bestfit %s %04X %02X\n", t.name, r, t.to[i])
|
|
|
}
|
|
|
}
|
|
|
return b.String()
|
|
|
}
|
|
|
|
|
|
// digest is the SHA-256 of Canonical, in hexadecimal.
|
|
|
func digest() string {
|
|
|
sum := sha256.Sum256([]byte(Canonical()))
|
|
|
return hex.EncodeToString(sum[:])
|
|
|
}
|