Format 3, step 1: Unicode 18.0.0 tables and the path and text rules
internal/pathrule checks the paths and the texts of a format 3 head
(spec 29.5, 29.6) with its own tables, never with the Unicode functions
of the platform, whose version changes with each runtime.
- gen reads the 19 pinned data files (UnicodeData, DerivedCoreProperties,
CaseFolding and emoji-variation-sequences of Unicode 18.0.0, and the
15 WindowsBestFit tables), checks their SHA-256 and writes tables.go:
assigned code points, Default_Ignorable_Code_Point, full canonical
decompositions and combining classes, C and F folding, the bases of
the emoji variation sequences, and the non-ASCII code points each
code page maps to ASCII. The data files stay out of git, in .cache.
- NFD, Fold and the key of R7; CheckPath with R2 to R6c and R10,
CheckTree with R7 and then R9, and CheckComment and CheckAuthor with
the invisible-character rule. Errors name the rule and never echo the
creator's text, so that another implementation can match them.
- The canonical text of the tables has a SHA-256, TablesDigest, which
the tests recompute and a TypeScript implementation will share.
- Checked against golang.org/x/text (Unicode 15.0.0) outside this
module: NFD matches on every code point both know, and folding only
differs on the 86 Cherokee letters that CaseFolding.txt folds to upper
case, as these tables do.
- The spec pins the SHA-256 of the 19 files in 29.5.1.
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
1 week ago
|
|
|
|
// Package pathrule checks the paths and the texts of the head of a format 3
|
|
|
|
|
|
// capsule (spec §29.5, §29.6) with fixed Unicode 18.0.0 and WindowsBestFit
|
|
|
|
|
|
// tables (spec §29.5.1), never with the Unicode functions of the platform,
|
|
|
|
|
|
// whose version changes with each runtime: Go 1.26.8 carries Unicode 15.0.0,
|
|
|
|
|
|
// and Node 24.9, 16.0.
|
|
|
|
|
|
//
|
|
|
|
|
|
// The tables are generated by internal/pathrule/gen; see tables.go.
|
|
|
|
|
|
package pathrule
|
|
|
|
|
|
|
|
|
|
|
|
//go:generate go run ./gen -data ../../.cache -go tables.go
|
|
|
|
|
|
|
|
|
|
|
|
import (
|
|
|
|
|
|
"crypto/sha256"
|
|
|
|
|
|
"encoding/hex"
|
|
|
|
|
|
"fmt"
|
|
|
|
|
|
"slices"
|
|
|
|
|
|
"strings"
|
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
// mapping is a sorted table from a code point to a sequence of code points:
|
|
|
|
|
|
// the value of keys[i] is data[start[i]:start[i+1]].
|
|
|
|
|
|
type mapping struct {
|
|
|
|
|
|
keys []rune
|
|
|
|
|
|
start []uint16
|
|
|
|
|
|
data []rune
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
func (m *mapping) lookup(r rune) ([]rune, bool) {
|
|
|
|
|
|
i, ok := slices.BinarySearch(m.keys, r)
|
|
|
|
|
|
if !ok {
|
|
|
|
|
|
return nil, false
|
|
|
|
|
|
}
|
|
|
|
|
|
return m.data[m.start[i]:m.start[i+1]], true
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// bestFitTable is one code page of WindowsBestFit, reduced to the non-ASCII
|
|
|
|
|
|
// code points it maps to an ASCII byte.
|
|
|
|
|
|
type bestFitTable struct {
|
|
|
|
|
|
name string
|
|
|
|
|
|
from []rune
|
|
|
|
|
|
to []byte
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
func (t *bestFitTable) lookup(r rune) (byte, bool) {
|
|
|
|
|
|
i, ok := slices.BinarySearch(t.from, r)
|
|
|
|
|
|
if !ok {
|
|
|
|
|
|
return 0, false
|
|
|
|
|
|
}
|
|
|
|
|
|
return t.to[i], true
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// inRanges reports whether r lies in one of the inclusive ranges lo, hi, ...
|
|
|
|
|
|
func inRanges(ranges []rune, r rune) bool {
|
|
|
|
|
|
// The first range whose hi is >= r.
|
|
|
|
|
|
lo, hi := 0, len(ranges)/2
|
|
|
|
|
|
for lo < hi {
|
|
|
|
|
|
m := int(uint(lo+hi) >> 1)
|
|
|
|
|
|
if ranges[2*m+1] < r {
|
|
|
|
|
|
lo = m + 1
|
|
|
|
|
|
} else {
|
|
|
|
|
|
hi = m
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
return lo < len(ranges)/2 && ranges[2*lo] <= r
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// Assigned reports whether r is an assigned code point of Unicode 18.0.0: a
|
|
|
|
|
|
// code point that is not of general category Cn.
|
|
|
|
|
|
func Assigned(r rune) bool { return inRanges(assignedRanges, r) }
|
|
|
|
|
|
|
|
|
|
|
|
// DefaultIgnorable reports whether r has the property
|
|
|
|
|
|
// Default_Ignorable_Code_Point in Unicode 18.0.0.
|
|
|
|
|
|
func DefaultIgnorable(r rune) bool { return inRanges(ignorableRanges, r) }
|
|
|
|
|
|
|
|
|
|
|
|
// ccc is the canonical combining class of r.
|
|
|
|
|
|
func ccc(r rune) uint8 {
|
|
|
|
|
|
i, ok := slices.BinarySearchFunc(cccTable, r, func(e uint32, r rune) int {
|
|
|
|
|
|
switch c := rune(e >> 8); {
|
|
|
|
|
|
case c < r:
|
|
|
|
|
|
return -1
|
|
|
|
|
|
case c > r:
|
|
|
|
|
|
return 1
|
|
|
|
|
|
}
|
|
|
|
|
|
return 0
|
|
|
|
|
|
})
|
|
|
|
|
|
if !ok {
|
|
|
|
|
|
return 0
|
|
|
|
|
|
}
|
|
|
|
|
|
return uint8(cccTable[i])
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// Hangul syllable decomposition (Unicode §3.12).
|
|
|
|
|
|
const (
|
|
|
|
|
|
hangulS = 0xAC00
|
|
|
|
|
|
hangulL = 0x1100
|
|
|
|
|
|
hangulV = 0x1161
|
|
|
|
|
|
hangulT = 0x11A7
|
|
|
|
|
|
hangulN = 21 * 28
|
|
|
|
|
|
hangulC = 19 * hangulN
|
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
// NFD returns the canonical decomposition of s, normalization form D of UAX
|
|
|
|
|
|
// #15 with the data of Unicode 18.0.0: every code point replaced by its full
|
|
|
|
|
|
// canonical decomposition, Hangul syllables decomposed algorithmically, and
|
|
|
|
|
|
// each run of combining marks put in canonical order, a stable sort by
|
|
|
|
|
|
// canonical combining class.
|
|
|
|
|
|
func NFD(s string) string {
|
|
|
|
|
|
var out []rune
|
|
|
|
|
|
for _, r := range s {
|
|
|
|
|
|
switch {
|
|
|
|
|
|
case r >= hangulS && r < hangulS+hangulC:
|
|
|
|
|
|
i := r - hangulS
|
|
|
|
|
|
out = append(out, hangulL+i/hangulN, hangulV+(i%hangulN)/28)
|
|
|
|
|
|
if t := i % 28; t != 0 {
|
|
|
|
|
|
out = append(out, hangulT+t)
|
|
|
|
|
|
}
|
|
|
|
|
|
default:
|
|
|
|
|
|
if d, ok := decompositions.lookup(r); ok {
|
|
|
|
|
|
out = append(out, d...)
|
|
|
|
|
|
} else {
|
|
|
|
|
|
out = append(out, r)
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
// Canonical ordering: insertion sort within each run of non-starters,
|
|
|
|
|
|
// stable, by combining class.
|
|
|
|
|
|
for i := 1; i < len(out); i++ {
|
|
|
|
|
|
c := ccc(out[i])
|
|
|
|
|
|
if c == 0 {
|
|
|
|
|
|
continue
|
|
|
|
|
|
}
|
|
|
|
|
|
for j := i; j > 0; j-- {
|
|
|
|
|
|
p := ccc(out[j-1])
|
|
|
|
|
|
if p <= c {
|
|
|
|
|
|
break
|
|
|
|
|
|
}
|
|
|
|
|
|
out[j-1], out[j] = out[j], out[j-1]
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
return string(out)
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// Fold returns the case folding of s: the mappings of CaseFolding.txt with
|
|
|
|
|
|
// status C or F, and U+0131 (ı) to U+0069 (i), as NTFS does (spec §29.5, R7).
|
|
|
|
|
|
func Fold(s string) string {
|
|
|
|
|
|
var b strings.Builder
|
|
|
|
|
|
for _, r := range s {
|
|
|
|
|
|
if r == 0x0131 {
|
|
|
|
|
|
b.WriteRune(0x0069)
|
|
|
|
|
|
} else if f, ok := foldings.lookup(r); ok {
|
|
|
|
|
|
for _, x := range f {
|
|
|
|
|
|
b.WriteRune(x)
|
|
|
|
|
|
}
|
|
|
|
|
|
} else {
|
|
|
|
|
|
b.WriteRune(r)
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
return b.String()
|
|
|
|
|
|
}
|
|
|
|
|
|
|
pathrule: the simple lower case of Unicode 18.0.0
The key of words lowers each code point with the simple mapping of
UnicodeData.txt, field 13, of Unicode 18.0.0 (spec v0.11, 38.1), not
with unicode.ToLower, which carries Unicode 15.0.0: U+A7CB lowers to
U+0264 in Unicode 16.0 and later, and a key made in the page, whose
platform knows it, would not open in the CLI.
The generator writes the table for Go and for datekeys-ts, and the
canonical text gains its lines, so TablesDigest changes; paths.json and
path_fold.json record the new one.
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
6 days ago
|
|
|
|
// Lower returns the simple lowercase mapping of r in Unicode 18.0.0, field
|
|
|
|
|
|
// 13 of UnicodeData.txt, or r itself. The key of words uses it (spec §38.1).
|
|
|
|
|
|
func Lower(r rune) rune {
|
|
|
|
|
|
if l, ok := lowercases.lookup(r); ok {
|
|
|
|
|
|
return l[0]
|
|
|
|
|
|
}
|
|
|
|
|
|
return r
|
|
|
|
|
|
}
|
|
|
|
|
|
|
Format 3, step 1: Unicode 18.0.0 tables and the path and text rules
internal/pathrule checks the paths and the texts of a format 3 head
(spec 29.5, 29.6) with its own tables, never with the Unicode functions
of the platform, whose version changes with each runtime.
- gen reads the 19 pinned data files (UnicodeData, DerivedCoreProperties,
CaseFolding and emoji-variation-sequences of Unicode 18.0.0, and the
15 WindowsBestFit tables), checks their SHA-256 and writes tables.go:
assigned code points, Default_Ignorable_Code_Point, full canonical
decompositions and combining classes, C and F folding, the bases of
the emoji variation sequences, and the non-ASCII code points each
code page maps to ASCII. The data files stay out of git, in .cache.
- NFD, Fold and the key of R7; CheckPath with R2 to R6c and R10,
CheckTree with R7 and then R9, and CheckComment and CheckAuthor with
the invisible-character rule. Errors name the rule and never echo the
creator's text, so that another implementation can match them.
- The canonical text of the tables has a SHA-256, TablesDigest, which
the tests recompute and a TypeScript implementation will share.
- Checked against golang.org/x/text (Unicode 15.0.0) outside this
module: NFD matches on every code point both know, and folding only
differs on the 86 Cherokee letters that CaseFolding.txt folds to upper
case, as these tables do.
- The spec pins the SHA-256 of the 19 files in 29.5.1.
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
1 week ago
|
|
|
|
// The whitelist of R4: the Default_Ignorable_Code_Point that a path may hold
|
|
|
|
|
|
// (spec §29.5).
|
|
|
|
|
|
const (
|
|
|
|
|
|
zwnj = 0x200C
|
|
|
|
|
|
zwj = 0x200D
|
|
|
|
|
|
vs15 = 0xFE0E
|
|
|
|
|
|
vs16 = 0xFE0F
|
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
func whitelisted(r rune) bool { return r == zwnj || r == zwj || r == vs15 || r == vs16 }
|
|
|
|
|
|
|
|
|
|
|
|
// stripWhitelist removes the whitelisted code points of R4 from s.
|
|
|
|
|
|
func stripWhitelist(s string) string {
|
|
|
|
|
|
return strings.Map(func(r rune) rune {
|
|
|
|
|
|
if whitelisted(r) {
|
|
|
|
|
|
return -1
|
|
|
|
|
|
}
|
|
|
|
|
|
return r
|
|
|
|
|
|
}, s)
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// Key is the key of a segment in R7: NFD(fold(NFD(s′))), where s′ is the
|
|
|
|
|
|
// segment without the whitelisted code points of R4. They are removed
|
|
|
|
|
|
// before normalizing because they have combining class 0: between two
|
|
|
|
|
|
// combining marks, they would change their canonical order.
|
|
|
|
|
|
func Key(segment string) string {
|
|
|
|
|
|
return NFD(Fold(NFD(stripWhitelist(segment))))
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// allowsSelector reports whether base forms an emoji variation sequence with
|
|
|
|
|
|
// the selector sel, U+FE0E or U+FE0F (R4b).
|
|
|
|
|
|
func allowsSelector(base, sel rune) bool {
|
|
|
|
|
|
bases := vs16Bases
|
|
|
|
|
|
if sel == vs15 {
|
|
|
|
|
|
bases = vs15Bases
|
|
|
|
|
|
}
|
|
|
|
|
|
_, ok := slices.BinarySearch(bases, base)
|
|
|
|
|
|
return ok
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// Canonical is the canonical text of the tables, whose SHA-256 is
|
|
|
|
|
|
// TablesDigest; the generator writes the same text from the data files. Each
|
|
|
|
|
|
// value is in upper-case hexadecimal of at least four digits, except the
|
|
|
|
|
|
// combining class and the code page, in decimal, and the best-fit byte, in
|
|
|
|
|
|
// two hexadecimal digits.
|
|
|
|
|
|
func Canonical() string {
|
|
|
|
|
|
var b strings.Builder
|
|
|
|
|
|
fmt.Fprintf(&b, "unicode %s\n", UnicodeVersion)
|
|
|
|
|
|
for i := 0; i < len(assignedRanges); i += 2 {
|
|
|
|
|
|
fmt.Fprintf(&b, "assigned %04X %04X\n", assignedRanges[i], assignedRanges[i+1])
|
|
|
|
|
|
}
|
|
|
|
|
|
for i := 0; i < len(ignorableRanges); i += 2 {
|
|
|
|
|
|
fmt.Fprintf(&b, "ignorable %04X %04X\n", ignorableRanges[i], ignorableRanges[i+1])
|
|
|
|
|
|
}
|
|
|
|
|
|
for _, e := range cccTable {
|
|
|
|
|
|
fmt.Fprintf(&b, "ccc %04X %d\n", e>>8, e&0xFF)
|
|
|
|
|
|
}
|
|
|
|
|
|
for _, m := range []struct {
|
|
|
|
|
|
name string
|
|
|
|
|
|
t *mapping
|
pathrule: the simple lower case of Unicode 18.0.0
The key of words lowers each code point with the simple mapping of
UnicodeData.txt, field 13, of Unicode 18.0.0 (spec v0.11, 38.1), not
with unicode.ToLower, which carries Unicode 15.0.0: U+A7CB lowers to
U+0264 in Unicode 16.0 and later, and a key made in the page, whose
platform knows it, would not open in the CLI.
The generator writes the table for Go and for datekeys-ts, and the
canonical text gains its lines, so TablesDigest changes; paths.json and
path_fold.json record the new one.
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
6 days ago
|
|
|
|
}{{"decomp", &decompositions}, {"fold", &foldings}, {"lower", &lowercases}} {
|
Format 3, step 1: Unicode 18.0.0 tables and the path and text rules
internal/pathrule checks the paths and the texts of a format 3 head
(spec 29.5, 29.6) with its own tables, never with the Unicode functions
of the platform, whose version changes with each runtime.
- gen reads the 19 pinned data files (UnicodeData, DerivedCoreProperties,
CaseFolding and emoji-variation-sequences of Unicode 18.0.0, and the
15 WindowsBestFit tables), checks their SHA-256 and writes tables.go:
assigned code points, Default_Ignorable_Code_Point, full canonical
decompositions and combining classes, C and F folding, the bases of
the emoji variation sequences, and the non-ASCII code points each
code page maps to ASCII. The data files stay out of git, in .cache.
- NFD, Fold and the key of R7; CheckPath with R2 to R6c and R10,
CheckTree with R7 and then R9, and CheckComment and CheckAuthor with
the invisible-character rule. Errors name the rule and never echo the
creator's text, so that another implementation can match them.
- The canonical text of the tables has a SHA-256, TablesDigest, which
the tests recompute and a TypeScript implementation will share.
- Checked against golang.org/x/text (Unicode 15.0.0) outside this
module: NFD matches on every code point both know, and folding only
differs on the 86 Cherokee letters that CaseFolding.txt folds to upper
case, as these tables do.
- The spec pins the SHA-256 of the 19 files in 29.5.1.
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
1 week ago
|
|
|
|
for i, k := range m.t.keys {
|
|
|
|
|
|
fmt.Fprintf(&b, "%s %04X", m.name, k)
|
|
|
|
|
|
for _, r := range m.t.data[m.t.start[i]:m.t.start[i+1]] {
|
|
|
|
|
|
fmt.Fprintf(&b, " %04X", r)
|
|
|
|
|
|
}
|
|
|
|
|
|
b.WriteByte('\n')
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
for _, r := range vs15Bases {
|
|
|
|
|
|
fmt.Fprintf(&b, "vs15 %04X\n", r)
|
|
|
|
|
|
}
|
|
|
|
|
|
for _, r := range vs16Bases {
|
|
|
|
|
|
fmt.Fprintf(&b, "vs16 %04X\n", r)
|
|
|
|
|
|
}
|
|
|
|
|
|
for _, t := range bestFitTables {
|
|
|
|
|
|
for i, r := range t.from {
|
|
|
|
|
|
fmt.Fprintf(&b, "bestfit %s %04X %02X\n", t.name, r, t.to[i])
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
return b.String()
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// digest is the SHA-256 of Canonical, in hexadecimal.
|
|
|
|
|
|
func digest() string {
|
|
|
|
|
|
sum := sha256.Sum256([]byte(Canonical()))
|
|
|
|
|
|
return hex.EncodeToString(sum[:])
|
|
|
|
|
|
}
|