You can not select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
DateKeys/internal/pathrule/unicode.go

258 lines
6.7 KiB

1 week ago
// Package pathrule checks the paths and the texts of the head of a format 3
// capsule (spec §29.5, §29.6) with fixed Unicode 18.0.0 and WindowsBestFit
// tables (spec §29.5.1), never with the Unicode functions of the platform,
// whose version changes with each runtime: Go 1.26.8 carries Unicode 15.0.0,
// and Node 24.9, 16.0.
//
// The tables are generated by internal/pathrule/gen; see tables.go.
package pathrule
//go:generate go run ./gen -data ../../.cache -go tables.go
import (
"crypto/sha256"
"encoding/hex"
"fmt"
"slices"
"strings"
)
// mapping is a sorted table from a code point to a sequence of code points:
// the value of keys[i] is data[start[i]:start[i+1]].
type mapping struct {
keys []rune
start []uint16
data []rune
}
func (m *mapping) lookup(r rune) ([]rune, bool) {
i, ok := slices.BinarySearch(m.keys, r)
if !ok {
return nil, false
}
return m.data[m.start[i]:m.start[i+1]], true
}
// bestFitTable is one code page of WindowsBestFit, reduced to the non-ASCII
// code points it maps to an ASCII byte.
type bestFitTable struct {
name string
from []rune
to []byte
}
func (t *bestFitTable) lookup(r rune) (byte, bool) {
i, ok := slices.BinarySearch(t.from, r)
if !ok {
return 0, false
}
return t.to[i], true
}
// inRanges reports whether r lies in one of the inclusive ranges lo, hi, ...
func inRanges(ranges []rune, r rune) bool {
// The first range whose hi is >= r.
lo, hi := 0, len(ranges)/2
for lo < hi {
m := int(uint(lo+hi) >> 1)
if ranges[2*m+1] < r {
lo = m + 1
} else {
hi = m
}
}
return lo < len(ranges)/2 && ranges[2*lo] <= r
}
// Assigned reports whether r is an assigned code point of Unicode 18.0.0: a
// code point that is not of general category Cn.
func Assigned(r rune) bool { return inRanges(assignedRanges, r) }
// DefaultIgnorable reports whether r has the property
// Default_Ignorable_Code_Point in Unicode 18.0.0.
func DefaultIgnorable(r rune) bool { return inRanges(ignorableRanges, r) }
// ccc is the canonical combining class of r.
func ccc(r rune) uint8 {
i, ok := slices.BinarySearchFunc(cccTable, r, func(e uint32, r rune) int {
switch c := rune(e >> 8); {
case c < r:
return -1
case c > r:
return 1
}
return 0
})
if !ok {
return 0
}
return uint8(cccTable[i])
}
// Hangul syllable decomposition (Unicode §3.12).
const (
hangulS = 0xAC00
hangulL = 0x1100
hangulV = 0x1161
hangulT = 0x11A7
hangulN = 21 * 28
hangulC = 19 * hangulN
)
// NFD returns the canonical decomposition of s, normalization form D of UAX
// #15 with the data of Unicode 18.0.0: every code point replaced by its full
// canonical decomposition, Hangul syllables decomposed algorithmically, and
// each run of combining marks put in canonical order, a stable sort by
// canonical combining class.
func NFD(s string) string {
var out []rune
for _, r := range s {
switch {
case r >= hangulS && r < hangulS+hangulC:
i := r - hangulS
out = append(out, hangulL+i/hangulN, hangulV+(i%hangulN)/28)
if t := i % 28; t != 0 {
out = append(out, hangulT+t)
}
default:
if d, ok := decompositions.lookup(r); ok {
out = append(out, d...)
} else {
out = append(out, r)
}
}
}
// Canonical ordering: insertion sort within each run of non-starters,
// stable, by combining class.
for i := 1; i < len(out); i++ {
c := ccc(out[i])
if c == 0 {
continue
}
for j := i; j > 0; j-- {
p := ccc(out[j-1])
if p <= c {
break
}
out[j-1], out[j] = out[j], out[j-1]
}
}
return string(out)
}
// Fold returns the case folding of s: the mappings of CaseFolding.txt with
// status C or F, and U+0131 (ı) to U+0069 (i), as NTFS does (spec §29.5, R7).
func Fold(s string) string {
var b strings.Builder
for _, r := range s {
if r == 0x0131 {
b.WriteRune(0x0069)
} else if f, ok := foldings.lookup(r); ok {
for _, x := range f {
b.WriteRune(x)
}
} else {
b.WriteRune(r)
}
}
return b.String()
}
// Lower returns the simple lowercase mapping of r in Unicode 18.0.0, field
// 13 of UnicodeData.txt, or r itself. The key of words uses it (spec §38.1).
func Lower(r rune) rune {
if l, ok := lowercases.lookup(r); ok {
return l[0]
}
return r
}
1 week ago
// The whitelist of R4: the Default_Ignorable_Code_Point that a path may hold
// (spec §29.5).
const (
zwnj = 0x200C
zwj = 0x200D
vs15 = 0xFE0E
vs16 = 0xFE0F
)
func whitelisted(r rune) bool { return r == zwnj || r == zwj || r == vs15 || r == vs16 }
// stripWhitelist removes the whitelisted code points of R4 from s.
func stripWhitelist(s string) string {
return strings.Map(func(r rune) rune {
if whitelisted(r) {
return -1
}
return r
}, s)
}
// Key is the key of a segment in R7: NFD(fold(NFD(s′))), where s′ is the
// segment without the whitelisted code points of R4. They are removed
// before normalizing because they have combining class 0: between two
// combining marks, they would change their canonical order.
func Key(segment string) string {
return NFD(Fold(NFD(stripWhitelist(segment))))
}
// allowsSelector reports whether base forms an emoji variation sequence with
// the selector sel, U+FE0E or U+FE0F (R4b).
func allowsSelector(base, sel rune) bool {
bases := vs16Bases
if sel == vs15 {
bases = vs15Bases
}
_, ok := slices.BinarySearch(bases, base)
return ok
}
// Canonical is the canonical text of the tables, whose SHA-256 is
// TablesDigest; the generator writes the same text from the data files. Each
// value is in upper-case hexadecimal of at least four digits, except the
// combining class and the code page, in decimal, and the best-fit byte, in
// two hexadecimal digits.
func Canonical() string {
var b strings.Builder
fmt.Fprintf(&b, "unicode %s\n", UnicodeVersion)
for i := 0; i < len(assignedRanges); i += 2 {
fmt.Fprintf(&b, "assigned %04X %04X\n", assignedRanges[i], assignedRanges[i+1])
}
for i := 0; i < len(ignorableRanges); i += 2 {
fmt.Fprintf(&b, "ignorable %04X %04X\n", ignorableRanges[i], ignorableRanges[i+1])
}
for _, e := range cccTable {
fmt.Fprintf(&b, "ccc %04X %d\n", e>>8, e&0xFF)
}
for _, m := range []struct {
name string
t *mapping
}{{"decomp", &decompositions}, {"fold", &foldings}, {"lower", &lowercases}} {
1 week ago
for i, k := range m.t.keys {
fmt.Fprintf(&b, "%s %04X", m.name, k)
for _, r := range m.t.data[m.t.start[i]:m.t.start[i+1]] {
fmt.Fprintf(&b, " %04X", r)
}
b.WriteByte('\n')
}
}
for _, r := range vs15Bases {
fmt.Fprintf(&b, "vs15 %04X\n", r)
}
for _, r := range vs16Bases {
fmt.Fprintf(&b, "vs16 %04X\n", r)
}
for _, t := range bestFitTables {
for i, r := range t.from {
fmt.Fprintf(&b, "bestfit %s %04X %02X\n", t.name, r, t.to[i])
}
}
return b.String()
}
// digest is the SHA-256 of Canonical, in hexadecimal.
func digest() string {
sum := sha256.Sum256([]byte(Canonical()))
return hex.EncodeToString(sum[:])
}

Powered by TurnKey Linux.