Format 3, step 1: Unicode 18.0.0 tables and the path and text rules

internal/pathrule checks the paths and the texts of a format 3 head
(spec 29.5, 29.6) with its own tables, never with the Unicode functions
of the platform, whose version changes with each runtime.

- gen reads the 19 pinned data files (UnicodeData, DerivedCoreProperties,
  CaseFolding and emoji-variation-sequences of Unicode 18.0.0, and the
  15 WindowsBestFit tables), checks their SHA-256 and writes tables.go:
  assigned code points, Default_Ignorable_Code_Point, full canonical
  decompositions and combining classes, C and F folding, the bases of
  the emoji variation sequences, and the non-ASCII code points each
  code page maps to ASCII. The data files stay out of git, in .cache.
- NFD, Fold and the key of R7; CheckPath with R2 to R6c and R10,
  CheckTree with R7 and then R9, and CheckComment and CheckAuthor with
  the invisible-character rule. Errors name the rule and never echo the
  creator's text, so that another implementation can match them.
- The canonical text of the tables has a SHA-256, TablesDigest, which
  the tests recompute and a TypeScript implementation will share.
- Checked against golang.org/x/text (Unicode 15.0.0) outside this
  module: NFD matches on every code point both know, and folding only
  differs on the 86 Cherokee letters that CaseFolding.txt folds to upper
  case, as these tables do.
- The spec pins the SHA-256 of the 19 files in 29.5.1.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
v0.10
dev 1 week ago
parent 631d09ca66
commit 6fba1ddc01

3
.gitignore vendored

@ -7,3 +7,6 @@
# Coverage and profiles
*.out
*.cdx.json
# Downloaded source data (Unicode, WindowsBestFit); only generated tables are committed
/.cache/

@ -0,0 +1,573 @@
// Command gen generates the Unicode and best-fit tables of package pathrule
// (spec §29.5.1) from the data files of Unicode 18.0.0 and WindowsBestFit:
//
// go run ./internal/pathrule/gen -data .cache
//
// It reads, below the -data directory,
//
// unicode/18.0.0/UnicodeData.txt
// unicode/18.0.0/DerivedCoreProperties.txt
// unicode/18.0.0/CaseFolding.txt
// unicode/18.0.0/emoji/emoji-variation-sequences.txt
// bestfit/bestfit<N>.txt, for the 15 code pages of WindowsBestFit
//
// checks the SHA-256 of each against the value pinned below, and writes
// internal/pathrule/tables.go. The files are not in the repository: only the
// generated tables are. Download them from https://www.unicode.org/Public/.
//
// The tables are also described by a canonical text, whose SHA-256 is
// TablesDigest: an implementation in another language generated from the
// same files computes the same digest (see canonical in package pathrule).
package main
import (
"bufio"
"bytes"
"crypto/sha256"
"encoding/hex"
"flag"
"fmt"
"go/format"
"os"
"path/filepath"
"slices"
"strconv"
"strings"
)
// UnicodeVersion is the version of Unicode the tables come from (spec
// §29.5.1). It is fixed with capsule format 3.
const UnicodeVersion = "18.0.0"
// source is one data file and the SHA-256 it must have.
type source struct {
path, sha256 string
}
// The pinned files (spec §29.5.1).
var (
unicodeData = source{"unicode/18.0.0/UnicodeData.txt", "0736451de439ae7baf1425136617da495e09ee5afbe6e394374db7009ea08950"}
derivedCore = source{"unicode/18.0.0/DerivedCoreProperties.txt", "09c928886a178fcafd93c29e4bd59073a058e5a100b716d425cb563ab50f68c9"}
caseFolding = source{"unicode/18.0.0/CaseFolding.txt", "a004797658a457bec4dc11683e39f69249ea3b595b752dbea6721c4c9f587b0d"}
emojiVariants = source{"unicode/18.0.0/emoji/emoji-variation-sequences.txt", "ff1707564aa1f1b2fcf4ec92d609d4cb26940bc0d4dcf07f5328ad6879e84da3"}
bestFitFiles = []source{
{"bestfit/bestfit874.txt", "663f43ca662e037c4534cb16298b560f29ce29c27b49b3589601ec3d97dd89fd"},
{"bestfit/bestfit932.txt", "2614cfea35c3c86c41d33198793a84ca44edee3cf0ee0013a61a43fba4ece331"},
{"bestfit/bestfit936.txt", "e5070a2d6ad26619f5872ddbe64d3381c11620af5adbb04cda0f0abb1a91fdae"},
{"bestfit/bestfit949.txt", "50e13b60ea8fda66a8223ecc85270e0f182303222244e2345d3d57f3e839d20a"},
{"bestfit/bestfit950.txt", "cf8c23389a42a226ea707f7ec32c665556d1fc3364db25bd765ce64d54eaee2a"},
{"bestfit/bestfit1250.txt", "cef9f171e67b09445bcb3f9ffccdc89418250ff825f1bd2d29a92d2074d7a53b"},
{"bestfit/bestfit1251.txt", "59ec85612ff908d9da0e877893c935941e56b13a2882b4fb9c9599be3d1ce4e7"},
{"bestfit/bestfit1252.txt", "72ea23c939c5b26fae7aded0207b327e2f3902d7d3c168d7087f5cfc38ee76a9"},
{"bestfit/bestfit1253.txt", "ea80c442aff7f09b36da6335f85f8e527f51c146beeb9825ec00d1b6ca99a99e"},
{"bestfit/bestfit1254.txt", "3d02512087634dc493b720992b590277736ffb2d5b0b665d69b6b9727e2c361a"},
{"bestfit/bestfit1255.txt", "fdd4bdda74f6571d89171b0070ac052cd3714c395dc3d1799bcd5e4a4da6f83a"},
{"bestfit/bestfit1256.txt", "745c447ada04a838da8bea406c13f446c7453b6371e8c6c7863a632443d56007"},
{"bestfit/bestfit1257.txt", "b8c5d7f3b8c25c3d5625d44dd3d6ee7a06e652ddf77373d050282c1cb7517366"},
{"bestfit/bestfit1258.txt", "5d52a9357b7d6b5b5014ed5a51be0ff9809b0c33625793d2a4feaf502e0682f1"},
{"bestfit/bestfit1361.txt", "7dcda2d5d2cfc5ddf43757d589a0106e020ef0d9b84de47f08e18297ae0fe1ec"},
}
)
func main() {
data := flag.String("data", ".cache", "directory with the downloaded data files")
out := flag.String("go", "internal/pathrule/tables.go", "Go file to write")
flag.Parse()
t, err := build(*data)
if err != nil {
fmt.Fprintln(os.Stderr, "gen:", err)
os.Exit(1)
}
src, err := t.goSource()
if err != nil {
fmt.Fprintln(os.Stderr, "gen:", err)
os.Exit(1)
}
if err := os.WriteFile(*out, src, 0o644); err != nil {
fmt.Fprintln(os.Stderr, "gen:", err)
os.Exit(1)
}
fmt.Printf("wrote %s: %d assigned ranges, %d ignorable ranges, %d ccc, %d decompositions, %d foldings, %d+%d emoji bases, %d best-fit tables; digest %s\n",
*out, len(t.assigned), len(t.ignorable), len(t.ccc), len(t.decomp), len(t.fold), len(t.vs15), len(t.vs16), len(t.bestFit), t.digest())
}
// tables are the data package pathrule needs, in canonical order.
type tables struct {
assigned [][2]rune // ranges of assigned code points
ignorable [][2]rune // ranges of Default_Ignorable_Code_Point
ccc map[rune]uint8 // canonical combining class, non-zero only
decomp map[rune][]rune // full canonical decomposition, Hangul excluded
fold map[rune][]rune // case folding, statuses C and F
vs15 []rune // bases of an emoji variation sequence with U+FE0E
vs16 []rune // bases of an emoji variation sequence with U+FE0F
bestFit []bestFitTable // one per code page
sources []source // the files, for the record
}
type bestFitTable struct {
name string // code page, "874"
to map[rune]byte // non-ASCII code point -> ASCII byte
}
// read returns the contents of a data file after checking its digest.
func read(dir string, s source) ([]byte, error) {
b, err := os.ReadFile(filepath.Join(dir, filepath.FromSlash(s.path)))
if err != nil {
return nil, err
}
sum := sha256.Sum256(b)
if got := hex.EncodeToString(sum[:]); got != s.sha256 {
return nil, fmt.Errorf("%s: SHA-256 %s, want %s", s.path, got, s.sha256)
}
return b, nil
}
func build(dir string) (*tables, error) {
t := &tables{ccc: map[rune]uint8{}, decomp: map[rune][]rune{}, fold: map[rune][]rune{}}
t.sources = append([]source{unicodeData, derivedCore, caseFolding, emojiVariants}, bestFitFiles...)
b, err := read(dir, unicodeData)
if err != nil {
return nil, err
}
if err := t.parseUnicodeData(b); err != nil {
return nil, fmt.Errorf("%s: %w", unicodeData.path, err)
}
if b, err = read(dir, derivedCore); err != nil {
return nil, err
}
if t.ignorable, err = parseProperty(b, "Default_Ignorable_Code_Point"); err != nil {
return nil, fmt.Errorf("%s: %w", derivedCore.path, err)
}
if b, err = read(dir, caseFolding); err != nil {
return nil, err
}
if err := t.parseCaseFolding(b); err != nil {
return nil, fmt.Errorf("%s: %w", caseFolding.path, err)
}
if b, err = read(dir, emojiVariants); err != nil {
return nil, err
}
if err := t.parseEmojiVariants(b); err != nil {
return nil, fmt.Errorf("%s: %w", emojiVariants.path, err)
}
for _, s := range bestFitFiles {
if b, err = read(dir, s); err != nil {
return nil, err
}
bt, err := parseBestFit(b)
if err != nil {
return nil, fmt.Errorf("%s: %w", s.path, err)
}
bt.name = strings.TrimSuffix(strings.TrimPrefix(filepath.Base(s.path), "bestfit"), ".txt")
t.bestFit = append(t.bestFit, bt)
}
return t, nil
}
func hexRune(s string) (rune, error) {
v, err := strconv.ParseUint(strings.TrimSpace(s), 16, 32)
if err != nil || v > 0x10FFFF {
return 0, fmt.Errorf("bad code point %q", s)
}
return rune(v), nil
}
func hexRunes(s string) ([]rune, error) {
var out []rune
for _, f := range strings.Fields(s) {
r, err := hexRune(f)
if err != nil {
return nil, err
}
out = append(out, r)
}
return out, nil
}
// lines yields the data lines of a UCD-style file: comments after '#'
// removed, blank lines skipped.
func lines(b []byte, f func(line string) error) error {
sc := bufio.NewScanner(bytes.NewReader(b))
sc.Buffer(make([]byte, 1<<16), 1<<20)
for n := 1; sc.Scan(); n++ {
line := sc.Text()
if i := strings.IndexByte(line, '#'); i >= 0 {
line = line[:i]
}
if strings.TrimSpace(line) == "" {
continue
}
if err := f(line); err != nil {
return fmt.Errorf("line %d: %w", n, err)
}
}
return sc.Err()
}
func (t *tables) parseUnicodeData(b []byte) error {
canonical := map[rune][]rune{}
var assigned []rune
var first rune = -1
err := lines(b, func(line string) error {
f := strings.Split(line, ";")
if len(f) != 15 {
return fmt.Errorf("%d fields", len(f))
}
cp, err := hexRune(f[0])
if err != nil {
return err
}
switch name := f[1]; {
case strings.HasSuffix(name, ", First>"):
first = cp
return nil
case strings.HasSuffix(name, ", Last>"):
if first < 0 {
return fmt.Errorf("range end %04X without start", cp)
}
for r := first; r <= cp; r++ {
assigned = append(assigned, r)
}
first = -1
default:
assigned = append(assigned, cp)
}
if c, err := strconv.ParseUint(f[3], 10, 8); err != nil {
return fmt.Errorf("ccc %q", f[3])
} else if c != 0 {
t.ccc[cp] = uint8(c)
}
if d := f[5]; d != "" && !strings.HasPrefix(d, "<") {
m, err := hexRunes(d)
if err != nil {
return err
}
canonical[cp] = m
}
return nil
})
if err != nil {
return err
}
t.assigned = ranges(assigned)
var expand func(r rune) []rune
expand = func(r rune) []rune {
m, ok := canonical[r]
if !ok {
return []rune{r}
}
var out []rune
for _, x := range m {
out = append(out, expand(x)...)
}
return out
}
for r := range canonical {
t.decomp[r] = expand(r)
}
return nil
}
// ranges turns a set of code points into sorted, merged, inclusive ranges.
func ranges(cps []rune) [][2]rune {
slices.Sort(cps)
cps = slices.Compact(cps)
var out [][2]rune
for _, r := range cps {
if n := len(out); n > 0 && out[n-1][1]+1 == r {
out[n-1][1] = r
continue
}
out = append(out, [2]rune{r, r})
}
return out
}
func parseProperty(b []byte, name string) ([][2]rune, error) {
var cps []rune
err := lines(b, func(line string) error {
f := strings.SplitN(line, ";", 2)
if len(f) != 2 || strings.TrimSpace(f[1]) != name {
return nil
}
lo, hi, ok := strings.Cut(strings.TrimSpace(f[0]), "..")
a, err := hexRune(lo)
if err != nil {
return err
}
z := a
if ok {
if z, err = hexRune(hi); err != nil {
return err
}
}
for r := a; r <= z; r++ {
cps = append(cps, r)
}
return nil
})
if len(cps) == 0 && err == nil {
err = fmt.Errorf("no %s entries", name)
}
return ranges(cps), err
}
func (t *tables) parseCaseFolding(b []byte) error {
return lines(b, func(line string) error {
f := strings.Split(line, ";")
if len(f) < 3 {
return fmt.Errorf("%d fields", len(f))
}
switch strings.TrimSpace(f[1]) {
case "C", "F":
default:
return nil
}
cp, err := hexRune(f[0])
if err != nil {
return err
}
m, err := hexRunes(f[2])
if err != nil {
return err
}
if _, dup := t.fold[cp]; dup {
return fmt.Errorf("two C or F foldings for %04X", cp)
}
t.fold[cp] = m
return nil
})
}
func (t *tables) parseEmojiVariants(b []byte) error {
err := lines(b, func(line string) error {
f := strings.SplitN(line, ";", 2)
seq, err := hexRunes(f[0])
if err != nil {
return err
}
if len(seq) != 2 {
return fmt.Errorf("sequence of %d code points", len(seq))
}
switch seq[1] {
case 0xFE0E:
t.vs15 = append(t.vs15, seq[0])
case 0xFE0F:
t.vs16 = append(t.vs16, seq[0])
default:
return fmt.Errorf("selector %04X", seq[1])
}
return nil
})
slices.Sort(t.vs15)
slices.Sort(t.vs16)
t.vs15, t.vs16 = slices.Compact(t.vs15), slices.Compact(t.vs16)
return err
}
// parseBestFit keeps, from the WCTABLE section, the code points of 0x80 and
// above that the code page maps to a single ASCII byte (spec §29.5, R6c).
func parseBestFit(b []byte) (bestFitTable, error) {
bt := bestFitTable{to: map[rune]byte{}}
sc := bufio.NewScanner(bytes.NewReader(b))
in, seen := false, false
for n := 1; sc.Scan(); n++ {
line := sc.Text()
if i := strings.IndexByte(line, ';'); i >= 0 {
line = line[:i]
}
f := strings.Fields(line)
if len(f) == 0 {
continue
}
if !strings.HasPrefix(f[0], "0x") {
in = f[0] == "WCTABLE"
seen = seen || in
continue
}
if !in {
continue
}
if len(f) < 2 {
return bt, fmt.Errorf("line %d: %q", n, line)
}
u, err1 := strconv.ParseUint(f[0][2:], 16, 32)
v, err2 := strconv.ParseUint(strings.TrimPrefix(f[1], "0x"), 16, 32)
if err1 != nil || err2 != nil || u > 0xFFFF {
return bt, fmt.Errorf("line %d: %q", n, line)
}
if u >= 0x80 && v <= 0x7F {
bt.to[rune(u)] = byte(v)
}
}
if !seen {
return bt, fmt.Errorf("no WCTABLE section")
}
return bt, sc.Err()
}
func sortedKeys[V any](m map[rune]V) []rune {
keys := make([]rune, 0, len(m))
for k := range m {
keys = append(keys, k)
}
slices.Sort(keys)
return keys
}
// canonical is the text whose SHA-256 is TablesDigest. Package pathrule
// rebuilds it from the generated tables, and so does any other
// implementation, line by line: every value in upper-case hexadecimal of at
// least four digits, except the combining class and the code page, in
// decimal.
func (t *tables) canonical() string {
var b strings.Builder
fmt.Fprintf(&b, "unicode %s\n", UnicodeVersion)
for _, r := range t.assigned {
fmt.Fprintf(&b, "assigned %04X %04X\n", r[0], r[1])
}
for _, r := range t.ignorable {
fmt.Fprintf(&b, "ignorable %04X %04X\n", r[0], r[1])
}
for _, k := range sortedKeys(t.ccc) {
fmt.Fprintf(&b, "ccc %04X %d\n", k, t.ccc[k])
}
for _, k := range sortedKeys(t.decomp) {
fmt.Fprintf(&b, "decomp %04X%s\n", k, hexList(t.decomp[k]))
}
for _, k := range sortedKeys(t.fold) {
fmt.Fprintf(&b, "fold %04X%s\n", k, hexList(t.fold[k]))
}
for _, r := range t.vs15 {
fmt.Fprintf(&b, "vs15 %04X\n", r)
}
for _, r := range t.vs16 {
fmt.Fprintf(&b, "vs16 %04X\n", r)
}
for _, bt := range t.bestFit {
for _, k := range sortedKeys(bt.to) {
fmt.Fprintf(&b, "bestfit %s %04X %02X\n", bt.name, k, bt.to[k])
}
}
return b.String()
}
func hexList(rs []rune) string {
var b strings.Builder
for _, r := range rs {
fmt.Fprintf(&b, " %04X", r)
}
return b.String()
}
func (t *tables) digest() string {
sum := sha256.Sum256([]byte(t.canonical()))
return hex.EncodeToString(sum[:])
}
// goSource renders tables.go.
func (t *tables) goSource() ([]byte, error) {
var b bytes.Buffer
w := func(format string, args ...any) { fmt.Fprintf(&b, format, args...) }
w("// Code generated by internal/pathrule/gen from Unicode %s and WindowsBestFit. DO NOT EDIT.\n\n", UnicodeVersion)
w("package pathrule\n\n")
w("// UnicodeVersion is the version of Unicode of these tables, fixed with\n// capsule format 3 (spec §29.5.1).\nconst UnicodeVersion = %q\n\n", UnicodeVersion)
w("// TablesDigest is the SHA-256 of the canonical text of these tables (see\n// Canonical). Another implementation generated from the same files has it too.\nconst TablesDigest = %q\n\n", t.digest())
w("// Sources are the data files of the tables and their SHA-256 (spec §29.5.1).\nvar Sources = []struct{ Path, SHA256 string }{\n")
for _, s := range t.sources {
w("\t{%q, %q},\n", s.path, s.sha256)
}
w("}\n\n")
pairs := func(name, doc string, rs [][2]rune) {
w("// %s\nvar %s = []rune{", doc, name)
for i, r := range rs {
if i%6 == 0 {
w("\n\t")
}
w("0x%04X, 0x%04X, ", r[0], r[1])
}
w("\n}\n\n")
}
pairs("assignedRanges", "assignedRanges are the assigned code points, as inclusive ranges: lo, hi, ...", t.assigned)
pairs("ignorableRanges", "ignorableRanges are Default_Ignorable_Code_Point, as inclusive ranges.", t.ignorable)
w("// cccTable holds each code point with a non-zero canonical combining class,\n// as code point << 8 | class, sorted.\nvar cccTable = []uint32{")
for i, k := range sortedKeys(t.ccc) {
if i%8 == 0 {
w("\n\t")
}
w("0x%06X, ", uint32(k)<<8|uint32(t.ccc[k]))
}
w("\n}\n\n")
mapping := func(name, doc string, m map[rune][]rune) {
keys := sortedKeys(m)
w("// %s\nvar %s = mapping{\n\tkeys: []rune{", doc, name)
for i, k := range keys {
if i%10 == 0 {
w("\n\t\t")
}
w("0x%04X, ", k)
}
w("\n\t},\n\tstart: []uint16{")
n := 0
for i, k := range keys {
if i%12 == 0 {
w("\n\t\t")
}
w("%d, ", n)
n += len(m[k])
}
w("%d,\n\t},\n\tdata: []rune{", n)
i := 0
for _, k := range keys {
for _, r := range m[k] {
if i%10 == 0 {
w("\n\t\t")
}
w("0x%04X, ", r)
i++
}
}
w("\n\t},\n}\n\n")
}
mapping("decompositions", "decompositions is the full canonical decomposition of each code point that has\n// one, Hangul syllables excepted.", t.decomp)
mapping("foldings", "foldings is the case folding of CaseFolding.txt, statuses C and F.", t.fold)
set := func(name, doc string, rs []rune) {
w("// %s\nvar %s = []rune{", doc, name)
for i, r := range rs {
if i%10 == 0 {
w("\n\t")
}
w("0x%04X, ", r)
}
w("\n}\n\n")
}
set("vs15Bases", "vs15Bases are the characters of an emoji variation sequence with U+FE0E.", t.vs15)
set("vs16Bases", "vs16Bases are the characters of an emoji variation sequence with U+FE0F.", t.vs16)
w("// bestFitTables are, for each code page of WindowsBestFit, the code points\n// of U+0080 and above that it maps to a single ASCII byte.\nvar bestFitTables = []bestFitTable{\n")
for _, bt := range t.bestFit {
keys := sortedKeys(bt.to)
w("\t{\n\t\tname: %q,\n\t\tfrom: []rune{", bt.name)
for i, k := range keys {
if i%10 == 0 {
w("\n\t\t\t")
}
w("0x%04X, ", k)
}
w("\n\t\t},\n\t\tto: []byte{")
for i, k := range keys {
if i%16 == 0 {
w("\n\t\t\t")
}
w("0x%02X, ", bt.to[k])
}
w("\n\t\t},\n\t},\n")
}
w("}\n")
return format.Source(b.Bytes())
}

@ -0,0 +1,268 @@
package pathrule
import (
"errors"
"fmt"
"strings"
"testing"
)
// The generated tables match their canonical text: nobody edited them.
func TestTablesDigest(t *testing.T) {
if got := digest(); got != TablesDigest {
t.Fatalf("digest of the tables %s, TablesDigest %s", got, TablesDigest)
}
if UnicodeVersion != "18.0.0" {
t.Fatalf("UnicodeVersion %q", UnicodeVersion)
}
if len(Sources) != 19 || len(bestFitTables) != 15 {
t.Fatalf("%d sources, %d best-fit tables", len(Sources), len(bestFitTables))
}
}
func TestProperties(t *testing.T) {
for _, c := range []struct {
r rune
assigned, ignorable bool
}{
{'a', true, false},
{0x0378, false, false}, // unassigned
{0x00AD, true, true}, // soft hyphen
{0x200D, true, true}, // ZWJ
{0xFE0F, true, true}, // VS16
{0x206A, true, true}, // deprecated format character
{0xE0041, true, true}, // tag A
{0xE0000, false, true}, // unassigned, still ignorable
{0x2028, true, false}, // line separator
{0xFFFE, false, false}, // noncharacter
{0xF03A, true, false}, // private use
{0x10FFFF, false, false},
} {
if Assigned(c.r) != c.assigned || DefaultIgnorable(c.r) != c.ignorable {
t.Errorf("U+%04X: assigned %v ignorable %v, want %v %v", c.r, Assigned(c.r), DefaultIgnorable(c.r), c.assigned, c.ignorable)
}
}
}
func TestNFD(t *testing.T) {
for _, c := range []struct{ in, want string }{
{"é", "é"},
{"Å", "Å"}, // ANGSTROM SIGN, a singleton decomposition
{"K", "K"}, // KELVIN SIGN
{"한", "한"}, // Hangul, with a final jamo
{"하", "하"}, // Hangul, without
{"ạ́", "ạ́"}, // canonical order
{"ΐ", "ΐ"},
{"ǖ", "ǖ"}, // recursive decomposition
{"á‍̣", "á‍̣"}, // ZWJ is a starter
} {
if got := NFD(c.in); got != c.want {
t.Errorf("NFD(%+q) = %+q, want %+q", c.in, got, c.want)
}
}
}
func TestKey(t *testing.T) {
for _, same := range [][2]string{
{"A.txt", "a.txt"},
{"Straße", "STRASSE"},
{"K", "k"},
{"ı", "i"},
{"é", "é"},
{"ab", "a‌b"},
{"á‍̣", "ạ́"},
{".DateKeys-x", ".datekeys-x"},
} {
if Key(same[0]) != Key(same[1]) {
t.Errorf("Key(%+q) = %+q, Key(%+q) = %+q", same[0], Key(same[0]), same[1], Key(same[1]))
}
}
if Key("a") == Key("b") {
t.Error("distinct keys expected")
}
}
// rule is the rule of a pathrule error, "" for nil.
func rule(err error) string {
var e *Error
if errors.As(err, &e) {
return e.Rule
}
if err != nil {
return "?" + err.Error()
}
return ""
}
func TestCheckPath(t *testing.T) {
const (
zwj = "‍"
zwnj = "‌"
vs16 = "️"
)
scotland := "\U0001F3F4\U000E0067\U000E0062\U000E0073\U000E0063\U000E0074\U000E007F"
for _, c := range []struct {
path, rule string
}{
// Accepted.
{"nota.txt", ""},
{"fotos/playa.jpg", ""},
{"¿Qué es esto.jpg", ""}, // bestfit1250 maps ¿ to '?', which R6c no longer rejects
{"Para ti ♥.jpg", ""}, // bestfit874 maps ♥ to 0x03
{"§ 3 contrato.pdf", ""}, // bestfit874 maps § to 0x15
{"Madrid → Lisboa", ""}, // bestfit1253 maps → to '>'
{"❤" + vs16 + ".txt", ""}, // ❤️
{"\U0001F3F3" + vs16 + zwj + "\U0001F308", ""}, // rainbow flag
{"\U0001F468" + zwj + "\U0001F469" + zwj + "\U0001F467", ""},
{"ab" + zwnj + "c", ""},
{"a/.datekeys-x", ""}, // only the first segment is reserved
{"ABCDEFGHI~1", ""}, // longer than 8 before the dot
{"a~1b", ""},
{"report~2023.txt", ""},
{strings.Repeat("a", 255), ""},
{strings.Repeat("ΐ", 85), ""}, // 170 bytes, 255 UTF-16 units after NFD
{" a ", ""}, // no table maps U+00A0 to U+0020... checked below
// R2.
{"/a", "R2"},
{"a//b", "R2"},
{"a/", "R2"},
{strings.Repeat("a/", 32) + "a", "R2"},
// R3.
{"..", "R3"},
{"a/./b", "R3"},
{zwj, "R4b"}, // R3 sees an empty stripped segment? No: R3 first.
{strings.Repeat("a", 256), "R3"},
{strings.Repeat("ΐ", 127), "R3"}, // 254 bytes, 381 UTF-16 units after NFD
// R4.
{"a:b", "R4"},
{"a\\b", "R4"},
{"a?b", "R4"},
{"a\tb", "R4"},
{"a
b", "R4"},
{"a­b", "R4"},
{"ab", "R4"},
{"a‮b", "R4"},
{"informeanexo", "R4"},
{"a͸b", "R4"},
{"a￾b", "R4"},
{scotland, "R4"},
{"\U0001F600\U000E0100", "R4"}, // VS17
// R4b.
{"a" + vs16, "R4b"},
{zwj + "a", "R4b"},
{"a" + zwj, "R4b"},
{"a" + zwj + zwj + "b", "R4b"},
{"a" + zwnj + zwj + "b", "R4b"},
// R5.
{" a", "R5"},
{"a ", "R5"},
{"a.", "R5"},
// R6.
{"CON.txt", "R6"},
{"con", "R6"},
{"Aux .log", "R6"},
{"COM¹", "R6"},
{"lpt9.doc", "R6"},
{"CONIN$", "R6"},
// R6b.
{"ABCDEF~1", "R6b"},
{"ABCDEF~1.TXT", "R6b"},
{"~1", "R6b"},
{"Ä~1.txt", "R6b"},
// R6c.
{"CON.txt", "R6c"}, // full-width CON
{"a∖b", "R6c"}, // SET MINUS
{"a∶b", "R6c"}, // RATIO
{" a", "R6c"}, // bestfit1252 maps U+3000 to U+0020, which breaks R5
{"a/b", "R6c"}, // full-width solidus
{"..", "R6c"}, // full-width dots project to ".."
// R10.
{".datekeys-x", "R10"},
{".DateKeys-abc/b", "R10"},
} {
got := rule(CheckPath(c.path))
if c.path == zwj {
// R3 comes before R4b in a segment: the stripped segment is empty.
if got != "R3" {
t.Errorf("CheckPath(%+q): %s, want R3", c.path, got)
}
continue
}
if c.path == " a " {
// Whatever the tables give, it must be R6c or nothing (R5 is
// about U+0020 only).
if got != "" && got != "R6c" {
t.Errorf("CheckPath(%+q): %s", c.path, got)
}
continue
}
if got != c.rule {
t.Errorf("CheckPath(%+q): %s (%v), want %s", c.path, got, CheckPath(c.path), c.rule)
}
}
}
func TestCheckTree(t *testing.T) {
for _, c := range []struct {
paths []string
rule string
}{
{[]string{"a/b", "a/c", "d"}, ""},
{[]string{"A.txt", "a.txt"}, "R7"},
{[]string{"A", "a/b"}, "R7"},
{[]string{"a/b", "a/b/c"}, "R7"},
{[]string{"Fotos/b", "fotos/a"}, "R7"},
{[]string{"STRASSE", "Straße"}, "R7"},
{[]string{"ab", "a‌b"}, "R7"},
{[]string{"k", "K"}, "R7"},
} {
if got := rule(CheckTree(c.paths)); got != c.rule {
t.Errorf("CheckTree(%+q): %s, want %s", c.paths, got, c.rule)
}
}
many := make([]string, MaxImplicitDirs+1)
for i := range many {
many[i] = fmt.Sprintf("d%05d/f", i)
}
if got := rule(CheckTree(many)); got != "R9" {
t.Errorf("%d folders: %s, want R9", len(many), got)
}
if got := rule(CheckTree(many[:MaxImplicitDirs])); got != "" {
t.Errorf("%d folders: %s", MaxImplicitDirs, got)
}
}
func TestTexts(t *testing.T) {
for _, c := range []struct {
text string
comment bool
rule string
}{
{"Hola\tmundo\nsegunda línea", true, ""},
{"❤️ para ti", true, ""},
{"Hola\r\n", true, "text"},
{"a‮b", true, "text"},
{"Hola\U000E0049\U000E0047", true, "text"}, // tag characters
{"\U0001F600\U000E0100", true, "text"}, // VS17
{"a️", true, "text"}, // VS16 after a letter
{"a‍\nb", true, "text"}, // ZWJ at the end of a line
{"a￾", true, "text"},
{"a﷐", true, "text"},
{"a­b", true, "text"},
{"Ana López", false, ""},
{"Ana\tLópez", false, "text"},
{"Ana\nLópez", false, "text"},
{" Ana", false, "text"},
{"Ana ", false, "text"},
{"a‎b", false, "text"},
} {
check := CheckAuthor
if c.comment {
check = CheckComment
}
if got := rule(check(c.text)); got != c.rule {
t.Errorf("%+q (comment %v): %s, want %s", c.text, c.comment, got, c.rule)
}
}
}

@ -0,0 +1,392 @@
package pathrule
import (
"fmt"
"regexp"
"strings"
"unicode/utf8"
)
// Limits of the paths and texts of a format 3 head, fixed with the format
// (spec §29.4, §29.5).
const (
MaxPathLen = 1024 // R1, bytes
MaxSegments = 32 // R2
MaxSegmentLen = 255 // R3, bytes
MaxSegmentUTF16 = 255 // R3, UTF-16 code units of the NFD of the segment
MaxImplicitDirs = 65535 // R9
MaxCommentLen = 16384 // bytes
MaxAuthorLen = 256 // bytes
reservedDirPrefix = ".datekeys-"
)
// Error is the violation of one rule of spec §29.5 or §29.6. Its text is
// the same in every implementation, so that the message of a rejected head
// does not depend on the reader.
type Error struct {
Rule string // "R2" to "R10", "text"
Detail string
}
func (e *Error) Error() string { return e.Rule + ": " + e.Detail }
func fail(rule, format string, args ...any) error {
return &Error{Rule: rule, Detail: fmt.Sprintf(format, args...)}
}
// codePoint formats r as in the spec: U+ and at least four upper-case
// hexadecimal digits.
func codePoint(r rune) string { return fmt.Sprintf("U+%04X", r) }
// forbiddenASCII are the ASCII characters of R4 that are not controls.
const forbiddenASCII = `"*:<>?\|`
// CheckPath checks one path of the head with the rules of the fourth layer
// that concern it alone, in this order: R2, and then for each segment R3,
// R4, R4b, R5, R6, R6b and R6c; then R10. R1 and R8 belong to the third
// layer, and R7 and R9 to the whole tree (CheckTree). The first violation
// is returned.
func CheckPath(path string) error {
segs := strings.Split(path, "/")
if len(segs) > MaxSegments {
return fail("R2", "%d segments, more than %d", len(segs), MaxSegments)
}
for i, s := range segs {
if s == "" {
return fail("R2", "segment %d is empty", i+1)
}
}
for i, s := range segs {
if err := checkSegment(s); err != nil {
e := err.(*Error)
e.Detail = fmt.Sprintf("segment %d: %s", i+1, e.Detail)
return e
}
}
if strings.HasPrefix(Key(segs[0]), reservedDirPrefix) {
return fail("R10", "the first segment starts with %q", reservedDirPrefix)
}
return nil
}
// checkSegment applies R3, R4, R4b, R5, R6, R6b and R6c to a segment.
func checkSegment(s string) error {
if err := checkR3(s); err != nil {
return err
}
for _, r := range s {
if err := checkR4(r); err != nil {
return err
}
}
if err := checkPlacement("R4b", s); err != nil {
return err
}
if err := checkR5(s); err != nil {
return err
}
if err := checkR6(s); err != nil {
return err
}
if err := checkR6b(s); err != nil {
return err
}
return checkR6c(s)
}
func checkR3(s string) error {
if len(s) > MaxSegmentLen {
return fail("R3", "%d bytes, more than %d", len(s), MaxSegmentLen)
}
switch s {
case ".":
return fail("R3", "the segment is a dot")
case "..":
return fail("R3", "the segment is two dots")
}
switch stripWhitelist(s) {
case "":
return fail("R3", "the segment is empty without ZWNJ, ZWJ, VS15 and VS16")
case ".":
return fail("R3", "the segment is a dot without ZWNJ, ZWJ, VS15 and VS16")
case "..":
return fail("R3", "the segment is two dots without ZWNJ, ZWJ, VS15 and VS16")
}
if n := utf16Len(NFD(s)); n > MaxSegmentUTF16 {
return fail("R3", "its NFD is %d UTF-16 code units, more than %d", n, MaxSegmentUTF16)
}
return nil
}
func utf16Len(s string) int {
n := 0
for _, r := range s {
if r >= 0x10000 {
n += 2
} else {
n++
}
}
return n
}
// checkR4 rejects the code points of R4.
func checkR4(r rune) error {
switch {
case r <= 0x1F || (r >= 0x7F && r <= 0x9F):
return fail("R4", "control %s", codePoint(r))
case strings.ContainsRune(forbiddenASCII, r):
return fail("R4", "character %s", codePoint(r))
case r == 0x2028 || r == 0x2029:
return fail("R4", "separator %s", codePoint(r))
case DefaultIgnorable(r) && !whitelisted(r):
return fail("R4", "invisible %s", codePoint(r))
case r >= 0xF000 && r <= 0xF0FF:
return fail("R4", "private use %s", codePoint(r))
case !Assigned(r):
return fail("R4", "unassigned %s", codePoint(r))
}
return nil
}
// checkPlacement applies R4b to s, a segment or a line of a text: VS15 and
// VS16 only right after a character that forms an emoji variation sequence
// with them, and ZWJ and ZWNJ never first, last or right after another ZWJ
// or ZWNJ.
func checkPlacement(rule, s string) error {
prev, i := rune(-1), 0
for _, r := range s {
switch r {
case vs15, vs16:
if prev < 0 || !allowsSelector(prev, r) {
return fail(rule, "%s is not part of an emoji variation sequence", codePoint(r))
}
case zwj, zwnj:
switch {
case prev < 0:
return fail(rule, "%s at the start", codePoint(r))
case prev == zwj || prev == zwnj:
return fail(rule, "%s right after %s", codePoint(r), codePoint(prev))
case i+utf8.RuneLen(r) == len(s):
return fail(rule, "%s at the end", codePoint(r))
}
}
prev = r
i += utf8.RuneLen(r)
}
return nil
}
func checkR5(s string) error {
switch {
case strings.HasPrefix(s, " "):
return fail("R5", "the segment starts with U+0020")
case strings.HasSuffix(s, " "):
return fail("R5", "the segment ends with U+0020")
case strings.HasSuffix(s, "."):
return fail("R5", "the segment ends with '.'")
}
return nil
}
// reserved are the device names of R6, in upper case.
var reserved = map[string]bool{
"CON": true, "PRN": true, "AUX": true, "NUL": true, "CONIN$": true, "CONOUT$": true,
"COM¹": true, "COM²": true, "COM³": true, "LPT¹": true, "LPT²": true, "LPT³": true,
}
func init() {
for d := '0'; d <= '9'; d++ {
reserved["COM"+string(d)] = true
reserved["LPT"+string(d)] = true
}
}
func checkR6(s string) error {
base, _, _ := strings.Cut(s, ".")
base = strings.TrimRight(base, " ")
if name := asciiUpper(base); reserved[name] {
return fail("R6", "%s is a reserved device name", name)
}
return nil
}
// asciiUpper upper-cases ASCII letters only.
func asciiUpper(s string) string {
return strings.Map(func(r rune) rune {
if r >= 'a' && r <= 'z' {
return r - 'a' + 'A'
}
return r
}, s)
}
// shortNameBase is the base of an 8.3 alias in R6b: it ends in '~' and 1 to
// 6 ASCII digits. It has no lookahead, which the regexp package lacks.
var shortNameBase = regexp.MustCompile(`^[^.]*~[0-9]{1,6}$`)
func checkR6b(s string) error {
if strings.Count(s, ".") > 1 {
return nil
}
base, ext, _ := strings.Cut(s, ".")
if n := utf8.RuneCountInString(base); n < 1 || n > 8 {
return nil
}
if utf8.RuneCountInString(ext) > 3 || !shortNameBase.MatchString(base) {
return nil
}
return fail("R6b", "the segment has the form of an 8.3 alias")
}
// checkR6c projects the segment with each best-fit table: every non-ASCII
// code point the table maps to an ASCII byte is replaced by that byte. A
// projection must not hold '/', '\', ':' or U+0000, and must pass R3, R5, R6
// and R6b.
func checkR6c(s string) error {
for i := range bestFitTables {
t := &bestFitTables[i]
p, changed := project(s, t)
if !changed {
continue
}
if j := strings.IndexAny(p, "/\\:\x00"); j >= 0 {
return fail("R6c", "code page %s maps the segment to one with %s", t.name, codePoint(rune(p[j])))
}
for _, check := range []func(string) error{checkR3, checkR5, checkR6, checkR6b} {
if err := check(p); err != nil {
e := err.(*Error)
return fail("R6c", "code page %s maps the segment to one that breaks %s: %s", t.name, e.Rule, e.Detail)
}
}
}
return nil
}
func project(s string, t *bestFitTable) (string, bool) {
var b strings.Builder
changed := false
for _, r := range s {
if r >= 0x80 {
if c, ok := t.lookup(r); ok {
b.WriteByte(c)
changed = true
continue
}
}
b.WriteRune(r)
}
return b.String(), changed
}
// CheckTree applies R7 and then R9 to the paths of a head, which have passed
// CheckPath: no two siblings with the same key, no path that is both a file
// and a folder, comparing keys, and at most MaxImplicitDirs folders. Paths
// are numbered from 1 in the messages.
func CheckTree(paths []string) error {
type node struct {
name string
dir bool
path int // the first path that reached it
}
// children maps the key of a folder, "" for the root, to its children by
// key; the key of a folder joins the keys of its segments with '/'.
children := map[string]map[string]node{}
for n, path := range paths {
segs := strings.Split(path, "/")
parent := ""
for i, s := range segs {
k := Key(s)
isDir := i < len(segs)-1
kids := children[parent]
if kids == nil {
kids = map[string]node{}
children[parent] = kids
}
switch old, ok := kids[k]; {
case !ok:
kids[k] = node{name: s, dir: isDir, path: n + 1}
case old.name != s:
return fail("R7", "path %d collides with path %d in segment %d", n+1, old.path, i+1)
case old.dir != isDir || !isDir:
return fail("R7", "path %d makes a file of path %d a folder, or the reverse, in segment %d", n+1, old.path, i+1)
}
if parent == "" {
parent = k
} else {
parent += "/" + k
}
}
}
dirs := map[string]bool{}
for _, path := range paths {
for i := 0; ; {
j := strings.IndexByte(path[i:], '/')
if j < 0 {
break
}
i += j
dirs[path[:i]] = true
i++
}
}
if len(dirs) > MaxImplicitDirs {
return fail("R9", "%d folders, more than %d", len(dirs), MaxImplicitDirs)
}
return nil
}
// CheckComment checks the comment of a head (spec §29.6).
func CheckComment(s string) error {
return checkText(s, true)
}
// CheckAuthor checks the declared author of a head (spec §29.6).
func CheckAuthor(s string) error {
if err := checkText(s, false); err != nil {
return err
}
if strings.HasPrefix(s, " ") || strings.HasSuffix(s, " ") {
return fail("text", "the declared author starts or ends with U+0020")
}
return nil
}
// checkText applies the rules of spec §29.6, code point by code point, and
// then R4b to each line.
func checkText(s string, comment bool) error {
for _, r := range s {
switch {
case r == '\t' || r == '\n':
if !comment {
return fail("text", "control %s in the declared author", codePoint(r))
}
case r <= 0x1F || (r >= 0x7F && r <= 0x9F):
return fail("text", "control %s", codePoint(r))
case (r >= 0x202A && r <= 0x202E) || (r >= 0x2066 && r <= 0x2069) || r == 0x061C || r == 0x200E || r == 0x200F:
return fail("text", "bidirectional control %s", codePoint(r))
case r == 0x2028 || r == 0x2029:
return fail("text", "separator %s", codePoint(r))
case r == 0xFEFF:
return fail("text", "byte order mark %s", codePoint(r))
case isNoncharacter(r):
return fail("text", "noncharacter %s", codePoint(r))
case DefaultIgnorable(r) && !whitelisted(r):
return fail("text", "invisible %s", codePoint(r))
}
}
for i, line := range strings.Split(s, "\n") {
if err := checkPlacement("text", line); err != nil {
e := err.(*Error)
e.Detail = fmt.Sprintf("line %d: %s", i+1, e.Detail)
return e
}
}
return nil
}
// isNoncharacter reports the 66 noncharacters: U+FDD0 to U+FDEF, and the last
// two code points of each plane.
func isNoncharacter(r rune) bool {
return (r >= 0xFDD0 && r <= 0xFDEF) || r&0xFFFE == 0xFFFE
}

File diff suppressed because it is too large Load Diff

@ -0,0 +1,248 @@
// Package pathrule checks the paths and the texts of the head of a format 3
// capsule (spec §29.5, §29.6) with fixed Unicode 18.0.0 and WindowsBestFit
// tables (spec §29.5.1), never with the Unicode functions of the platform,
// whose version changes with each runtime: Go 1.26.8 carries Unicode 15.0.0,
// and Node 24.9, 16.0.
//
// The tables are generated by internal/pathrule/gen; see tables.go.
package pathrule
//go:generate go run ./gen -data ../../.cache -go tables.go
import (
"crypto/sha256"
"encoding/hex"
"fmt"
"slices"
"strings"
)
// mapping is a sorted table from a code point to a sequence of code points:
// the value of keys[i] is data[start[i]:start[i+1]].
type mapping struct {
keys []rune
start []uint16
data []rune
}
func (m *mapping) lookup(r rune) ([]rune, bool) {
i, ok := slices.BinarySearch(m.keys, r)
if !ok {
return nil, false
}
return m.data[m.start[i]:m.start[i+1]], true
}
// bestFitTable is one code page of WindowsBestFit, reduced to the non-ASCII
// code points it maps to an ASCII byte.
type bestFitTable struct {
name string
from []rune
to []byte
}
func (t *bestFitTable) lookup(r rune) (byte, bool) {
i, ok := slices.BinarySearch(t.from, r)
if !ok {
return 0, false
}
return t.to[i], true
}
// inRanges reports whether r lies in one of the inclusive ranges lo, hi, ...
func inRanges(ranges []rune, r rune) bool {
// The first range whose hi is >= r.
lo, hi := 0, len(ranges)/2
for lo < hi {
m := int(uint(lo+hi) >> 1)
if ranges[2*m+1] < r {
lo = m + 1
} else {
hi = m
}
}
return lo < len(ranges)/2 && ranges[2*lo] <= r
}
// Assigned reports whether r is an assigned code point of Unicode 18.0.0: a
// code point that is not of general category Cn.
func Assigned(r rune) bool { return inRanges(assignedRanges, r) }
// DefaultIgnorable reports whether r has the property
// Default_Ignorable_Code_Point in Unicode 18.0.0.
func DefaultIgnorable(r rune) bool { return inRanges(ignorableRanges, r) }
// ccc is the canonical combining class of r.
func ccc(r rune) uint8 {
i, ok := slices.BinarySearchFunc(cccTable, r, func(e uint32, r rune) int {
switch c := rune(e >> 8); {
case c < r:
return -1
case c > r:
return 1
}
return 0
})
if !ok {
return 0
}
return uint8(cccTable[i])
}
// Hangul syllable decomposition (Unicode §3.12).
const (
hangulS = 0xAC00
hangulL = 0x1100
hangulV = 0x1161
hangulT = 0x11A7
hangulN = 21 * 28
hangulC = 19 * hangulN
)
// NFD returns the canonical decomposition of s, normalization form D of UAX
// #15 with the data of Unicode 18.0.0: every code point replaced by its full
// canonical decomposition, Hangul syllables decomposed algorithmically, and
// each run of combining marks put in canonical order, a stable sort by
// canonical combining class.
func NFD(s string) string {
var out []rune
for _, r := range s {
switch {
case r >= hangulS && r < hangulS+hangulC:
i := r - hangulS
out = append(out, hangulL+i/hangulN, hangulV+(i%hangulN)/28)
if t := i % 28; t != 0 {
out = append(out, hangulT+t)
}
default:
if d, ok := decompositions.lookup(r); ok {
out = append(out, d...)
} else {
out = append(out, r)
}
}
}
// Canonical ordering: insertion sort within each run of non-starters,
// stable, by combining class.
for i := 1; i < len(out); i++ {
c := ccc(out[i])
if c == 0 {
continue
}
for j := i; j > 0; j-- {
p := ccc(out[j-1])
if p <= c {
break
}
out[j-1], out[j] = out[j], out[j-1]
}
}
return string(out)
}
// Fold returns the case folding of s: the mappings of CaseFolding.txt with
// status C or F, and U+0131 (ı) to U+0069 (i), as NTFS does (spec §29.5, R7).
func Fold(s string) string {
var b strings.Builder
for _, r := range s {
if r == 0x0131 {
b.WriteRune(0x0069)
} else if f, ok := foldings.lookup(r); ok {
for _, x := range f {
b.WriteRune(x)
}
} else {
b.WriteRune(r)
}
}
return b.String()
}
// The whitelist of R4: the Default_Ignorable_Code_Point that a path may hold
// (spec §29.5).
const (
zwnj = 0x200C
zwj = 0x200D
vs15 = 0xFE0E
vs16 = 0xFE0F
)
func whitelisted(r rune) bool { return r == zwnj || r == zwj || r == vs15 || r == vs16 }
// stripWhitelist removes the whitelisted code points of R4 from s.
func stripWhitelist(s string) string {
return strings.Map(func(r rune) rune {
if whitelisted(r) {
return -1
}
return r
}, s)
}
// Key is the key of a segment in R7: NFD(fold(NFD(s′))), where s′ is the
// segment without the whitelisted code points of R4. They are removed
// before normalizing because they have combining class 0: between two
// combining marks, they would change their canonical order.
func Key(segment string) string {
return NFD(Fold(NFD(stripWhitelist(segment))))
}
// allowsSelector reports whether base forms an emoji variation sequence with
// the selector sel, U+FE0E or U+FE0F (R4b).
func allowsSelector(base, sel rune) bool {
bases := vs16Bases
if sel == vs15 {
bases = vs15Bases
}
_, ok := slices.BinarySearch(bases, base)
return ok
}
// Canonical is the canonical text of the tables, whose SHA-256 is
// TablesDigest; the generator writes the same text from the data files. Each
// value is in upper-case hexadecimal of at least four digits, except the
// combining class and the code page, in decimal, and the best-fit byte, in
// two hexadecimal digits.
func Canonical() string {
var b strings.Builder
fmt.Fprintf(&b, "unicode %s\n", UnicodeVersion)
for i := 0; i < len(assignedRanges); i += 2 {
fmt.Fprintf(&b, "assigned %04X %04X\n", assignedRanges[i], assignedRanges[i+1])
}
for i := 0; i < len(ignorableRanges); i += 2 {
fmt.Fprintf(&b, "ignorable %04X %04X\n", ignorableRanges[i], ignorableRanges[i+1])
}
for _, e := range cccTable {
fmt.Fprintf(&b, "ccc %04X %d\n", e>>8, e&0xFF)
}
for _, m := range []struct {
name string
t *mapping
}{{"decomp", &decompositions}, {"fold", &foldings}} {
for i, k := range m.t.keys {
fmt.Fprintf(&b, "%s %04X", m.name, k)
for _, r := range m.t.data[m.t.start[i]:m.t.start[i+1]] {
fmt.Fprintf(&b, " %04X", r)
}
b.WriteByte('\n')
}
}
for _, r := range vs15Bases {
fmt.Fprintf(&b, "vs15 %04X\n", r)
}
for _, r := range vs16Bases {
fmt.Fprintf(&b, "vs16 %04X\n", r)
}
for _, t := range bestFitTables {
for i, r := range t.from {
fmt.Fprintf(&b, "bestfit %s %04X %02X\n", t.name, r, t.to[i])
}
}
return b.String()
}
// digest is the SHA-256 of Canonical, in hexadecimal.
func digest() string {
sum := sha256.Sum256([]byte(Canonical()))
return hex.EncodeToString(sum[:])
}

@ -1189,7 +1189,29 @@ R3, R4, R4b, R6c, R7 y la regla de invisibles de §29.6 usan datos fijos, no los
- **Unicode 18.0.0:** `UnicodeData.txt`, con la categoría general, la clase de combinación canónica y las descomposiciones canónicas; `DerivedCoreProperties.txt`, con `Default_Ignorable_Code_Point`; `CaseFolding.txt`, con sus entradas C y F; `emoji/emoji-variation-sequences.txt`, con los caracteres que admiten VS15 y VS16 (R4b); y NFD según UAX #15, con la descomposición algorítmica de Hangul.
- **WindowsBestFit,** de unicode.org: los quince ficheros `bestfit874.txt`, `bestfit932.txt`, `bestfit936.txt`, `bestfit949.txt`, `bestfit950.txt`, de `bestfit1250.txt` a `bestfit1258.txt` y `bestfit1361.txt`, con la conversión de Unicode a su código de página de su sección `WCTABLE`.
Una implementación MUST aplicar las tablas generadas de esos ficheros, fijados por su SHA-256 (por fijar al implementar), y MUST NOT usar las funciones de Unicode de su plataforma para estas reglas: `normalize`, `toLowerCase`, las clases `\p{…}` de las expresiones regulares, el paquete `unicode` de Go o `golang.org/x/text`. Su versión de Unicode cambia con cada motor: Go 1.26.8 trae la 15.0.0, y Node 24.9, la 16.0. La implementación de referencia genera desde esos ficheros el código de las dos implementaciones, y sus pruebas comprueban los digests.
Una implementación MUST aplicar las tablas generadas de esos ficheros, fijados por el SHA-256 de la tabla de abajo, y MUST NOT usar las funciones de Unicode de su plataforma para estas reglas: `normalize`, `toLowerCase`, las clases `\p{…}` de las expresiones regulares, el paquete `unicode` de Go o `golang.org/x/text`. Su versión de Unicode cambia con cada motor: Go 1.26.8 trae la 15.0.0, y Node 24.9, la 16.0. La implementación de referencia genera desde esos ficheros el código de las dos implementaciones, y sus pruebas comprueban los digests. Las rutas de la tabla son las de los ficheros bajo `https://www.unicode.org/Public/18.0.0/ucd/` y `https://www.unicode.org/Public/MAPPINGS/VENDORS/MICSFT/WindowsBestFit/`.
| Fichero | SHA-256 |
|---|---|
| `unicode/18.0.0/UnicodeData.txt` | `0736451de439ae7baf1425136617da495e09ee5afbe6e394374db7009ea08950` |
| `unicode/18.0.0/DerivedCoreProperties.txt` | `09c928886a178fcafd93c29e4bd59073a058e5a100b716d425cb563ab50f68c9` |
| `unicode/18.0.0/CaseFolding.txt` | `a004797658a457bec4dc11683e39f69249ea3b595b752dbea6721c4c9f587b0d` |
| `unicode/18.0.0/emoji/emoji-variation-sequences.txt` | `ff1707564aa1f1b2fcf4ec92d609d4cb26940bc0d4dcf07f5328ad6879e84da3` |
| `bestfit/bestfit874.txt` | `663f43ca662e037c4534cb16298b560f29ce29c27b49b3589601ec3d97dd89fd` |
| `bestfit/bestfit932.txt` | `2614cfea35c3c86c41d33198793a84ca44edee3cf0ee0013a61a43fba4ece331` |
| `bestfit/bestfit936.txt` | `e5070a2d6ad26619f5872ddbe64d3381c11620af5adbb04cda0f0abb1a91fdae` |
| `bestfit/bestfit949.txt` | `50e13b60ea8fda66a8223ecc85270e0f182303222244e2345d3d57f3e839d20a` |
| `bestfit/bestfit950.txt` | `cf8c23389a42a226ea707f7ec32c665556d1fc3364db25bd765ce64d54eaee2a` |
| `bestfit/bestfit1250.txt` | `cef9f171e67b09445bcb3f9ffccdc89418250ff825f1bd2d29a92d2074d7a53b` |
| `bestfit/bestfit1251.txt` | `59ec85612ff908d9da0e877893c935941e56b13a2882b4fb9c9599be3d1ce4e7` |
| `bestfit/bestfit1252.txt` | `72ea23c939c5b26fae7aded0207b327e2f3902d7d3c168d7087f5cfc38ee76a9` |
| `bestfit/bestfit1253.txt` | `ea80c442aff7f09b36da6335f85f8e527f51c146beeb9825ec00d1b6ca99a99e` |
| `bestfit/bestfit1254.txt` | `3d02512087634dc493b720992b590277736ffb2d5b0b665d69b6b9727e2c361a` |
| `bestfit/bestfit1255.txt` | `fdd4bdda74f6571d89171b0070ac052cd3714c395dc3d1799bcd5e4a4da6f83a` |
| `bestfit/bestfit1256.txt` | `745c447ada04a838da8bea406c13f446c7453b6371e8c6c7863a632443d56007` |
| `bestfit/bestfit1257.txt` | `b8c5d7f3b8c25c3d5625d44dd3d6ee7a06e652ddf77373d050282c1cb7517366` |
| `bestfit/bestfit1258.txt` | `5d52a9357b7d6b5b5014ed5a51be0ff9809b0c33625793d2a4feaf502e0682f1` |
| `bestfit/bestfit1361.txt` | `7dcda2d5d2cfc5ddf43757d589a0106e020ef0d9b84de47f08e18297ae0fe1ec` |
Un punto de código posterior a Unicode 18.0.0 es Cn para estas tablas: R4 lo rechaza, y el escritor lo dice con un mensaje que nombra el carácter (§62.1, regla 15).

Loading…
Cancel
Save

Powered by TurnKey Linux.