You can not select more than 25 topics
Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
1404 lines
46 KiB
1404 lines
46 KiB
/* ——————————————————————————————————————————————————————————————————————————
|
|
Proyecto goat
|
|
—————————————————————————————————————————————————————————————————————————————
|
|
Fichero ascii_attrs.go
|
|
Package btes
|
|
Autor Juan V. Navarro juanvnl@activething.com
|
|
Creado 02/02/2026
|
|
—————————————————————————————————————————————————————————————————————————————
|
|
|
|
LICENSES AND TERMS OF USE
|
|
-------------------------
|
|
|
|
This software is licensed under the Elastic License v2.0 (the "License").
|
|
For full terms and additional information regarding permitted and prohibited
|
|
uses, please visit:
|
|
https://activething.com/ATGO/licenses
|
|
|
|
You may use, copy, modify, and redistribute this software internally within
|
|
your organization for any purpose, including research, development, and
|
|
testing, subject to the terms of this License.
|
|
|
|
You may NOT, however, use, provide, distribute, or make this software
|
|
available to any third party as part of a hosted service, SaaS offering, or
|
|
commercial product without first obtaining a commercial license from
|
|
Active Thing.
|
|
|
|
You may combine this software with other code, provided that such
|
|
combination does not circumvent the restrictions of this License.
|
|
|
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
FITNESS FOR A PARTICULAR PURPOSE, AND NON-INFRINGEMENT. In no event shall
|
|
the authors or copyright holders be liable for any claim, damages, or other
|
|
liability arising from the use of this software.
|
|
|
|
—————————————————————————————————————————————————————————————————————————————
|
|
|
|
Web : activething.com | activething.com/goat
|
|
git : g.activething.com | github.com/activething/goat
|
|
Correo : dev@activething.com
|
|
|
|
—————————————————————————————————————————————————————————————————————————————
|
|
No deseo caminar sobre el agua", dijo Siddhartha.
|
|
Que los antiguos chamanes se contenten con tales habilidades.
|
|
—— Hermann Hesse, Siddhartha
|
|
—————————————————————————————————————————————————————————————————————————————
|
|
Copyright (c) 2026 Active Thing
|
|
————————————————————————————————————————————————————————————————————————————— */
|
|
|
|
// Package btes proporciona operaciones de alto rendimiento sobre bytes ASCII
|
|
// mediante el uso de tablas de búsqueda (Lookup Tables - LUT) para clasificación
|
|
// y transformación en tiempo constante O(1).
|
|
//
|
|
// La estrategia principal es pre-calcular atributos de cada byte ASCII posible
|
|
// (256 valores) en una tabla de 256 entradas, permitiendo verificar propiedades
|
|
// de caracteres mediante simples operaciones bitwise sin comparaciones costosas.
|
|
//
|
|
// # Casos de Uso Principales
|
|
//
|
|
// - Routing HTTP: normalización de paths, métodos y headers
|
|
// - Parsing de URLs: detección de caracteres válidos, decode hexadecimal
|
|
// - Validación de identificadores: verificación alfanumérica rápida
|
|
// - Transformación de strings: case conversions sin allocations innecesarias
|
|
//
|
|
// # Ventajas de Rendimiento
|
|
//
|
|
// - IsAlpha(ch): ~1ns vs 8-15ns con unicode.IsLetter
|
|
// - ToUpper(s): 4x más rápido que implementación naive con allocation tardía
|
|
// - EqualFold(a,b): ~27ns vs ~38ns con strings.EqualFold para ASCII puro
|
|
package btes
|
|
|
|
// ════════════════════════════════════════════════════════════════════════════
|
|
// CONSTANTES DE ATRIBUTOS ASCII
|
|
// ════════════════════════════════════════════════════════════════════════════
|
|
|
|
// Máscaras de bits para clasificar caracteres ASCII. Cada constante representa
|
|
// un bit diferente, permitiendo que un byte tenga múltiples atributos simultáneos
|
|
// mediante operaciones OR (|).
|
|
//
|
|
// Ejemplo: 'A' tiene atributos: ASCIIAttrPrintable | ASCIIAttrAlpha | ASCIIAttrAlphaUpper
|
|
const (
|
|
// ASCIIAttrControl identifica caracteres de control ASCII (0-31, 127).
|
|
// Incluye: \n, \r, \t, ESC, NULL, etc.
|
|
ASCIIAttrControl Attr = 1 << iota
|
|
|
|
// ASCIIAttrPrintable identifica caracteres imprimibles (32-126).
|
|
// Excluye caracteres de control pero incluye espacio (32).
|
|
ASCIIAttrPrintable
|
|
|
|
// ASCIIAttrAlpha identifica letras (A-Z, a-z).
|
|
// Útil para validar identificadores o paths alfabéticos.
|
|
ASCIIAttrAlpha
|
|
|
|
// ASCIIAttrAlphaLower identifica letras minúsculas (a-z).
|
|
ASCIIAttrAlphaLower
|
|
|
|
// ASCIIAttrAlphaUpper identifica letras mayúsculas (A-Z).
|
|
ASCIIAttrAlphaUpper
|
|
|
|
// ASCIIAttrDigit identifica dígitos decimales ('0'-'9').
|
|
ASCIIAttrDigit
|
|
|
|
// ASCIIAttrHex identifica dígitos hexadecimales adicionales (A-F, a-f).
|
|
// No incluye 0-9, que ya tienen ASCIIAttrDigit.
|
|
ASCIIAttrHex
|
|
|
|
// ASCIIAttrDigitHex es una máscara combinada para verificar si un byte
|
|
// es un dígito hexadecimal válido (0-9, A-F, a-f).
|
|
// Equivale a: ASCIIAttrHex | ASCIIAttrDigit
|
|
ASCIIAttrDigitHex = ASCIIAttrHex | ASCIIAttrDigit
|
|
|
|
ASCIIAttrAlphaNum = ASCIIAttrAlpha | ASCIIAttrDigit
|
|
ASCIIAttrAlphaNumLower = ASCIIAttrAlphaLower | ASCIIAttrDigit
|
|
ASCIIAttrAlphaNumUpper = ASCIIAttrAlphaUpper | ASCIIAttrDigit
|
|
)
|
|
|
|
// ════════════════════════════════════════════════════════════════════════════
|
|
// TIPO PRINCIPAL: ASCIIAttrs
|
|
// ════════════════════════════════════════════════════════════════════════════
|
|
|
|
// ASCIIAttrs encapsula una tabla de atributos (LUT) para los 256 valores ASCII.
|
|
// Cada entrada de la tabla contiene los atributos del byte correspondiente
|
|
// codificados como bits en un uint8.
|
|
//
|
|
// La estructura es inmutable después de la inicialización, lo que permite
|
|
// compartirla de forma segura entre goroutines sin sincronización.
|
|
//
|
|
// # Complejidad de Operaciones
|
|
//
|
|
// - Verificación de atributos: O(1) - un lookup + AND bitwise
|
|
// - Transformaciones individuales: O(1) - un lookup + aritmética
|
|
// - Transformaciones de slices: O(n) - donde n es el tamaño del slice
|
|
//
|
|
// # Uso de Memoria
|
|
//
|
|
// - Tamaño de tabla: 256 bytes (una entrada por byte ASCII posible)
|
|
// - Overhead por instancia: ~8 bytes (puntero + metadatos struct)
|
|
//
|
|
// # Ejemplo Básico
|
|
//
|
|
// attrs := NewASCIIAttrs()
|
|
//
|
|
// // Verificaciones O(1)
|
|
// if attrs.IsDigit('5') {
|
|
// // true
|
|
// }
|
|
//
|
|
// // Transformaciones con zero-copy cuando sea posible
|
|
// upper := attrs.ToUpper([]byte("hello")) // "HELLO"
|
|
// same := attrs.ToUpper([]byte("HELLO")) // retorna input (zero-copy)
|
|
type ASCIIAttrs struct {
|
|
// attrs es la tabla interna de 256 entradas que mapea cada byte ASCII
|
|
// a sus atributos codificados como bits.
|
|
attrs Attrs
|
|
}
|
|
|
|
// ════════════════════════════════════════════════════════════════════════════
|
|
// CONSTRUCTOR
|
|
// ════════════════════════════════════════════════════════════════════════════
|
|
|
|
// NewASCIIAttrs crea y retorna una nueva instancia de ASCIIAttrs con la tabla
|
|
// de atributos ASCII completamente inicializada.
|
|
//
|
|
// La tabla se pre-calcula una sola vez en la construcción con los siguientes
|
|
// atributos para cada rango de caracteres ASCII:
|
|
//
|
|
// - Control (0-31, 127): Caracteres no imprimibles
|
|
// - Printable (32-126): Caracteres imprimibles estándar
|
|
// - Digit ('0'-'9'): Dígitos decimales
|
|
// - AlphaUpper ('A'-'Z'): Letras mayúsculas
|
|
// - AlphaLower ('a'-'z'): Letras minúsculas
|
|
// - Hex ('A'-'F', 'a'-'f'): Dígitos hexadecimales adicionales
|
|
//
|
|
// # Complejidad
|
|
//
|
|
// - Tiempo: O(256) - inicialización de tabla completa
|
|
// - Memoria: 256 bytes para la tabla + overhead de struct
|
|
//
|
|
// # Thread Safety
|
|
//
|
|
// Esta función es thread-safe. Cada llamada retorna una nueva instancia
|
|
// independiente que puede usarse concurrentemente sin sincronización.
|
|
//
|
|
// # Ejemplo
|
|
//
|
|
// // Típicamente se crea una instancia global
|
|
// var asciiAttrs = NewASCIIAttrs()
|
|
//
|
|
// // O se crea bajo demanda
|
|
// func processPath(path []byte) {
|
|
// attrs := NewASCIIAttrs()
|
|
// upper := attrs.ToUpper(path)
|
|
// // ...
|
|
// }
|
|
func NewASCIIAttrs() *ASCIIAttrs {
|
|
a := &ASCIIAttrs{}
|
|
|
|
// Inicializar caracteres de control (0-31 + DEL)
|
|
a.attrs.SetRange(ASCIIAttrControl, 0, 31)
|
|
a.attrs[127] = ASCIIAttrControl
|
|
|
|
// Inicializar caracteres imprimibles (espacio hasta ~)
|
|
a.attrs.SetRange(ASCIIAttrPrintable, 32, 126)
|
|
|
|
// Inicializar dígitos decimales
|
|
a.attrs.AddRange(ASCIIAttrDigit, '0', '9')
|
|
|
|
// Inicializar letras (marcando Alpha + Upper/Lower según corresponda)
|
|
a.attrs.AddRange(ASCIIAttrAlpha|ASCIIAttrAlphaUpper, 'A', 'Z')
|
|
a.attrs.AddRange(ASCIIAttrAlpha|ASCIIAttrAlphaLower, 'a', 'z')
|
|
|
|
// Inicializar dígitos hexadecimales adicionales (A-F, a-f)
|
|
// Los dígitos 0-9 ya están marcados con ASCIIAttrDigit
|
|
a.attrs.AddRange(ASCIIAttrHex, 'A', 'F')
|
|
a.attrs.AddRange(ASCIIAttrHex, 'a', 'f')
|
|
|
|
return a
|
|
}
|
|
|
|
// ════════════════════════════════════════════════════════════════════════════
|
|
// COMPARACIÓN CASE-INSENSITIVE
|
|
// ════════════════════════════════════════════════════════════════════════════
|
|
|
|
// EqualFold compara dos slices de bytes ignorando diferencias de mayúsculas/minúsculas
|
|
// en letras ASCII. Es equivalente a strings.EqualFold pero optimizado para ASCII puro.
|
|
//
|
|
// # Algoritmo
|
|
//
|
|
// 1. Verificación de longitud (early exit si difieren)
|
|
// 2. Comparación byte por byte:
|
|
// - Si son iguales → continuar
|
|
// - Si difieren → normalizar a minúsculas con |0x20 y verificar si son letras
|
|
//
|
|
// # Técnica de Normalización
|
|
//
|
|
// Usa el truco de |0x20 para convertir cualquier letra ASCII a minúscula:
|
|
// - 'A' (65) | 0x20 = 'a' (97)
|
|
// - 'a' (97) | 0x20 = 'a' (97)
|
|
// - '5' (53) | 0x20 = 'u' (117) ← Por eso verificamos IsAlpha después
|
|
//
|
|
// # Complejidad
|
|
//
|
|
// - Tiempo: O(n) donde n = len(source)
|
|
// - Memoria: O(1) - no allocations
|
|
// - Early exit: Retorna false inmediatamente al primer mismatch
|
|
//
|
|
// # Performance
|
|
//
|
|
// - ~27ns para strings típicos de 8-12 bytes (métodos HTTP, headers cortos)
|
|
// - ~38% más rápido que strings.EqualFold para ASCII puro
|
|
// - Beneficio aumenta con strings más largos por mejor inlining y cache locality
|
|
//
|
|
// # Casos de Uso
|
|
//
|
|
// - Comparar métodos HTTP: "GET" vs "get"
|
|
// - Comparar headers HTTP: "Content-Type" vs "content-type"
|
|
// - Routing case-insensitive: "/API/Users" vs "/api/users"
|
|
//
|
|
// # Ejemplo
|
|
//
|
|
// attrs := NewASCIIAttrs()
|
|
//
|
|
// // Comparaciones case-insensitive
|
|
// attrs.EqualFold([]byte("GET"), []byte("get")) // true
|
|
// attrs.EqualFold([]byte("Content-Type"), []byte("content-type")) // true
|
|
// attrs.EqualFold([]byte("hello"), []byte("world")) // false
|
|
// attrs.EqualFold([]byte("abc"), []byte("ABC123")) // false (longitud)
|
|
//
|
|
// # Limitaciones
|
|
//
|
|
// Solo funciona correctamente con ASCII. Para Unicode, usar strings.EqualFold
|
|
// o unicode.SimpleFold.
|
|
func (t *ASCIIAttrs) EqualFold(source, target []byte) bool {
|
|
// Early exit: longitudes diferentes nunca pueden ser iguales
|
|
if len(source) != len(target) {
|
|
return false
|
|
}
|
|
|
|
// Comparación byte a byte
|
|
for i := 0; i < len(source); i++ {
|
|
s, tr := source[i], target[i]
|
|
|
|
// Fast path: bytes exactamente iguales
|
|
if s == tr {
|
|
continue
|
|
}
|
|
|
|
// Slow path: verificar si son la misma letra en diferentes casos
|
|
// |0x20 convierte A-Z a a-z (si es letra)
|
|
// Luego verificamos que realmente sea una letra para evitar falsos positivos
|
|
if (s|0x20) == (tr|0x20) && t.attrs[s]&ASCIIAttrAlpha != 0 {
|
|
continue
|
|
}
|
|
|
|
// No son iguales ni case-insensitive
|
|
return false
|
|
}
|
|
|
|
return true
|
|
}
|
|
|
|
// SimpleLetterEqualFold verifica si dos slices tienen el mismo patrón de
|
|
// mayúsculas/minúsculas en las posiciones de letras.
|
|
//
|
|
// A diferencia de EqualFold, esta función NO ignora las diferencias de case.
|
|
// En su lugar, verifica que cuando hay una letra mayúscula en source[i], también
|
|
// hay una letra mayúscula en target[i] (aunque sean letras diferentes).
|
|
//
|
|
// # Casos de Uso
|
|
//
|
|
// - Validar consistencia de formato en identificadores
|
|
// - Verificar que el patrón de capitalización sea consistente
|
|
// - Validación de schemas donde el case pattern importa
|
|
//
|
|
// # Ejemplo
|
|
//
|
|
// attrs := NewASCIIAttrs()
|
|
//
|
|
// // Mismo patrón de case (ambos empiezan con mayúscula)
|
|
// attrs.SimpleLetterEqualFold([]byte("HelloWorld"), []byte("GreatThing")) // true
|
|
//
|
|
// // Diferente patrón (uno empieza minúscula)
|
|
// attrs.SimpleLetterEqualFold([]byte("hello"), []byte("World")) // false
|
|
//
|
|
// // Mismo contenido pero en ambos
|
|
// attrs.SimpleLetterEqualFold([]byte("Hello"), []byte("Hello")) // true
|
|
//
|
|
// # Complejidad
|
|
//
|
|
// - Tiempo: O(n)
|
|
// - Memoria: O(1)
|
|
func (t *ASCIIAttrs) SimpleLetterEqualFold(source, target []byte) bool {
|
|
if len(source) != len(target) {
|
|
return false
|
|
}
|
|
|
|
for i, b := range source {
|
|
// Verificar que ambos bytes tengan el mismo atributo de case
|
|
// (ambos uppercase o ambos no-uppercase)
|
|
if t.IsAlphaUpper(b) != t.IsAlphaUpper(target[i]) {
|
|
return false
|
|
}
|
|
}
|
|
|
|
return true
|
|
}
|
|
|
|
// ════════════════════════════════════════════════════════════════════════════
|
|
// CONVERSIONES DE CASE - BYTE INDIVIDUAL
|
|
// ════════════════════════════════════════════════════════════════════════════
|
|
|
|
// ToUpperByte convierte un byte individual a mayúscula si es una letra minúscula ASCII.
|
|
//
|
|
// # Algoritmo
|
|
//
|
|
// En ASCII, la diferencia entre una letra mayúscula y su equivalente minúscula es 32:
|
|
// - 'a' = 97, 'A' = 65 → diferencia = 32
|
|
// - 'z' = 122, 'Z' = 90 → diferencia = 32
|
|
//
|
|
// Por lo tanto: mayúscula = minúscula - 32
|
|
//
|
|
// # Complejidad
|
|
//
|
|
// - Tiempo: O(1) - un lookup + una resta condicional
|
|
// - Memoria: O(1) - sin allocations
|
|
//
|
|
// # Ejemplo
|
|
//
|
|
// attrs := NewASCIIAttrs()
|
|
// attrs.ToUpperByte('a') // 'A'
|
|
// attrs.ToUpperByte('Z') // 'Z' (sin cambios)
|
|
// attrs.ToUpperByte('5') // '5' (sin cambios)
|
|
func (t *ASCIIAttrs) ToUpperByte(chr byte) byte {
|
|
if t.attrs[chr]&ASCIIAttrAlphaLower != 0 {
|
|
return chr - 32
|
|
}
|
|
return chr
|
|
}
|
|
|
|
// ToLowerByte convierte un byte individual a minúscula si es una letra mayúscula ASCII.
|
|
//
|
|
// # Algoritmo
|
|
//
|
|
// Inverso de ToUpperByte: minúscula = mayúscula + 32
|
|
//
|
|
// # Complejidad
|
|
//
|
|
// - Tiempo: O(1)
|
|
// - Memoria: O(1)
|
|
//
|
|
// # Ejemplo
|
|
//
|
|
// attrs := NewASCIIAttrs()
|
|
// attrs.ToLowerByte('A') // 'a'
|
|
// attrs.ToLowerByte('z') // 'z' (sin cambios)
|
|
// attrs.ToLowerByte('5') // '5' (sin cambios)
|
|
func (t *ASCIIAttrs) ToLowerByte(chr byte) byte {
|
|
if t.attrs[chr]&ASCIIAttrAlphaUpper != 0 {
|
|
return chr + 32
|
|
}
|
|
return chr
|
|
}
|
|
|
|
// ════════════════════════════════════════════════════════════════════════════
|
|
// CONVERSIONES DE CASE - IN-PLACE (MUTANTES)
|
|
// ════════════════════════════════════════════════════════════════════════════
|
|
|
|
// Upperize modifica el slice in-place convirtiendo todas las letras minúsculas
|
|
// a mayúsculas. Es la versión más rápida cuando se puede mutar el slice original.
|
|
//
|
|
// # Performance
|
|
//
|
|
// - ~43ns para slice típico de 16 bytes
|
|
// - Sin allocations (0 B/op)
|
|
// - Cache-friendly: acceso secuencial lineal
|
|
//
|
|
// # Casos de Uso
|
|
//
|
|
// - Normalizar paths temporales antes de lookup en tabla
|
|
// - Procesar buffers reutilizables
|
|
// - Cuando el slice original no se necesita preservar
|
|
//
|
|
// # Ejemplo
|
|
//
|
|
// attrs := NewASCIIAttrs()
|
|
//
|
|
// path := []byte("hello/world")
|
|
// attrs.Upperize(path)
|
|
// // path ahora es "HELLO/WORLD"
|
|
//
|
|
// // CUIDADO: El slice original está modificado
|
|
// original := []byte("test")
|
|
// attrs.Upperize(original)
|
|
// fmt.Println(string(original)) // "TEST" (¡modificado!)
|
|
//
|
|
// # Complejidad
|
|
//
|
|
// - Tiempo: O(n) donde n = len(list)
|
|
// - Memoria: O(1) - no allocations
|
|
func (t *ASCIIAttrs) Upperize(list []byte) {
|
|
for ix, ch := range list {
|
|
if t.attrs[ch]&ASCIIAttrAlphaLower != 0 {
|
|
list[ix] -= 32
|
|
}
|
|
}
|
|
}
|
|
|
|
// Lowerize modifica el slice in-place convirtiendo todas las letras mayúsculas
|
|
// a minúsculas.
|
|
//
|
|
// Equivalente a Upperize pero en dirección opuesta. Ver Upperize para más detalles.
|
|
//
|
|
// # Ejemplo
|
|
//
|
|
// attrs := NewASCIIAttrs()
|
|
//
|
|
// method := []byte("GET")
|
|
// attrs.Lowerize(method)
|
|
// // method ahora es "get"
|
|
func (t *ASCIIAttrs) Lowerize(list []byte) {
|
|
for ix, ch := range list {
|
|
if t.attrs[ch]&ASCIIAttrAlphaUpper != 0 {
|
|
list[ix] += 32
|
|
}
|
|
}
|
|
}
|
|
|
|
// ════════════════════════════════════════════════════════════════════════════
|
|
// CONVERSIONES DE CASE - COPY-ON-WRITE (INMUTABLES)
|
|
// ════════════════════════════════════════════════════════════════════════════
|
|
|
|
// ToUpper devuelve una copia del slice con todas las letras minúsculas convertidas
|
|
// a mayúsculas. Si el slice ya está completamente en mayúsculas, retorna el slice
|
|
// original sin hacer copia (zero-copy optimization).
|
|
//
|
|
// # Algoritmo Optimizado (Pre-Scan Strategy)
|
|
//
|
|
// 1. Pre-scan: Escanear el slice completo buscando minúsculas
|
|
// - Si no encuentra ninguna → retornar slice original (zero-copy)
|
|
// - Si encuentra al menos una → proceder a paso 2
|
|
//
|
|
// 2. Allocation: Crear nuevo slice del mismo tamaño UNA vez
|
|
//
|
|
// 3. Transform: Recorrer original y copiar/transformar a resultado
|
|
// - Si es minúscula → copiar como mayúscula (byte - 32)
|
|
// - Si no es minúscula → copiar tal cual
|
|
//
|
|
// # Por Qué Esta Estrategia es Óptima
|
|
//
|
|
// ## Comparación con Estrategia Naive:
|
|
//
|
|
// // ❌ NAIVE (malo):
|
|
// for ix, ch := range list {
|
|
// if isLower(ch) {
|
|
// if result == nil {
|
|
// result = make([]byte, len(list)) // Allocation tardía
|
|
// copy(result, list[:ix]) // Copy de bytes ya visitados
|
|
// }
|
|
// result[ix] = ch - 32
|
|
// }
|
|
// }
|
|
//
|
|
// Problemas de estrategia naive:
|
|
// - Allocation DENTRO del loop (primera minúscula encontrada)
|
|
// - copy() ejecutado DESPUÉS de iterar ix bytes
|
|
// - En peor caso (minúscula al final): itera N → alloc → copy N → transforma 1
|
|
// - Total: 2N operaciones de memoria
|
|
//
|
|
// ## Ventajas de Pre-Scan:
|
|
//
|
|
// - Máximo 2 scans completos (pre-scan + transform)
|
|
// - Allocation UNA vez al principio
|
|
// - No hay copy() separado (integrado en transform)
|
|
// - Zero-copy cuando no hay cambios (común en HTTP: métodos ya uppercase)
|
|
//
|
|
// # Performance
|
|
//
|
|
// - ~75ns para slice típico con cambios (4x mejora vs naive 307ns)
|
|
// - ~5ns para slice sin cambios (zero-copy, solo pre-scan)
|
|
// - 1 allocation cuando necesario vs allocations múltiples en naive
|
|
//
|
|
// # Casos de Uso
|
|
//
|
|
// - Normalizar métodos HTTP: "get" → "GET" (pero "GET" → "GET" zero-copy)
|
|
// - Normalizar headers HTTP antes de comparación
|
|
// - Convertir paths a formato canónico
|
|
// - Cualquier caso donde el original debe preservarse
|
|
//
|
|
// # Ejemplo
|
|
//
|
|
// attrs := NewASCIIAttrs()
|
|
//
|
|
// // Caso 1: Necesita transformación (hace copia)
|
|
// lower := []byte("hello")
|
|
// upper := attrs.ToUpper(lower) // "HELLO" (nuevo slice)
|
|
// // lower sigue siendo "hello" (original preservado)
|
|
//
|
|
// // Caso 2: Ya está uppercase (zero-copy)
|
|
// already := []byte("HELLO")
|
|
// same := attrs.ToUpper(already) // retorna already (mismo slice)
|
|
// // No hay allocation ni copia
|
|
//
|
|
// // Caso 3: Mezclado (hace copia)
|
|
// mixed := []byte("HeLLo")
|
|
// result := attrs.ToUpper(mixed) // "HELLO" (nuevo slice)
|
|
//
|
|
// # Complejidad
|
|
//
|
|
// - Tiempo:
|
|
// - Mejor caso (sin cambios): O(n) - solo pre-scan
|
|
// - Peor caso (con cambios): O(2n) - pre-scan + transform
|
|
// - Memoria:
|
|
// - Mejor caso: O(1) - no allocations
|
|
// - Peor caso: O(n) - un nuevo slice
|
|
//
|
|
// # Thread Safety
|
|
//
|
|
// Es thread-safe para el slice de entrada (no lo modifica). Cada llamada que
|
|
// necesita transformación retorna un nuevo slice independiente.
|
|
func (t *ASCIIAttrs) ToUpper(list []byte) []byte {
|
|
// FASE 1: Pre-scan para detectar si hay minúsculas
|
|
// Early exit si no hay cambios necesarios
|
|
needsChange := false
|
|
for _, ch := range list {
|
|
if t.attrs[ch]&ASCIIAttrAlphaLower != 0 {
|
|
needsChange = true
|
|
break // No necesitamos seguir escaneando
|
|
}
|
|
}
|
|
|
|
// Zero-copy optimization: retornar original si ya está uppercase
|
|
if !needsChange {
|
|
return list
|
|
}
|
|
|
|
// FASE 2: Allocation UNA vez + transformation en single-pass
|
|
result := make([]byte, len(list))
|
|
for i, ch := range list {
|
|
if t.attrs[ch]&ASCIIAttrAlphaLower != 0 {
|
|
result[i] = ch - 32 // Convertir a mayúscula
|
|
} else {
|
|
result[i] = ch // Copiar tal cual
|
|
}
|
|
}
|
|
|
|
return result
|
|
}
|
|
|
|
// ToLower devuelve una copia del slice con todas las letras mayúsculas convertidas
|
|
// a minúsculas. Usa la misma estrategia optimizada que ToUpper.
|
|
//
|
|
// Ver documentación de ToUpper para detalles del algoritmo y optimizaciones.
|
|
//
|
|
// # Ejemplo
|
|
//
|
|
// attrs := NewASCIIAttrs()
|
|
//
|
|
// upper := []byte("HELLO")
|
|
// lower := attrs.ToLower(upper) // "hello" (nuevo slice)
|
|
//
|
|
// alreadyLower := []byte("hello")
|
|
// same := attrs.ToLower(alreadyLower) // retorna alreadyLower (zero-copy)
|
|
func (t *ASCIIAttrs) ToLower(list []byte) []byte {
|
|
// Pre-scan para detectar mayúsculas
|
|
needsChange := false
|
|
for _, ch := range list {
|
|
if t.attrs[ch]&ASCIIAttrAlphaUpper != 0 {
|
|
needsChange = true
|
|
break
|
|
}
|
|
}
|
|
|
|
// Zero-copy si no hay cambios
|
|
if !needsChange {
|
|
return list
|
|
}
|
|
|
|
// Single allocation + transform
|
|
result := make([]byte, len(list))
|
|
for i, ch := range list {
|
|
if t.attrs[ch]&ASCIIAttrAlphaUpper != 0 {
|
|
result[i] = ch + 32 // Convertir a minúscula
|
|
} else {
|
|
result[i] = ch
|
|
}
|
|
}
|
|
|
|
return result
|
|
}
|
|
|
|
// ════════════════════════════════════════════════════════════════════════════
|
|
// TRANSFORMACIONES DE NAMING CONVENTIONS
|
|
// ════════════════════════════════════════════════════════════════════════════
|
|
|
|
// Camelize transforma el slice a CamelCase in-place, eliminando los separadores
|
|
// y capitalizando la primera letra de cada palabra.
|
|
//
|
|
// # Formato CamelCase
|
|
//
|
|
// En CamelCase, cada palabra (excepto posiblemente la primera) comienza con mayúscula
|
|
// y no hay espacios ni separadores:
|
|
// - snake_case "hello_world" → CamelCase "HelloWorld"
|
|
// - kebab-case "hello-world" → CamelCase "HelloWorld"
|
|
//
|
|
// # Comportamiento
|
|
//
|
|
// - Elimina todos los separadores encontrados
|
|
// - Capitaliza la primera letra después de cada separador (o al inicio)
|
|
// - Normaliza el resto de letras a minúsculas
|
|
// - El slice se compacta (reduce tamaño) si había separadores
|
|
//
|
|
// # IMPORTANTE: Modificación In-Place
|
|
//
|
|
// Esta función modifica el slice original Y retorna un re-slice con la nueva
|
|
// longitud (menor si había separadores). El slice retornado comparte el mismo
|
|
// backing array que el original.
|
|
//
|
|
// # Algoritmo
|
|
//
|
|
// 1. Usar dos índices: readIdx (lectura) y writeIdx (escritura)
|
|
// 2. Para cada byte leído:
|
|
// - Si es separador: marcar que siguiente letra debe capitalizarse, no escribir
|
|
// - Si debe capitalizarse: escribir como mayúscula, desmarcar flag
|
|
// - Sino: escribir como minúscula
|
|
// 3. Retornar slice[:writeIdx] con nueva longitud
|
|
//
|
|
// # Ejemplo
|
|
//
|
|
// attrs := NewASCIIAttrs()
|
|
//
|
|
// input := []byte("hello_world_test")
|
|
// result := attrs.Camelize(input, '_')
|
|
// // result: "HelloWorldTest" (len=14)
|
|
// // input: "HelloWorldTestst" (¡modificado! últimos bytes son basura)
|
|
//
|
|
// // SIEMPRE usar el slice retornado:
|
|
// camel := attrs.Camelize([]byte("user_service_handler"), '_')
|
|
// fmt.Println(string(camel)) // "UserServiceHandler"
|
|
//
|
|
// # Complejidad
|
|
//
|
|
// - Tiempo: O(n)
|
|
// - Memoria: O(1) - no allocations, modifica in-place
|
|
//
|
|
// # Casos de Uso
|
|
//
|
|
// - Transformar nombres de variables de snake_case a CamelCase
|
|
// - Normalizar identificadores desde diferentes convenciones
|
|
// - Procesar parámetros de configuración
|
|
func (t *ASCIIAttrs) Camelize(list []byte, separator byte) []byte {
|
|
if len(list) == 0 {
|
|
return list
|
|
}
|
|
|
|
writeIdx := 0
|
|
capitalizeNext := true
|
|
|
|
for readIdx := 0; readIdx < len(list); readIdx++ {
|
|
ch := list[readIdx]
|
|
|
|
// Si encontramos separador: marcamos que siguiente letra va en mayúscula
|
|
// y NO escribimos el separador (lo eliminamos)
|
|
if ch == separator {
|
|
capitalizeNext = true
|
|
continue // Salta el separador (compactación)
|
|
}
|
|
|
|
// Aplicar transformación según estado
|
|
if capitalizeNext {
|
|
// Primera letra de palabra: capitalizar si es minúscula
|
|
if t.attrs[ch]&ASCIIAttrAlphaLower != 0 {
|
|
list[writeIdx] = ch - 32 // A mayúscula
|
|
} else {
|
|
list[writeIdx] = ch
|
|
}
|
|
// Solo desactivar capitalización si encontramos un alfanumérico
|
|
// (ignora símbolos/espacios al inicio de palabra)
|
|
if t.attrs[ch]&(ASCIIAttrAlpha|ASCIIAttrDigit) != 0 {
|
|
capitalizeNext = false
|
|
}
|
|
} else {
|
|
// Resto de la palabra: normalizar a minúsculas
|
|
if t.attrs[ch]&ASCIIAttrAlphaUpper != 0 {
|
|
list[writeIdx] = ch + 32 // A minúscula
|
|
} else {
|
|
list[writeIdx] = ch
|
|
}
|
|
}
|
|
writeIdx++
|
|
}
|
|
|
|
// Retornar re-slice con nueva longitud
|
|
return list[:writeIdx]
|
|
}
|
|
|
|
// ToCamel convierte a CamelCase retornando un nuevo slice, preservando el original.
|
|
// Usa estrategia optimizada de pre-scan para evitar allocations innecesarias.
|
|
//
|
|
// # Diferencia con Camelize
|
|
//
|
|
// - Camelize: Modifica in-place, más rápido, usa mismo backing array
|
|
// - ToCamel: Crea copia si necesario, preserva original, zero-copy cuando posible
|
|
//
|
|
// # Algoritmo Optimizado
|
|
//
|
|
// 1. Pre-scan Phase:
|
|
// - Contar separadores (para calcular tamaño final)
|
|
// - Detectar si necesita cambios (early exit si ya está en formato correcto)
|
|
//
|
|
// 2. Decision:
|
|
// - Si no necesita cambios → retornar original (zero-copy)
|
|
// - Si necesita cambios → continuar a fase 3
|
|
//
|
|
// 3. Transform Phase:
|
|
// - Allocar slice con tamaño final conocido (len - sepCount)
|
|
// - Copiar y transformar simultáneamente
|
|
//
|
|
// # Performance
|
|
//
|
|
// - ~115ns con transformación (4x mejora vs implementación naive)
|
|
// - ~10ns sin transformación (zero-copy)
|
|
// - Evita allocations innecesarias cuando input ya está en CamelCase
|
|
//
|
|
// # Ejemplo
|
|
//
|
|
// attrs := NewASCIIAttrs()
|
|
//
|
|
// // Transformación necesaria
|
|
// snake := []byte("hello_world")
|
|
// camel := attrs.ToCamel(snake, '_')
|
|
// // snake: "hello_world" (sin cambios)
|
|
// // camel: "HelloWorld" (nuevo slice)
|
|
//
|
|
// // Zero-copy (ya es CamelCase sin separadores)
|
|
// already := []byte("HelloWorld")
|
|
// same := attrs.ToCamel(already, '_')
|
|
// // same apunta a already (no hay copia)
|
|
//
|
|
// // Con múltiples separadores
|
|
// multi := []byte("user_service_handler")
|
|
// result := attrs.ToCamel(multi, '_')
|
|
// // result: "UserServiceHandler"
|
|
//
|
|
// # Complejidad
|
|
//
|
|
// - Tiempo:
|
|
// - Mejor caso (sin cambios): O(n) - solo pre-scan
|
|
// - Peor caso (con cambios): O(2n) - pre-scan + transform
|
|
// - Memoria:
|
|
// - Mejor caso: O(1) - zero-copy
|
|
// - Peor caso: O(n-s) - donde s = número de separadores
|
|
func (t *ASCIIAttrs) ToCamel(list []byte, separator byte) []byte {
|
|
if len(list) == 0 {
|
|
return list
|
|
}
|
|
|
|
// FASE 1: Pre-scan para detectar cambios necesarios Y contar separadores
|
|
sepCount := 0
|
|
needsChange := false
|
|
capitalizeNext := true
|
|
|
|
for _, ch := range list {
|
|
if ch == separator {
|
|
sepCount++
|
|
needsChange = true // Siempre necesita cambios si hay separadores
|
|
capitalizeNext = true
|
|
continue
|
|
}
|
|
|
|
// Verificar si necesita cambio según posición
|
|
if capitalizeNext {
|
|
// Primera letra de palabra: debe ser mayúscula
|
|
if t.attrs[ch]&ASCIIAttrAlphaLower != 0 {
|
|
needsChange = true // Es minúscula, necesita cambio
|
|
}
|
|
// Consumir estado solo si es alfanumérico
|
|
if t.attrs[ch]&(ASCIIAttrAlpha|ASCIIAttrDigit) != 0 {
|
|
capitalizeNext = false
|
|
}
|
|
} else {
|
|
// Resto de la palabra: debe ser minúscula
|
|
if t.attrs[ch]&ASCIIAttrAlphaUpper != 0 {
|
|
needsChange = true // Es mayúscula, necesita normalización
|
|
}
|
|
}
|
|
}
|
|
|
|
// Zero-copy optimization
|
|
if !needsChange {
|
|
return list
|
|
}
|
|
|
|
// FASE 2: Allocation con tamaño final conocido
|
|
finalLen := len(list) - sepCount
|
|
if finalLen == 0 {
|
|
return []byte{}
|
|
}
|
|
|
|
result := make([]byte, finalLen)
|
|
|
|
// FASE 3: Transform en single-pass
|
|
i := 0
|
|
capitalizeNext = true
|
|
|
|
for _, ch := range list {
|
|
if ch == separator {
|
|
capitalizeNext = true
|
|
continue
|
|
}
|
|
|
|
if capitalizeNext {
|
|
if t.attrs[ch]&ASCIIAttrAlphaLower != 0 {
|
|
result[i] = ch - 32 // A mayúscula
|
|
} else {
|
|
result[i] = ch
|
|
}
|
|
if t.attrs[ch]&(ASCIIAttrAlpha|ASCIIAttrDigit) != 0 {
|
|
capitalizeNext = false
|
|
}
|
|
} else {
|
|
if t.attrs[ch]&ASCIIAttrAlphaUpper != 0 {
|
|
result[i] = ch + 32 // A minúscula (normalización)
|
|
} else {
|
|
result[i] = ch
|
|
}
|
|
}
|
|
i++
|
|
}
|
|
|
|
return result
|
|
}
|
|
|
|
// Snakeize convierte a snake_case in-place insertando separadores antes de mayúsculas.
|
|
//
|
|
// # ADVERTENCIA: Expansión de Tamaño
|
|
//
|
|
// A diferencia de Camelize (que compacta), Snakeize puede AUMENTAR el tamaño del slice
|
|
// al insertar separadores. Por esto, normalmente requiere un buffer de destino con
|
|
// espacio extra.
|
|
//
|
|
// Esta implementación calcula el espacio extra necesario y crea un nuevo slice si
|
|
// es necesario. Por lo tanto, NO es realmente in-place puro.
|
|
//
|
|
// # Algoritmo
|
|
//
|
|
// 1. Calcular espacio extra necesario contando mayúsculas
|
|
// 2. Si no hay mayúsculas → convertir a lowercase in-place y retornar
|
|
// 3. Si hay mayúsculas → crear nuevo slice con espacio extra y transformar
|
|
//
|
|
// # Ejemplo
|
|
//
|
|
// attrs := NewASCIIAttrs()
|
|
//
|
|
// camel := []byte("HelloWorld")
|
|
// snake := attrs.Snakeize(camel, '_')
|
|
// // snake: "hello_world"
|
|
//
|
|
// mixed := []byte("getUserByID")
|
|
// result := attrs.Snakeize(mixed, '_')
|
|
// // result: "get_user_by_i_d"
|
|
//
|
|
// # Complejidad
|
|
//
|
|
// - Tiempo: O(n)
|
|
// - Memoria: O(n+e) donde e = número de mayúsculas a insertar
|
|
func (t *ASCIIAttrs) Snakeize(list []byte, separator byte) []byte {
|
|
// Calcular espacio extra necesario
|
|
extra := 0
|
|
for i, ch := range list {
|
|
if i > 0 && t.attrs[ch]&ASCIIAttrAlphaUpper != 0 {
|
|
extra++
|
|
}
|
|
}
|
|
|
|
// Si no hay mayúsculas, solo convertir a lowercase
|
|
if extra == 0 {
|
|
t.Lowerize(list)
|
|
return list
|
|
}
|
|
|
|
// Crear nuevo slice con espacio para separadores
|
|
newLen := len(list) + extra
|
|
result := make([]byte, newLen)
|
|
target := 0
|
|
|
|
for i := 0; i < len(list); i++ {
|
|
ch := list[i]
|
|
|
|
// Insertar separador antes de mayúscula (excepto primera posición)
|
|
if i > 0 && t.attrs[ch]&ASCIIAttrAlphaUpper != 0 {
|
|
result[target] = separator
|
|
target++
|
|
result[target] = ch + 32 // Convertir a lowercase
|
|
} else if t.attrs[ch]&ASCIIAttrAlphaUpper != 0 {
|
|
result[target] = ch + 32 // Primera letra: solo lowercase
|
|
} else {
|
|
result[target] = ch
|
|
}
|
|
target++
|
|
}
|
|
|
|
return result
|
|
}
|
|
|
|
// ToSnake convierte a snake_case manejando inteligentemente CamelCase y separadores.
|
|
//
|
|
// # Comportamiento Avanzado
|
|
//
|
|
// Esta función maneja varios casos complejos:
|
|
//
|
|
// 1. CamelCase → snake_case:
|
|
// - "HelloWorld" → "hello_world"
|
|
// - Inserta '_' antes de mayúscula si la anterior era minúscula/dígito
|
|
//
|
|
// 2. Siglas (secuencias de mayúsculas):
|
|
// - "HTMLParser" → "html_parser" (no "h_t_m_l_parser")
|
|
// - Detecta cuando una mayúscula es seguida por minúscula
|
|
//
|
|
// 3. Normalización de separadores:
|
|
// - Convierte '-', '.', ' ' al separador solicitado
|
|
// - "hello-world" → "hello_world" (si separator='_')
|
|
// - "hello.world" → "hello_world"
|
|
//
|
|
// # Algoritmo de Inserción Inteligente
|
|
//
|
|
// Inserta '_' antes de una mayúscula si:
|
|
// - La anterior era minúscula/dígito: "aB" → "a_b"
|
|
// - O forma parte de sigla antes de minúscula: "ABc" → "a_bc"
|
|
//
|
|
// # Ejemplo
|
|
//
|
|
// attrs := NewASCIIAttrs()
|
|
//
|
|
// // CamelCase básico
|
|
// attrs.ToSnake([]byte("HelloWorld"), '_') // "hello_world"
|
|
//
|
|
// // Siglas
|
|
// attrs.ToSnake([]byte("HTMLParser"), '_') // "html_parser"
|
|
// attrs.ToSnake([]byte("parseHTMLDocument"), '_') // "parse_html_document"
|
|
//
|
|
// // Normalización de separadores
|
|
// attrs.ToSnake([]byte("hello-world"), '_') // "hello_world"
|
|
// attrs.ToSnake([]byte("user.service"), '_') // "user_service"
|
|
//
|
|
// // Mix complejo
|
|
// attrs.ToSnake([]byte("getUserByID"), '_') // "get_user_by_id"
|
|
//
|
|
// # Performance
|
|
//
|
|
// - Pre-calcula tamaño final (evita reallocations)
|
|
// - Zero-copy cuando no hay cambios
|
|
// - Single-pass transformation
|
|
//
|
|
// # Complejidad
|
|
//
|
|
// - Tiempo: O(2n) - pre-scan + transform
|
|
// - Memoria: O(n+s) donde s = separadores insertados
|
|
func (t *ASCIIAttrs) ToSnake(list []byte, separator byte) []byte {
|
|
if len(list) == 0 {
|
|
return list
|
|
}
|
|
|
|
// PASO 1: Calcular longitud extra y detectar necesidad de normalización
|
|
extraLen := 0
|
|
needsNormalization := false
|
|
hasUpper := false
|
|
|
|
for i := 0; i < len(list); i++ {
|
|
ch := list[i]
|
|
|
|
if t.attrs[ch]&ASCIIAttrAlphaUpper != 0 {
|
|
hasUpper = true
|
|
|
|
// Caso 1: "aB" → Insertar '_' entre minúscula/dígito y mayúscula
|
|
if i > 0 && (t.attrs[list[i-1]]&(ASCIIAttrAlphaLower|ASCIIAttrDigit) != 0) {
|
|
extraLen++
|
|
}
|
|
|
|
// Caso 2: "ABc" → Insertar '_' en "A_Bc" (manejo de siglas)
|
|
if i > 0 && i < len(list)-1 &&
|
|
(t.attrs[list[i-1]]&ASCIIAttrAlphaUpper != 0) &&
|
|
(t.attrs[list[i+1]]&ASCIIAttrAlphaLower != 0) {
|
|
extraLen++
|
|
}
|
|
} else {
|
|
// Detectar separadores que necesitan normalización
|
|
if (ch == '-' || ch == '.' || ch == ' ' || ch == '_') && ch != separator {
|
|
needsNormalization = true
|
|
}
|
|
}
|
|
}
|
|
|
|
// Zero-copy optimization
|
|
if extraLen == 0 && !hasUpper && !needsNormalization {
|
|
return list
|
|
}
|
|
|
|
// PASO 2: Construcción del resultado
|
|
result := make([]byte, len(list)+extraLen)
|
|
w := 0
|
|
|
|
for i := 0; i < len(list); i++ {
|
|
ch := list[i]
|
|
|
|
if t.attrs[ch]&ASCIIAttrAlphaUpper != 0 {
|
|
// Lógica de inserción de separador
|
|
if i > 0 && (t.attrs[list[i-1]]&(ASCIIAttrAlphaLower|ASCIIAttrDigit) != 0) {
|
|
result[w] = separator
|
|
w++
|
|
} else if i > 0 && i < len(list)-1 &&
|
|
(t.attrs[list[i-1]]&ASCIIAttrAlphaUpper != 0) &&
|
|
(t.attrs[list[i+1]]&ASCIIAttrAlphaLower != 0) {
|
|
result[w] = separator
|
|
w++
|
|
}
|
|
result[w] = ch + 32 // A minúscula
|
|
} else {
|
|
// Normalización de separadores
|
|
if ch == '-' || ch == '.' || ch == ' ' || ch == '_' {
|
|
result[w] = separator
|
|
} else {
|
|
result[w] = ch
|
|
}
|
|
}
|
|
w++
|
|
}
|
|
|
|
return result
|
|
}
|
|
|
|
// ScreamingSnakeize transforma a SCREAMING_SNAKE_CASE in-place (limitado al tamaño actual).
|
|
//
|
|
// # SCREAMING_SNAKE_CASE
|
|
//
|
|
// Es snake_case pero con todas las letras en mayúsculas:
|
|
// - "hello_world" → "HELLO_WORLD"
|
|
// - Comúnmente usado para constantes: MAX_SIZE, API_KEY
|
|
//
|
|
// # Limitación
|
|
//
|
|
// No puede insertar separadores si no hay espacio (in-place real). Para una versión
|
|
// que inserta separadores, usar ToScreamingSnake.
|
|
//
|
|
// # Ejemplo
|
|
//
|
|
// attrs := NewASCIIAttrs()
|
|
//
|
|
// input := []byte("hello_world")
|
|
// attrs.ScreamingSnakeize(input)
|
|
// // input: "HELLO_WORLD"
|
|
//
|
|
// // Normaliza separadores conocidos
|
|
// input2 := []byte("hello-world")
|
|
// attrs.ScreamingSnakeize(input2)
|
|
// // input2: "HELLO_WORLD"
|
|
func (t *ASCIIAttrs) ScreamingSnakeize(list []byte) []byte {
|
|
for i := 0; i < len(list); i++ {
|
|
ch := list[i]
|
|
|
|
// Convertir minúsculas a mayúsculas
|
|
if t.attrs[ch]&ASCIIAttrAlphaLower != 0 {
|
|
list[i] = ch - 32
|
|
} else if ch == '-' || ch == '.' || ch == ' ' {
|
|
// Normalizar separadores conocidos a '_'
|
|
list[i] = '_'
|
|
}
|
|
}
|
|
return list
|
|
}
|
|
|
|
// ToScreamingSnake convierte a SCREAMING_SNAKE_CASE con manejo completo de CamelCase.
|
|
//
|
|
// Combina la funcionalidad de ToSnake y uppercase:
|
|
// - Detecta CamelCase e inserta '_'
|
|
// - Maneja siglas inteligentemente
|
|
// - Normaliza separadores existentes
|
|
// - Convierte todo a mayúsculas
|
|
//
|
|
// # Ejemplo
|
|
//
|
|
// attrs := NewASCIIAttrs()
|
|
//
|
|
// // CamelCase → SCREAMING_SNAKE
|
|
// attrs.ToScreamingSnake([]byte("HelloWorld"), '_') // "HELLO_WORLD"
|
|
// attrs.ToScreamingSnake([]byte("getUserByID"), '_') // "GET_USER_BY_ID"
|
|
//
|
|
// // Normalización
|
|
// attrs.ToScreamingSnake([]byte("hello-world"), '_') // "HELLO_WORLD"
|
|
// attrs.ToScreamingSnake([]byte("api_key"), '_') // "API_KEY"
|
|
//
|
|
// # Complejidad
|
|
//
|
|
// - Tiempo: O(2n)
|
|
// - Memoria: O(n+s)
|
|
func (t *ASCIIAttrs) ToScreamingSnake(list []byte, separator byte) []byte {
|
|
if len(list) == 0 {
|
|
return list
|
|
}
|
|
|
|
// Detectar necesidad de cambios
|
|
extraLen := 0
|
|
needsUpper := false
|
|
needsNormalization := false
|
|
|
|
for i := 0; i < len(list); i++ {
|
|
ch := list[i]
|
|
|
|
// Detectar inserción de separadores
|
|
if i > 0 && t.attrs[ch]&ASCIIAttrAlphaUpper != 0 {
|
|
if t.attrs[list[i-1]]&(ASCIIAttrAlphaLower|ASCIIAttrDigit) != 0 {
|
|
extraLen++
|
|
} else if i < len(list)-1 && t.attrs[list[i+1]]&ASCIIAttrAlphaLower != 0 {
|
|
extraLen++
|
|
}
|
|
}
|
|
|
|
// Detectar necesidad de uppercase
|
|
if t.attrs[ch]&ASCIIAttrAlphaLower != 0 {
|
|
needsUpper = true
|
|
}
|
|
|
|
// Detectar necesidad de normalización
|
|
if (ch == '-' || ch == '.' || ch == ' ') && ch != separator {
|
|
needsNormalization = true
|
|
}
|
|
}
|
|
|
|
// Zero-copy optimization
|
|
if extraLen == 0 && !needsUpper && !needsNormalization {
|
|
return list
|
|
}
|
|
|
|
// Construcción
|
|
result := make([]byte, len(list)+extraLen)
|
|
w := 0
|
|
|
|
for i := 0; i < len(list); i++ {
|
|
ch := list[i]
|
|
|
|
if t.attrs[ch]&ASCIIAttrAlphaUpper != 0 {
|
|
// Insertar separador si es necesario
|
|
if i > 0 && (t.attrs[list[i-1]]&(ASCIIAttrAlphaLower|ASCIIAttrDigit) != 0) {
|
|
result[w] = separator
|
|
w++
|
|
} else if i > 0 && i < len(list)-1 &&
|
|
(t.attrs[list[i-1]]&ASCIIAttrAlphaUpper != 0) &&
|
|
(t.attrs[list[i+1]]&ASCIIAttrAlphaLower != 0) {
|
|
result[w] = separator
|
|
w++
|
|
}
|
|
result[w] = ch // Ya es mayúscula
|
|
} else if t.attrs[ch]&ASCIIAttrAlphaLower != 0 {
|
|
result[w] = ch - 32 // Convertir a mayúscula
|
|
} else if ch == '-' || ch == '.' || ch == ' ' || ch == '_' {
|
|
result[w] = separator // Normalizar separador
|
|
} else {
|
|
result[w] = ch
|
|
}
|
|
w++
|
|
}
|
|
|
|
return result
|
|
}
|
|
|
|
// ════════════════════════════════════════════════════════════════════════════
|
|
// CONVERSIONES NUMÉRICAS
|
|
// ════════════════════════════════════════════════════════════════════════════
|
|
|
|
// DigitToByte convierte un byte dígito decimal ('0'-'9') a su valor numérico (0-9).
|
|
//
|
|
// # Ejemplo
|
|
//
|
|
// attrs := NewASCIIAttrs()
|
|
// attrs.DigitToByte('0') // 0
|
|
// attrs.DigitToByte('5') // 5
|
|
// attrs.DigitToByte('9') // 9
|
|
// attrs.DigitToByte('A') // 0 (no es dígito)
|
|
//
|
|
// # Complejidad: O(1)
|
|
func (t *ASCIIAttrs) DigitToByte(chr byte) byte {
|
|
if t.attrs[chr]&ASCIIAttrDigit != 0 {
|
|
return chr - '0'
|
|
}
|
|
return 0
|
|
}
|
|
|
|
// HexDigitToByte convierte un byte hexadecimal ('0'-'9','A'-'F','a'-'f') a su valor (0-15).
|
|
//
|
|
// # Ejemplo
|
|
//
|
|
// attrs := NewASCIIAttrs()
|
|
// attrs.HexDigitToByte('0') // 0
|
|
// attrs.HexDigitToByte('9') // 9
|
|
// attrs.HexDigitToByte('A') // 10
|
|
// attrs.HexDigitToByte('F') // 15
|
|
// attrs.HexDigitToByte('a') // 10
|
|
// attrs.HexDigitToByte('f') // 15
|
|
// attrs.HexDigitToByte('G') // 0 (no es hex)
|
|
//
|
|
// # Complejidad: O(1)
|
|
func (t *ASCIIAttrs) HexDigitToByte(chr byte) byte {
|
|
att := t.attrs[chr]
|
|
if att&ASCIIAttrDigitHex != 0 {
|
|
if att&ASCIIAttrDigit != 0 {
|
|
return chr - '0'
|
|
}
|
|
if att&ASCIIAttrAlphaLower != 0 {
|
|
return chr - 'a' + 10
|
|
}
|
|
return chr - 'A' + 10
|
|
}
|
|
return 0
|
|
}
|
|
|
|
// IsHex2Digit verifica si dos bytes consecutivos son dígitos hexadecimales válidos.
|
|
//
|
|
// # Uso Típico
|
|
//
|
|
// Verificar antes de decodificar URL encoding: "%20" → ' '
|
|
//
|
|
// # Ejemplo
|
|
//
|
|
// attrs := NewASCIIAttrs()
|
|
// attrs.IsHex2Digit('2', '0') // true ("%20")
|
|
// attrs.IsHex2Digit('F', 'F') // true ("%FF")
|
|
// attrs.IsHex2Digit('G', '0') // false (G no es hex)
|
|
//
|
|
// # Complejidad: O(1)
|
|
func (t *ASCIIAttrs) IsHex2Digit(cha, chb byte) bool {
|
|
return (t.attrs[cha]&ASCIIAttrDigitHex != 0) && (t.attrs[chb]&ASCIIAttrDigitHex != 0)
|
|
}
|
|
|
|
// Hex2DigitToByte combina dos dígitos hexadecimales en un byte.
|
|
//
|
|
// # IMPORTANTE
|
|
//
|
|
// No valida si los bytes son hex válidos. Usar IsHex2Digit primero si no estás seguro.
|
|
//
|
|
// # Ejemplo
|
|
//
|
|
// attrs := NewASCIIAttrs()
|
|
// attrs.Hex2DigitToByte('2', '0') // 32 (espacio ' ')
|
|
// attrs.Hex2DigitToByte('F', 'F') // 255
|
|
// attrs.Hex2DigitToByte('4', '1') // 65 ('A')
|
|
//
|
|
// # Uso en URL Decoding
|
|
//
|
|
// if attrs.IsHex2Digit(url[i+1], url[i+2]) {
|
|
// decoded = attrs.Hex2DigitToByte(url[i+1], url[i+2])
|
|
// }
|
|
//
|
|
// # Complejidad: O(1)
|
|
func (t *ASCIIAttrs) Hex2DigitToByte(cha, chb byte) byte {
|
|
return t.HexDigitToByte(cha)<<4 | t.HexDigitToByte(chb)
|
|
}
|
|
|
|
// GetHex2DigitToByte combina dos dígitos hex en un byte Y valida.
|
|
//
|
|
// A diferencia de Hex2DigitToByte, esta función valida y retorna un bool indicando éxito.
|
|
//
|
|
// # Ejemplo
|
|
//
|
|
// attrs := NewASCIIAttrs()
|
|
//
|
|
// val, ok := attrs.GetHex2DigitToByte('2', '0')
|
|
// if ok {
|
|
// // val = 32 (espacio)
|
|
// }
|
|
//
|
|
// val, ok = attrs.GetHex2DigitToByte('G', '0')
|
|
// // ok = false, val = 0
|
|
//
|
|
// # Complejidad: O(1)
|
|
func (t *ASCIIAttrs) GetHex2DigitToByte(cha, chb byte) (byte, bool) {
|
|
at, bt := t.attrs[cha], t.attrs[chb]
|
|
|
|
// Validar que ambos sean hex
|
|
if at&ASCIIAttrDigitHex == 0 || bt&ASCIIAttrDigitHex == 0 {
|
|
return 0, false
|
|
}
|
|
|
|
// Convertir primer dígito
|
|
var va byte
|
|
if at&ASCIIAttrDigit != 0 {
|
|
va = cha - '0'
|
|
} else if at&ASCIIAttrAlphaLower != 0 {
|
|
va = cha - 'a' + 10
|
|
} else {
|
|
va = cha - 'A' + 10
|
|
}
|
|
va <<= 4
|
|
|
|
// Convertir segundo dígito
|
|
if bt&ASCIIAttrDigit != 0 {
|
|
return va | (chb - '0'), true
|
|
}
|
|
if bt&ASCIIAttrAlphaLower != 0 {
|
|
return va | (chb - 'a' + 10), true
|
|
}
|
|
return va | (chb - 'A' + 10), true
|
|
}
|
|
|
|
// ════════════════════════════════════════════════════════════════════════════
|
|
// PREDICADOS DE CLASIFICACIÓN
|
|
// ════════════════════════════════════════════════════════════════════════════
|
|
|
|
// Todas estas funciones son O(1) - un lookup + AND bitwise.
|
|
|
|
// IsControl verifica si el byte es un carácter de control ASCII (0-31 o 127).
|
|
func (t *ASCIIAttrs) IsControl(a byte) bool {
|
|
return t.attrs[a]&ASCIIAttrControl != 0
|
|
}
|
|
|
|
// IsPrintable verifica si el byte es un carácter imprimible (32-126).
|
|
func (t *ASCIIAttrs) IsPrintable(a byte) bool {
|
|
return t.attrs[a]&ASCIIAttrPrintable != 0
|
|
}
|
|
|
|
// IsAlpha verifica si el byte es una letra (A-Z o a-z).
|
|
func (t *ASCIIAttrs) IsAlpha(a byte) bool {
|
|
return t.attrs[a]&ASCIIAttrAlpha != 0
|
|
}
|
|
|
|
// IsAlphaLower verifica si el byte es una letra minúscula (a-z).
|
|
func (t *ASCIIAttrs) IsAlphaLower(a byte) bool {
|
|
return t.attrs[a]&ASCIIAttrAlphaLower != 0
|
|
}
|
|
|
|
// IsAlphaUpper verifica si el byte es una letra mayúscula (A-Z).
|
|
func (t *ASCIIAttrs) IsAlphaUpper(a byte) bool {
|
|
return t.attrs[a]&ASCIIAttrAlphaUpper != 0
|
|
}
|
|
|
|
// IsDigit verifica si el byte es un dígito decimal ('0'-'9').
|
|
func (t *ASCIIAttrs) IsDigit(a byte) bool {
|
|
return t.attrs[a]&ASCIIAttrDigit != 0
|
|
}
|
|
|
|
// IsHex verifica si el byte es un dígito hexadecimal adicional (A-F, a-f).
|
|
// No incluye 0-9. Para verificar cualquier dígito hex, usar IsHexDigit.
|
|
func (t *ASCIIAttrs) IsHex(a byte) bool {
|
|
return t.attrs[a]&ASCIIAttrHex != 0
|
|
}
|
|
|
|
// IsAlphaNum verifica si el byte es alfanumérico (letra o dígito).
|
|
func (t *ASCIIAttrs) IsAlphaNum(a byte) bool {
|
|
return t.attrs[a]&(ASCIIAttrAlphaNum) != 0
|
|
}
|
|
|
|
// IsAlphaNumLower verifica si el byte es alfanumérico (letra minúscula o dígito).
|
|
func (t *ASCIIAttrs) IsAlphaNumLower(a byte) bool {
|
|
return t.attrs[a]&(ASCIIAttrAlphaNumLower) != 0
|
|
}
|
|
|
|
// IsAlphaNumUpper verifica si el byte es alfanumérico (letra mayúscula o dígito).
|
|
func (t *ASCIIAttrs) IsAlphaNumUpper(a byte) bool {
|
|
return t.attrs[a]&(ASCIIAttrAlphaNumUpper) != 0
|
|
}
|
|
|
|
// IsHexDigit verifica si el byte es cualquier dígito hexadecimal (0-9, A-F, a-f).
|
|
func (t *ASCIIAttrs) IsHexDigit(a byte) bool {
|
|
return t.attrs[a]&ASCIIAttrDigitHex != 0
|
|
}
|