/* —————————————————————————————————————————————————————————————————————————— Proyecto goat ————————————————————————————————————————————————————————————————————————————— Fichero ascii_attrs.go Package btes Autor Juan V. Navarro juanvnl@activething.com Creado 02/02/2026 ————————————————————————————————————————————————————————————————————————————— LICENSES AND TERMS OF USE ------------------------- This software is licensed under the Elastic License v2.0 (the "License"). For full terms and additional information regarding permitted and prohibited uses, please visit: https://activething.com/ATGO/licenses You may use, copy, modify, and redistribute this software internally within your organization for any purpose, including research, development, and testing, subject to the terms of this License. You may NOT, however, use, provide, distribute, or make this software available to any third party as part of a hosted service, SaaS offering, or commercial product without first obtaining a commercial license from Active Thing. You may combine this software with other code, provided that such combination does not circumvent the restrictions of this License. THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE, AND NON-INFRINGEMENT. In no event shall the authors or copyright holders be liable for any claim, damages, or other liability arising from the use of this software. ————————————————————————————————————————————————————————————————————————————— Web : activething.com | activething.com/goat git : g.activething.com | github.com/activething/goat Correo : dev@activething.com ————————————————————————————————————————————————————————————————————————————— No deseo caminar sobre el agua", dijo Siddhartha. Que los antiguos chamanes se contenten con tales habilidades. —— Hermann Hesse, Siddhartha ————————————————————————————————————————————————————————————————————————————— Copyright (c) 2026 Active Thing ————————————————————————————————————————————————————————————————————————————— */ // Package btes proporciona operaciones de alto rendimiento sobre bytes ASCII // mediante el uso de tablas de búsqueda (Lookup Tables - LUT) para clasificación // y transformación en tiempo constante O(1). // // La estrategia principal es pre-calcular atributos de cada byte ASCII posible // (256 valores) en una tabla de 256 entradas, permitiendo verificar propiedades // de caracteres mediante simples operaciones bitwise sin comparaciones costosas. // // # Casos de Uso Principales // // - Routing HTTP: normalización de paths, métodos y headers // - Parsing de URLs: detección de caracteres válidos, decode hexadecimal // - Validación de identificadores: verificación alfanumérica rápida // - Transformación de strings: case conversions sin allocations innecesarias // // # Ventajas de Rendimiento // // - IsAlpha(ch): ~1ns vs 8-15ns con unicode.IsLetter // - ToUpper(s): 4x más rápido que implementación naive con allocation tardía // - EqualFold(a,b): ~27ns vs ~38ns con strings.EqualFold para ASCII puro package btes // ════════════════════════════════════════════════════════════════════════════ // CONSTANTES DE ATRIBUTOS ASCII // ════════════════════════════════════════════════════════════════════════════ // Máscaras de bits para clasificar caracteres ASCII. Cada constante representa // un bit diferente, permitiendo que un byte tenga múltiples atributos simultáneos // mediante operaciones OR (|). // // Ejemplo: 'A' tiene atributos: ASCIIAttrPrintable | ASCIIAttrAlpha | ASCIIAttrAlphaUpper const ( // ASCIIAttrControl identifica caracteres de control ASCII (0-31, 127). // Incluye: \n, \r, \t, ESC, NULL, etc. ASCIIAttrControl Attr = 1 << iota // ASCIIAttrPrintable identifica caracteres imprimibles (32-126). // Excluye caracteres de control pero incluye espacio (32). ASCIIAttrPrintable // ASCIIAttrAlpha identifica letras (A-Z, a-z). // Útil para validar identificadores o paths alfabéticos. ASCIIAttrAlpha // ASCIIAttrAlphaLower identifica letras minúsculas (a-z). ASCIIAttrAlphaLower // ASCIIAttrAlphaUpper identifica letras mayúsculas (A-Z). ASCIIAttrAlphaUpper // ASCIIAttrDigit identifica dígitos decimales ('0'-'9'). ASCIIAttrDigit // ASCIIAttrHex identifica dígitos hexadecimales adicionales (A-F, a-f). // No incluye 0-9, que ya tienen ASCIIAttrDigit. ASCIIAttrHex // ASCIIAttrDigitHex es una máscara combinada para verificar si un byte // es un dígito hexadecimal válido (0-9, A-F, a-f). // Equivale a: ASCIIAttrHex | ASCIIAttrDigit ASCIIAttrDigitHex = ASCIIAttrHex | ASCIIAttrDigit ASCIIAttrAlphaNum = ASCIIAttrAlpha | ASCIIAttrDigit ASCIIAttrAlphaNumLower = ASCIIAttrAlphaLower | ASCIIAttrDigit ASCIIAttrAlphaNumUpper = ASCIIAttrAlphaUpper | ASCIIAttrDigit ) // ════════════════════════════════════════════════════════════════════════════ // TIPO PRINCIPAL: ASCIIAttrs // ════════════════════════════════════════════════════════════════════════════ // ASCIIAttrs encapsula una tabla de atributos (LUT) para los 256 valores ASCII. // Cada entrada de la tabla contiene los atributos del byte correspondiente // codificados como bits en un uint8. // // La estructura es inmutable después de la inicialización, lo que permite // compartirla de forma segura entre goroutines sin sincronización. // // # Complejidad de Operaciones // // - Verificación de atributos: O(1) - un lookup + AND bitwise // - Transformaciones individuales: O(1) - un lookup + aritmética // - Transformaciones de slices: O(n) - donde n es el tamaño del slice // // # Uso de Memoria // // - Tamaño de tabla: 256 bytes (una entrada por byte ASCII posible) // - Overhead por instancia: ~8 bytes (puntero + metadatos struct) // // # Ejemplo Básico // // attrs := NewASCIIAttrs() // // // Verificaciones O(1) // if attrs.IsDigit('5') { // // true // } // // // Transformaciones con zero-copy cuando sea posible // upper := attrs.ToUpper([]byte("hello")) // "HELLO" // same := attrs.ToUpper([]byte("HELLO")) // retorna input (zero-copy) type ASCIIAttrs struct { // attrs es la tabla interna de 256 entradas que mapea cada byte ASCII // a sus atributos codificados como bits. attrs Attrs } // ════════════════════════════════════════════════════════════════════════════ // CONSTRUCTOR // ════════════════════════════════════════════════════════════════════════════ // NewASCIIAttrs crea y retorna una nueva instancia de ASCIIAttrs con la tabla // de atributos ASCII completamente inicializada. // // La tabla se pre-calcula una sola vez en la construcción con los siguientes // atributos para cada rango de caracteres ASCII: // // - Control (0-31, 127): Caracteres no imprimibles // - Printable (32-126): Caracteres imprimibles estándar // - Digit ('0'-'9'): Dígitos decimales // - AlphaUpper ('A'-'Z'): Letras mayúsculas // - AlphaLower ('a'-'z'): Letras minúsculas // - Hex ('A'-'F', 'a'-'f'): Dígitos hexadecimales adicionales // // # Complejidad // // - Tiempo: O(256) - inicialización de tabla completa // - Memoria: 256 bytes para la tabla + overhead de struct // // # Thread Safety // // Esta función es thread-safe. Cada llamada retorna una nueva instancia // independiente que puede usarse concurrentemente sin sincronización. // // # Ejemplo // // // Típicamente se crea una instancia global // var asciiAttrs = NewASCIIAttrs() // // // O se crea bajo demanda // func processPath(path []byte) { // attrs := NewASCIIAttrs() // upper := attrs.ToUpper(path) // // ... // } func NewASCIIAttrs() *ASCIIAttrs { a := &ASCIIAttrs{} // Inicializar caracteres de control (0-31 + DEL) a.attrs.SetRange(ASCIIAttrControl, 0, 31) a.attrs[127] = ASCIIAttrControl // Inicializar caracteres imprimibles (espacio hasta ~) a.attrs.SetRange(ASCIIAttrPrintable, 32, 126) // Inicializar dígitos decimales a.attrs.AddRange(ASCIIAttrDigit, '0', '9') // Inicializar letras (marcando Alpha + Upper/Lower según corresponda) a.attrs.AddRange(ASCIIAttrAlpha|ASCIIAttrAlphaUpper, 'A', 'Z') a.attrs.AddRange(ASCIIAttrAlpha|ASCIIAttrAlphaLower, 'a', 'z') // Inicializar dígitos hexadecimales adicionales (A-F, a-f) // Los dígitos 0-9 ya están marcados con ASCIIAttrDigit a.attrs.AddRange(ASCIIAttrHex, 'A', 'F') a.attrs.AddRange(ASCIIAttrHex, 'a', 'f') return a } // ════════════════════════════════════════════════════════════════════════════ // COMPARACIÓN CASE-INSENSITIVE // ════════════════════════════════════════════════════════════════════════════ // EqualFold compara dos slices de bytes ignorando diferencias de mayúsculas/minúsculas // en letras ASCII. Es equivalente a strings.EqualFold pero optimizado para ASCII puro. // // # Algoritmo // // 1. Verificación de longitud (early exit si difieren) // 2. Comparación byte por byte: // - Si son iguales → continuar // - Si difieren → normalizar a minúsculas con |0x20 y verificar si son letras // // # Técnica de Normalización // // Usa el truco de |0x20 para convertir cualquier letra ASCII a minúscula: // - 'A' (65) | 0x20 = 'a' (97) // - 'a' (97) | 0x20 = 'a' (97) // - '5' (53) | 0x20 = 'u' (117) ← Por eso verificamos IsAlpha después // // # Complejidad // // - Tiempo: O(n) donde n = len(source) // - Memoria: O(1) - no allocations // - Early exit: Retorna false inmediatamente al primer mismatch // // # Performance // // - ~27ns para strings típicos de 8-12 bytes (métodos HTTP, headers cortos) // - ~38% más rápido que strings.EqualFold para ASCII puro // - Beneficio aumenta con strings más largos por mejor inlining y cache locality // // # Casos de Uso // // - Comparar métodos HTTP: "GET" vs "get" // - Comparar headers HTTP: "Content-Type" vs "content-type" // - Routing case-insensitive: "/API/Users" vs "/api/users" // // # Ejemplo // // attrs := NewASCIIAttrs() // // // Comparaciones case-insensitive // attrs.EqualFold([]byte("GET"), []byte("get")) // true // attrs.EqualFold([]byte("Content-Type"), []byte("content-type")) // true // attrs.EqualFold([]byte("hello"), []byte("world")) // false // attrs.EqualFold([]byte("abc"), []byte("ABC123")) // false (longitud) // // # Limitaciones // // Solo funciona correctamente con ASCII. Para Unicode, usar strings.EqualFold // o unicode.SimpleFold. func (t *ASCIIAttrs) EqualFold(source, target []byte) bool { // Early exit: longitudes diferentes nunca pueden ser iguales if len(source) != len(target) { return false } // Comparación byte a byte for i := 0; i < len(source); i++ { s, tr := source[i], target[i] // Fast path: bytes exactamente iguales if s == tr { continue } // Slow path: verificar si son la misma letra en diferentes casos // |0x20 convierte A-Z a a-z (si es letra) // Luego verificamos que realmente sea una letra para evitar falsos positivos if (s|0x20) == (tr|0x20) && t.attrs[s]&ASCIIAttrAlpha != 0 { continue } // No son iguales ni case-insensitive return false } return true } // SimpleLetterEqualFold verifica si dos slices tienen el mismo patrón de // mayúsculas/minúsculas en las posiciones de letras. // // A diferencia de EqualFold, esta función NO ignora las diferencias de case. // En su lugar, verifica que cuando hay una letra mayúscula en source[i], también // hay una letra mayúscula en target[i] (aunque sean letras diferentes). // // # Casos de Uso // // - Validar consistencia de formato en identificadores // - Verificar que el patrón de capitalización sea consistente // - Validación de schemas donde el case pattern importa // // # Ejemplo // // attrs := NewASCIIAttrs() // // // Mismo patrón de case (ambos empiezan con mayúscula) // attrs.SimpleLetterEqualFold([]byte("HelloWorld"), []byte("GreatThing")) // true // // // Diferente patrón (uno empieza minúscula) // attrs.SimpleLetterEqualFold([]byte("hello"), []byte("World")) // false // // // Mismo contenido pero en ambos // attrs.SimpleLetterEqualFold([]byte("Hello"), []byte("Hello")) // true // // # Complejidad // // - Tiempo: O(n) // - Memoria: O(1) func (t *ASCIIAttrs) SimpleLetterEqualFold(source, target []byte) bool { if len(source) != len(target) { return false } for i, b := range source { // Verificar que ambos bytes tengan el mismo atributo de case // (ambos uppercase o ambos no-uppercase) if t.IsAlphaUpper(b) != t.IsAlphaUpper(target[i]) { return false } } return true } // ════════════════════════════════════════════════════════════════════════════ // CONVERSIONES DE CASE - BYTE INDIVIDUAL // ════════════════════════════════════════════════════════════════════════════ // ToUpperByte convierte un byte individual a mayúscula si es una letra minúscula ASCII. // // # Algoritmo // // En ASCII, la diferencia entre una letra mayúscula y su equivalente minúscula es 32: // - 'a' = 97, 'A' = 65 → diferencia = 32 // - 'z' = 122, 'Z' = 90 → diferencia = 32 // // Por lo tanto: mayúscula = minúscula - 32 // // # Complejidad // // - Tiempo: O(1) - un lookup + una resta condicional // - Memoria: O(1) - sin allocations // // # Ejemplo // // attrs := NewASCIIAttrs() // attrs.ToUpperByte('a') // 'A' // attrs.ToUpperByte('Z') // 'Z' (sin cambios) // attrs.ToUpperByte('5') // '5' (sin cambios) func (t *ASCIIAttrs) ToUpperByte(chr byte) byte { if t.attrs[chr]&ASCIIAttrAlphaLower != 0 { return chr - 32 } return chr } // ToLowerByte convierte un byte individual a minúscula si es una letra mayúscula ASCII. // // # Algoritmo // // Inverso de ToUpperByte: minúscula = mayúscula + 32 // // # Complejidad // // - Tiempo: O(1) // - Memoria: O(1) // // # Ejemplo // // attrs := NewASCIIAttrs() // attrs.ToLowerByte('A') // 'a' // attrs.ToLowerByte('z') // 'z' (sin cambios) // attrs.ToLowerByte('5') // '5' (sin cambios) func (t *ASCIIAttrs) ToLowerByte(chr byte) byte { if t.attrs[chr]&ASCIIAttrAlphaUpper != 0 { return chr + 32 } return chr } // ════════════════════════════════════════════════════════════════════════════ // CONVERSIONES DE CASE - IN-PLACE (MUTANTES) // ════════════════════════════════════════════════════════════════════════════ // Upperize modifica el slice in-place convirtiendo todas las letras minúsculas // a mayúsculas. Es la versión más rápida cuando se puede mutar el slice original. // // # Performance // // - ~43ns para slice típico de 16 bytes // - Sin allocations (0 B/op) // - Cache-friendly: acceso secuencial lineal // // # Casos de Uso // // - Normalizar paths temporales antes de lookup en tabla // - Procesar buffers reutilizables // - Cuando el slice original no se necesita preservar // // # Ejemplo // // attrs := NewASCIIAttrs() // // path := []byte("hello/world") // attrs.Upperize(path) // // path ahora es "HELLO/WORLD" // // // CUIDADO: El slice original está modificado // original := []byte("test") // attrs.Upperize(original) // fmt.Println(string(original)) // "TEST" (¡modificado!) // // # Complejidad // // - Tiempo: O(n) donde n = len(list) // - Memoria: O(1) - no allocations func (t *ASCIIAttrs) Upperize(list []byte) { for ix, ch := range list { if t.attrs[ch]&ASCIIAttrAlphaLower != 0 { list[ix] -= 32 } } } // Lowerize modifica el slice in-place convirtiendo todas las letras mayúsculas // a minúsculas. // // Equivalente a Upperize pero en dirección opuesta. Ver Upperize para más detalles. // // # Ejemplo // // attrs := NewASCIIAttrs() // // method := []byte("GET") // attrs.Lowerize(method) // // method ahora es "get" func (t *ASCIIAttrs) Lowerize(list []byte) { for ix, ch := range list { if t.attrs[ch]&ASCIIAttrAlphaUpper != 0 { list[ix] += 32 } } } // ════════════════════════════════════════════════════════════════════════════ // CONVERSIONES DE CASE - COPY-ON-WRITE (INMUTABLES) // ════════════════════════════════════════════════════════════════════════════ // ToUpper devuelve una copia del slice con todas las letras minúsculas convertidas // a mayúsculas. Si el slice ya está completamente en mayúsculas, retorna el slice // original sin hacer copia (zero-copy optimization). // // # Algoritmo Optimizado (Pre-Scan Strategy) // // 1. Pre-scan: Escanear el slice completo buscando minúsculas // - Si no encuentra ninguna → retornar slice original (zero-copy) // - Si encuentra al menos una → proceder a paso 2 // // 2. Allocation: Crear nuevo slice del mismo tamaño UNA vez // // 3. Transform: Recorrer original y copiar/transformar a resultado // - Si es minúscula → copiar como mayúscula (byte - 32) // - Si no es minúscula → copiar tal cual // // # Por Qué Esta Estrategia es Óptima // // ## Comparación con Estrategia Naive: // // // ❌ NAIVE (malo): // for ix, ch := range list { // if isLower(ch) { // if result == nil { // result = make([]byte, len(list)) // Allocation tardía // copy(result, list[:ix]) // Copy de bytes ya visitados // } // result[ix] = ch - 32 // } // } // // Problemas de estrategia naive: // - Allocation DENTRO del loop (primera minúscula encontrada) // - copy() ejecutado DESPUÉS de iterar ix bytes // - En peor caso (minúscula al final): itera N → alloc → copy N → transforma 1 // - Total: 2N operaciones de memoria // // ## Ventajas de Pre-Scan: // // - Máximo 2 scans completos (pre-scan + transform) // - Allocation UNA vez al principio // - No hay copy() separado (integrado en transform) // - Zero-copy cuando no hay cambios (común en HTTP: métodos ya uppercase) // // # Performance // // - ~75ns para slice típico con cambios (4x mejora vs naive 307ns) // - ~5ns para slice sin cambios (zero-copy, solo pre-scan) // - 1 allocation cuando necesario vs allocations múltiples en naive // // # Casos de Uso // // - Normalizar métodos HTTP: "get" → "GET" (pero "GET" → "GET" zero-copy) // - Normalizar headers HTTP antes de comparación // - Convertir paths a formato canónico // - Cualquier caso donde el original debe preservarse // // # Ejemplo // // attrs := NewASCIIAttrs() // // // Caso 1: Necesita transformación (hace copia) // lower := []byte("hello") // upper := attrs.ToUpper(lower) // "HELLO" (nuevo slice) // // lower sigue siendo "hello" (original preservado) // // // Caso 2: Ya está uppercase (zero-copy) // already := []byte("HELLO") // same := attrs.ToUpper(already) // retorna already (mismo slice) // // No hay allocation ni copia // // // Caso 3: Mezclado (hace copia) // mixed := []byte("HeLLo") // result := attrs.ToUpper(mixed) // "HELLO" (nuevo slice) // // # Complejidad // // - Tiempo: // - Mejor caso (sin cambios): O(n) - solo pre-scan // - Peor caso (con cambios): O(2n) - pre-scan + transform // - Memoria: // - Mejor caso: O(1) - no allocations // - Peor caso: O(n) - un nuevo slice // // # Thread Safety // // Es thread-safe para el slice de entrada (no lo modifica). Cada llamada que // necesita transformación retorna un nuevo slice independiente. func (t *ASCIIAttrs) ToUpper(list []byte) []byte { // FASE 1: Pre-scan para detectar si hay minúsculas // Early exit si no hay cambios necesarios needsChange := false for _, ch := range list { if t.attrs[ch]&ASCIIAttrAlphaLower != 0 { needsChange = true break // No necesitamos seguir escaneando } } // Zero-copy optimization: retornar original si ya está uppercase if !needsChange { return list } // FASE 2: Allocation UNA vez + transformation en single-pass result := make([]byte, len(list)) for i, ch := range list { if t.attrs[ch]&ASCIIAttrAlphaLower != 0 { result[i] = ch - 32 // Convertir a mayúscula } else { result[i] = ch // Copiar tal cual } } return result } // ToLower devuelve una copia del slice con todas las letras mayúsculas convertidas // a minúsculas. Usa la misma estrategia optimizada que ToUpper. // // Ver documentación de ToUpper para detalles del algoritmo y optimizaciones. // // # Ejemplo // // attrs := NewASCIIAttrs() // // upper := []byte("HELLO") // lower := attrs.ToLower(upper) // "hello" (nuevo slice) // // alreadyLower := []byte("hello") // same := attrs.ToLower(alreadyLower) // retorna alreadyLower (zero-copy) func (t *ASCIIAttrs) ToLower(list []byte) []byte { // Pre-scan para detectar mayúsculas needsChange := false for _, ch := range list { if t.attrs[ch]&ASCIIAttrAlphaUpper != 0 { needsChange = true break } } // Zero-copy si no hay cambios if !needsChange { return list } // Single allocation + transform result := make([]byte, len(list)) for i, ch := range list { if t.attrs[ch]&ASCIIAttrAlphaUpper != 0 { result[i] = ch + 32 // Convertir a minúscula } else { result[i] = ch } } return result } // ════════════════════════════════════════════════════════════════════════════ // TRANSFORMACIONES DE NAMING CONVENTIONS // ════════════════════════════════════════════════════════════════════════════ // Camelize transforma el slice a CamelCase in-place, eliminando los separadores // y capitalizando la primera letra de cada palabra. // // # Formato CamelCase // // En CamelCase, cada palabra (excepto posiblemente la primera) comienza con mayúscula // y no hay espacios ni separadores: // - snake_case "hello_world" → CamelCase "HelloWorld" // - kebab-case "hello-world" → CamelCase "HelloWorld" // // # Comportamiento // // - Elimina todos los separadores encontrados // - Capitaliza la primera letra después de cada separador (o al inicio) // - Normaliza el resto de letras a minúsculas // - El slice se compacta (reduce tamaño) si había separadores // // # IMPORTANTE: Modificación In-Place // // Esta función modifica el slice original Y retorna un re-slice con la nueva // longitud (menor si había separadores). El slice retornado comparte el mismo // backing array que el original. // // # Algoritmo // // 1. Usar dos índices: readIdx (lectura) y writeIdx (escritura) // 2. Para cada byte leído: // - Si es separador: marcar que siguiente letra debe capitalizarse, no escribir // - Si debe capitalizarse: escribir como mayúscula, desmarcar flag // - Sino: escribir como minúscula // 3. Retornar slice[:writeIdx] con nueva longitud // // # Ejemplo // // attrs := NewASCIIAttrs() // // input := []byte("hello_world_test") // result := attrs.Camelize(input, '_') // // result: "HelloWorldTest" (len=14) // // input: "HelloWorldTestst" (¡modificado! últimos bytes son basura) // // // SIEMPRE usar el slice retornado: // camel := attrs.Camelize([]byte("user_service_handler"), '_') // fmt.Println(string(camel)) // "UserServiceHandler" // // # Complejidad // // - Tiempo: O(n) // - Memoria: O(1) - no allocations, modifica in-place // // # Casos de Uso // // - Transformar nombres de variables de snake_case a CamelCase // - Normalizar identificadores desde diferentes convenciones // - Procesar parámetros de configuración func (t *ASCIIAttrs) Camelize(list []byte, separator byte) []byte { if len(list) == 0 { return list } writeIdx := 0 capitalizeNext := true for readIdx := 0; readIdx < len(list); readIdx++ { ch := list[readIdx] // Si encontramos separador: marcamos que siguiente letra va en mayúscula // y NO escribimos el separador (lo eliminamos) if ch == separator { capitalizeNext = true continue // Salta el separador (compactación) } // Aplicar transformación según estado if capitalizeNext { // Primera letra de palabra: capitalizar si es minúscula if t.attrs[ch]&ASCIIAttrAlphaLower != 0 { list[writeIdx] = ch - 32 // A mayúscula } else { list[writeIdx] = ch } // Solo desactivar capitalización si encontramos un alfanumérico // (ignora símbolos/espacios al inicio de palabra) if t.attrs[ch]&(ASCIIAttrAlpha|ASCIIAttrDigit) != 0 { capitalizeNext = false } } else { // Resto de la palabra: normalizar a minúsculas if t.attrs[ch]&ASCIIAttrAlphaUpper != 0 { list[writeIdx] = ch + 32 // A minúscula } else { list[writeIdx] = ch } } writeIdx++ } // Retornar re-slice con nueva longitud return list[:writeIdx] } // ToCamel convierte a CamelCase retornando un nuevo slice, preservando el original. // Usa estrategia optimizada de pre-scan para evitar allocations innecesarias. // // # Diferencia con Camelize // // - Camelize: Modifica in-place, más rápido, usa mismo backing array // - ToCamel: Crea copia si necesario, preserva original, zero-copy cuando posible // // # Algoritmo Optimizado // // 1. Pre-scan Phase: // - Contar separadores (para calcular tamaño final) // - Detectar si necesita cambios (early exit si ya está en formato correcto) // // 2. Decision: // - Si no necesita cambios → retornar original (zero-copy) // - Si necesita cambios → continuar a fase 3 // // 3. Transform Phase: // - Allocar slice con tamaño final conocido (len - sepCount) // - Copiar y transformar simultáneamente // // # Performance // // - ~115ns con transformación (4x mejora vs implementación naive) // - ~10ns sin transformación (zero-copy) // - Evita allocations innecesarias cuando input ya está en CamelCase // // # Ejemplo // // attrs := NewASCIIAttrs() // // // Transformación necesaria // snake := []byte("hello_world") // camel := attrs.ToCamel(snake, '_') // // snake: "hello_world" (sin cambios) // // camel: "HelloWorld" (nuevo slice) // // // Zero-copy (ya es CamelCase sin separadores) // already := []byte("HelloWorld") // same := attrs.ToCamel(already, '_') // // same apunta a already (no hay copia) // // // Con múltiples separadores // multi := []byte("user_service_handler") // result := attrs.ToCamel(multi, '_') // // result: "UserServiceHandler" // // # Complejidad // // - Tiempo: // - Mejor caso (sin cambios): O(n) - solo pre-scan // - Peor caso (con cambios): O(2n) - pre-scan + transform // - Memoria: // - Mejor caso: O(1) - zero-copy // - Peor caso: O(n-s) - donde s = número de separadores func (t *ASCIIAttrs) ToCamel(list []byte, separator byte) []byte { if len(list) == 0 { return list } // FASE 1: Pre-scan para detectar cambios necesarios Y contar separadores sepCount := 0 needsChange := false capitalizeNext := true for _, ch := range list { if ch == separator { sepCount++ needsChange = true // Siempre necesita cambios si hay separadores capitalizeNext = true continue } // Verificar si necesita cambio según posición if capitalizeNext { // Primera letra de palabra: debe ser mayúscula if t.attrs[ch]&ASCIIAttrAlphaLower != 0 { needsChange = true // Es minúscula, necesita cambio } // Consumir estado solo si es alfanumérico if t.attrs[ch]&(ASCIIAttrAlpha|ASCIIAttrDigit) != 0 { capitalizeNext = false } } else { // Resto de la palabra: debe ser minúscula if t.attrs[ch]&ASCIIAttrAlphaUpper != 0 { needsChange = true // Es mayúscula, necesita normalización } } } // Zero-copy optimization if !needsChange { return list } // FASE 2: Allocation con tamaño final conocido finalLen := len(list) - sepCount if finalLen == 0 { return []byte{} } result := make([]byte, finalLen) // FASE 3: Transform en single-pass i := 0 capitalizeNext = true for _, ch := range list { if ch == separator { capitalizeNext = true continue } if capitalizeNext { if t.attrs[ch]&ASCIIAttrAlphaLower != 0 { result[i] = ch - 32 // A mayúscula } else { result[i] = ch } if t.attrs[ch]&(ASCIIAttrAlpha|ASCIIAttrDigit) != 0 { capitalizeNext = false } } else { if t.attrs[ch]&ASCIIAttrAlphaUpper != 0 { result[i] = ch + 32 // A minúscula (normalización) } else { result[i] = ch } } i++ } return result } // Snakeize convierte a snake_case in-place insertando separadores antes de mayúsculas. // // # ADVERTENCIA: Expansión de Tamaño // // A diferencia de Camelize (que compacta), Snakeize puede AUMENTAR el tamaño del slice // al insertar separadores. Por esto, normalmente requiere un buffer de destino con // espacio extra. // // Esta implementación calcula el espacio extra necesario y crea un nuevo slice si // es necesario. Por lo tanto, NO es realmente in-place puro. // // # Algoritmo // // 1. Calcular espacio extra necesario contando mayúsculas // 2. Si no hay mayúsculas → convertir a lowercase in-place y retornar // 3. Si hay mayúsculas → crear nuevo slice con espacio extra y transformar // // # Ejemplo // // attrs := NewASCIIAttrs() // // camel := []byte("HelloWorld") // snake := attrs.Snakeize(camel, '_') // // snake: "hello_world" // // mixed := []byte("getUserByID") // result := attrs.Snakeize(mixed, '_') // // result: "get_user_by_i_d" // // # Complejidad // // - Tiempo: O(n) // - Memoria: O(n+e) donde e = número de mayúsculas a insertar func (t *ASCIIAttrs) Snakeize(list []byte, separator byte) []byte { // Calcular espacio extra necesario extra := 0 for i, ch := range list { if i > 0 && t.attrs[ch]&ASCIIAttrAlphaUpper != 0 { extra++ } } // Si no hay mayúsculas, solo convertir a lowercase if extra == 0 { t.Lowerize(list) return list } // Crear nuevo slice con espacio para separadores newLen := len(list) + extra result := make([]byte, newLen) target := 0 for i := 0; i < len(list); i++ { ch := list[i] // Insertar separador antes de mayúscula (excepto primera posición) if i > 0 && t.attrs[ch]&ASCIIAttrAlphaUpper != 0 { result[target] = separator target++ result[target] = ch + 32 // Convertir a lowercase } else if t.attrs[ch]&ASCIIAttrAlphaUpper != 0 { result[target] = ch + 32 // Primera letra: solo lowercase } else { result[target] = ch } target++ } return result } // ToSnake convierte a snake_case manejando inteligentemente CamelCase y separadores. // // # Comportamiento Avanzado // // Esta función maneja varios casos complejos: // // 1. CamelCase → snake_case: // - "HelloWorld" → "hello_world" // - Inserta '_' antes de mayúscula si la anterior era minúscula/dígito // // 2. Siglas (secuencias de mayúsculas): // - "HTMLParser" → "html_parser" (no "h_t_m_l_parser") // - Detecta cuando una mayúscula es seguida por minúscula // // 3. Normalización de separadores: // - Convierte '-', '.', ' ' al separador solicitado // - "hello-world" → "hello_world" (si separator='_') // - "hello.world" → "hello_world" // // # Algoritmo de Inserción Inteligente // // Inserta '_' antes de una mayúscula si: // - La anterior era minúscula/dígito: "aB" → "a_b" // - O forma parte de sigla antes de minúscula: "ABc" → "a_bc" // // # Ejemplo // // attrs := NewASCIIAttrs() // // // CamelCase básico // attrs.ToSnake([]byte("HelloWorld"), '_') // "hello_world" // // // Siglas // attrs.ToSnake([]byte("HTMLParser"), '_') // "html_parser" // attrs.ToSnake([]byte("parseHTMLDocument"), '_') // "parse_html_document" // // // Normalización de separadores // attrs.ToSnake([]byte("hello-world"), '_') // "hello_world" // attrs.ToSnake([]byte("user.service"), '_') // "user_service" // // // Mix complejo // attrs.ToSnake([]byte("getUserByID"), '_') // "get_user_by_id" // // # Performance // // - Pre-calcula tamaño final (evita reallocations) // - Zero-copy cuando no hay cambios // - Single-pass transformation // // # Complejidad // // - Tiempo: O(2n) - pre-scan + transform // - Memoria: O(n+s) donde s = separadores insertados func (t *ASCIIAttrs) ToSnake(list []byte, separator byte) []byte { if len(list) == 0 { return list } // PASO 1: Calcular longitud extra y detectar necesidad de normalización extraLen := 0 needsNormalization := false hasUpper := false for i := 0; i < len(list); i++ { ch := list[i] if t.attrs[ch]&ASCIIAttrAlphaUpper != 0 { hasUpper = true // Caso 1: "aB" → Insertar '_' entre minúscula/dígito y mayúscula if i > 0 && (t.attrs[list[i-1]]&(ASCIIAttrAlphaLower|ASCIIAttrDigit) != 0) { extraLen++ } // Caso 2: "ABc" → Insertar '_' en "A_Bc" (manejo de siglas) if i > 0 && i < len(list)-1 && (t.attrs[list[i-1]]&ASCIIAttrAlphaUpper != 0) && (t.attrs[list[i+1]]&ASCIIAttrAlphaLower != 0) { extraLen++ } } else { // Detectar separadores que necesitan normalización if (ch == '-' || ch == '.' || ch == ' ' || ch == '_') && ch != separator { needsNormalization = true } } } // Zero-copy optimization if extraLen == 0 && !hasUpper && !needsNormalization { return list } // PASO 2: Construcción del resultado result := make([]byte, len(list)+extraLen) w := 0 for i := 0; i < len(list); i++ { ch := list[i] if t.attrs[ch]&ASCIIAttrAlphaUpper != 0 { // Lógica de inserción de separador if i > 0 && (t.attrs[list[i-1]]&(ASCIIAttrAlphaLower|ASCIIAttrDigit) != 0) { result[w] = separator w++ } else if i > 0 && i < len(list)-1 && (t.attrs[list[i-1]]&ASCIIAttrAlphaUpper != 0) && (t.attrs[list[i+1]]&ASCIIAttrAlphaLower != 0) { result[w] = separator w++ } result[w] = ch + 32 // A minúscula } else { // Normalización de separadores if ch == '-' || ch == '.' || ch == ' ' || ch == '_' { result[w] = separator } else { result[w] = ch } } w++ } return result } // ScreamingSnakeize transforma a SCREAMING_SNAKE_CASE in-place (limitado al tamaño actual). // // # SCREAMING_SNAKE_CASE // // Es snake_case pero con todas las letras en mayúsculas: // - "hello_world" → "HELLO_WORLD" // - Comúnmente usado para constantes: MAX_SIZE, API_KEY // // # Limitación // // No puede insertar separadores si no hay espacio (in-place real). Para una versión // que inserta separadores, usar ToScreamingSnake. // // # Ejemplo // // attrs := NewASCIIAttrs() // // input := []byte("hello_world") // attrs.ScreamingSnakeize(input) // // input: "HELLO_WORLD" // // // Normaliza separadores conocidos // input2 := []byte("hello-world") // attrs.ScreamingSnakeize(input2) // // input2: "HELLO_WORLD" func (t *ASCIIAttrs) ScreamingSnakeize(list []byte) []byte { for i := 0; i < len(list); i++ { ch := list[i] // Convertir minúsculas a mayúsculas if t.attrs[ch]&ASCIIAttrAlphaLower != 0 { list[i] = ch - 32 } else if ch == '-' || ch == '.' || ch == ' ' { // Normalizar separadores conocidos a '_' list[i] = '_' } } return list } // ToScreamingSnake convierte a SCREAMING_SNAKE_CASE con manejo completo de CamelCase. // // Combina la funcionalidad de ToSnake y uppercase: // - Detecta CamelCase e inserta '_' // - Maneja siglas inteligentemente // - Normaliza separadores existentes // - Convierte todo a mayúsculas // // # Ejemplo // // attrs := NewASCIIAttrs() // // // CamelCase → SCREAMING_SNAKE // attrs.ToScreamingSnake([]byte("HelloWorld"), '_') // "HELLO_WORLD" // attrs.ToScreamingSnake([]byte("getUserByID"), '_') // "GET_USER_BY_ID" // // // Normalización // attrs.ToScreamingSnake([]byte("hello-world"), '_') // "HELLO_WORLD" // attrs.ToScreamingSnake([]byte("api_key"), '_') // "API_KEY" // // # Complejidad // // - Tiempo: O(2n) // - Memoria: O(n+s) func (t *ASCIIAttrs) ToScreamingSnake(list []byte, separator byte) []byte { if len(list) == 0 { return list } // Detectar necesidad de cambios extraLen := 0 needsUpper := false needsNormalization := false for i := 0; i < len(list); i++ { ch := list[i] // Detectar inserción de separadores if i > 0 && t.attrs[ch]&ASCIIAttrAlphaUpper != 0 { if t.attrs[list[i-1]]&(ASCIIAttrAlphaLower|ASCIIAttrDigit) != 0 { extraLen++ } else if i < len(list)-1 && t.attrs[list[i+1]]&ASCIIAttrAlphaLower != 0 { extraLen++ } } // Detectar necesidad de uppercase if t.attrs[ch]&ASCIIAttrAlphaLower != 0 { needsUpper = true } // Detectar necesidad de normalización if (ch == '-' || ch == '.' || ch == ' ') && ch != separator { needsNormalization = true } } // Zero-copy optimization if extraLen == 0 && !needsUpper && !needsNormalization { return list } // Construcción result := make([]byte, len(list)+extraLen) w := 0 for i := 0; i < len(list); i++ { ch := list[i] if t.attrs[ch]&ASCIIAttrAlphaUpper != 0 { // Insertar separador si es necesario if i > 0 && (t.attrs[list[i-1]]&(ASCIIAttrAlphaLower|ASCIIAttrDigit) != 0) { result[w] = separator w++ } else if i > 0 && i < len(list)-1 && (t.attrs[list[i-1]]&ASCIIAttrAlphaUpper != 0) && (t.attrs[list[i+1]]&ASCIIAttrAlphaLower != 0) { result[w] = separator w++ } result[w] = ch // Ya es mayúscula } else if t.attrs[ch]&ASCIIAttrAlphaLower != 0 { result[w] = ch - 32 // Convertir a mayúscula } else if ch == '-' || ch == '.' || ch == ' ' || ch == '_' { result[w] = separator // Normalizar separador } else { result[w] = ch } w++ } return result } // ════════════════════════════════════════════════════════════════════════════ // CONVERSIONES NUMÉRICAS // ════════════════════════════════════════════════════════════════════════════ // DigitToByte convierte un byte dígito decimal ('0'-'9') a su valor numérico (0-9). // // # Ejemplo // // attrs := NewASCIIAttrs() // attrs.DigitToByte('0') // 0 // attrs.DigitToByte('5') // 5 // attrs.DigitToByte('9') // 9 // attrs.DigitToByte('A') // 0 (no es dígito) // // # Complejidad: O(1) func (t *ASCIIAttrs) DigitToByte(chr byte) byte { if t.attrs[chr]&ASCIIAttrDigit != 0 { return chr - '0' } return 0 } // HexDigitToByte convierte un byte hexadecimal ('0'-'9','A'-'F','a'-'f') a su valor (0-15). // // # Ejemplo // // attrs := NewASCIIAttrs() // attrs.HexDigitToByte('0') // 0 // attrs.HexDigitToByte('9') // 9 // attrs.HexDigitToByte('A') // 10 // attrs.HexDigitToByte('F') // 15 // attrs.HexDigitToByte('a') // 10 // attrs.HexDigitToByte('f') // 15 // attrs.HexDigitToByte('G') // 0 (no es hex) // // # Complejidad: O(1) func (t *ASCIIAttrs) HexDigitToByte(chr byte) byte { att := t.attrs[chr] if att&ASCIIAttrDigitHex != 0 { if att&ASCIIAttrDigit != 0 { return chr - '0' } if att&ASCIIAttrAlphaLower != 0 { return chr - 'a' + 10 } return chr - 'A' + 10 } return 0 } // IsHex2Digit verifica si dos bytes consecutivos son dígitos hexadecimales válidos. // // # Uso Típico // // Verificar antes de decodificar URL encoding: "%20" → ' ' // // # Ejemplo // // attrs := NewASCIIAttrs() // attrs.IsHex2Digit('2', '0') // true ("%20") // attrs.IsHex2Digit('F', 'F') // true ("%FF") // attrs.IsHex2Digit('G', '0') // false (G no es hex) // // # Complejidad: O(1) func (t *ASCIIAttrs) IsHex2Digit(cha, chb byte) bool { return (t.attrs[cha]&ASCIIAttrDigitHex != 0) && (t.attrs[chb]&ASCIIAttrDigitHex != 0) } // Hex2DigitToByte combina dos dígitos hexadecimales en un byte. // // # IMPORTANTE // // No valida si los bytes son hex válidos. Usar IsHex2Digit primero si no estás seguro. // // # Ejemplo // // attrs := NewASCIIAttrs() // attrs.Hex2DigitToByte('2', '0') // 32 (espacio ' ') // attrs.Hex2DigitToByte('F', 'F') // 255 // attrs.Hex2DigitToByte('4', '1') // 65 ('A') // // # Uso en URL Decoding // // if attrs.IsHex2Digit(url[i+1], url[i+2]) { // decoded = attrs.Hex2DigitToByte(url[i+1], url[i+2]) // } // // # Complejidad: O(1) func (t *ASCIIAttrs) Hex2DigitToByte(cha, chb byte) byte { return t.HexDigitToByte(cha)<<4 | t.HexDigitToByte(chb) } // GetHex2DigitToByte combina dos dígitos hex en un byte Y valida. // // A diferencia de Hex2DigitToByte, esta función valida y retorna un bool indicando éxito. // // # Ejemplo // // attrs := NewASCIIAttrs() // // val, ok := attrs.GetHex2DigitToByte('2', '0') // if ok { // // val = 32 (espacio) // } // // val, ok = attrs.GetHex2DigitToByte('G', '0') // // ok = false, val = 0 // // # Complejidad: O(1) func (t *ASCIIAttrs) GetHex2DigitToByte(cha, chb byte) (byte, bool) { at, bt := t.attrs[cha], t.attrs[chb] // Validar que ambos sean hex if at&ASCIIAttrDigitHex == 0 || bt&ASCIIAttrDigitHex == 0 { return 0, false } // Convertir primer dígito var va byte if at&ASCIIAttrDigit != 0 { va = cha - '0' } else if at&ASCIIAttrAlphaLower != 0 { va = cha - 'a' + 10 } else { va = cha - 'A' + 10 } va <<= 4 // Convertir segundo dígito if bt&ASCIIAttrDigit != 0 { return va | (chb - '0'), true } if bt&ASCIIAttrAlphaLower != 0 { return va | (chb - 'a' + 10), true } return va | (chb - 'A' + 10), true } // ════════════════════════════════════════════════════════════════════════════ // PREDICADOS DE CLASIFICACIÓN // ════════════════════════════════════════════════════════════════════════════ // Todas estas funciones son O(1) - un lookup + AND bitwise. // IsControl verifica si el byte es un carácter de control ASCII (0-31 o 127). func (t *ASCIIAttrs) IsControl(a byte) bool { return t.attrs[a]&ASCIIAttrControl != 0 } // IsPrintable verifica si el byte es un carácter imprimible (32-126). func (t *ASCIIAttrs) IsPrintable(a byte) bool { return t.attrs[a]&ASCIIAttrPrintable != 0 } // IsAlpha verifica si el byte es una letra (A-Z o a-z). func (t *ASCIIAttrs) IsAlpha(a byte) bool { return t.attrs[a]&ASCIIAttrAlpha != 0 } // IsAlphaLower verifica si el byte es una letra minúscula (a-z). func (t *ASCIIAttrs) IsAlphaLower(a byte) bool { return t.attrs[a]&ASCIIAttrAlphaLower != 0 } // IsAlphaUpper verifica si el byte es una letra mayúscula (A-Z). func (t *ASCIIAttrs) IsAlphaUpper(a byte) bool { return t.attrs[a]&ASCIIAttrAlphaUpper != 0 } // IsDigit verifica si el byte es un dígito decimal ('0'-'9'). func (t *ASCIIAttrs) IsDigit(a byte) bool { return t.attrs[a]&ASCIIAttrDigit != 0 } // IsHex verifica si el byte es un dígito hexadecimal adicional (A-F, a-f). // No incluye 0-9. Para verificar cualquier dígito hex, usar IsHexDigit. func (t *ASCIIAttrs) IsHex(a byte) bool { return t.attrs[a]&ASCIIAttrHex != 0 } // IsAlphaNum verifica si el byte es alfanumérico (letra o dígito). func (t *ASCIIAttrs) IsAlphaNum(a byte) bool { return t.attrs[a]&(ASCIIAttrAlphaNum) != 0 } // IsAlphaNumLower verifica si el byte es alfanumérico (letra minúscula o dígito). func (t *ASCIIAttrs) IsAlphaNumLower(a byte) bool { return t.attrs[a]&(ASCIIAttrAlphaNumLower) != 0 } // IsAlphaNumUpper verifica si el byte es alfanumérico (letra mayúscula o dígito). func (t *ASCIIAttrs) IsAlphaNumUpper(a byte) bool { return t.attrs[a]&(ASCIIAttrAlphaNumUpper) != 0 } // IsHexDigit verifica si el byte es cualquier dígito hexadecimal (0-9, A-F, a-f). func (t *ASCIIAttrs) IsHexDigit(a byte) bool { return t.attrs[a]&ASCIIAttrDigitHex != 0 }