You can not select more than 25 topics
Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
240 lines
5.0 KiB
240 lines
5.0 KiB
// ------------------------------------------------------------------------
|
|
// Project active2
|
|
// Active Thing (activething.com) git.activething.com/go
|
|
//
|
|
// File name atokenizer.go
|
|
// Created by DEV
|
|
// Modified 24/04/2024
|
|
//
|
|
// Copyright 2024 activething.com
|
|
// ------------------------------------------------------------------------
|
|
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
// you may not use this file except in compliance with the License.
|
|
// You may obtain a copy of the License at
|
|
//
|
|
// http://www.apache.org/licenses/LICENSE-2.0
|
|
//
|
|
// Unless required by applicable law or agreed to in writing, software
|
|
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
// See the License for the specific language governing permissions and
|
|
// limitations under the License.
|
|
// ------------------------------------------------------------------------
|
|
|
|
package tokens
|
|
|
|
import "bins/matchs"
|
|
|
|
const (
|
|
|
|
topicMinLen int = 1
|
|
topicMaxLen int = 255
|
|
topicMaxTokens int = 32
|
|
topicMinTokens int = 1
|
|
|
|
tokenMinLen int = 1
|
|
tokenMaxLen int = 256
|
|
|
|
|
|
)
|
|
|
|
|
|
type (
|
|
|
|
// ATokenizer
|
|
// Struct type
|
|
// Implements -> Tokenizer
|
|
//
|
|
ATokenizer struct {
|
|
aTokenizerChars
|
|
|
|
MinLenToken int
|
|
MaxLenToken int
|
|
MinLenTopic int
|
|
MaxLenTopic int
|
|
|
|
MinTokens int
|
|
MaxTokens int
|
|
|
|
MatchBuilder matchs.Builder
|
|
}
|
|
|
|
|
|
ATokenizerFnc func (tokenizer *ATokenizer)
|
|
|
|
)
|
|
|
|
// NewATokenizer
|
|
// Function constructor -> ATokenizer
|
|
//
|
|
//
|
|
func NewATokenizer(options ...ATokenizerFnc) *ATokenizer {
|
|
tk := &ATokenizer{
|
|
aTokenizerChars : *newATokenizerChars(),
|
|
MinLenToken : tokenMinLen,
|
|
MaxLenToken : tokenMaxLen,
|
|
MinLenTopic : topicMinLen,
|
|
MaxLenTopic : topicMaxLen,
|
|
MinTokens : topicMinTokens,
|
|
MaxTokens : topicMaxTokens,
|
|
MatchBuilder : matchs.NopBuilder{},
|
|
}
|
|
for _,o := range options {
|
|
o(tk)
|
|
}
|
|
if tk.MaxTokens > topicMaxTokens {
|
|
tk.MaxTokens = topicMaxTokens
|
|
}
|
|
return tk
|
|
}
|
|
|
|
|
|
func (t ATokenizer) NewTopic (topic []byte, allow Kind) *Topic {
|
|
return NewTopic(t, topic, allow) }
|
|
|
|
|
|
func (t ATokenizer) Tokenize(topic []byte, allow Kind) (Tokens, Kind){
|
|
ln := len(topic)
|
|
if ln < t.MinLenTopic || ln > t.MaxLenTopic {
|
|
return []Token{*(NewInvalidToken(G.ErrInvalidTopicLen,topic)) }, TokenKindInvalid
|
|
}
|
|
|
|
tl := [topicMaxTokens]Token{}
|
|
tt := TokenKindNone
|
|
m,c,x := 0,0,0
|
|
|
|
for i:=0; i<ln; i++ {
|
|
switch topic[i] {
|
|
case t.TokenMatchIni: m++
|
|
case t.TokenMatchEnd : m--
|
|
case t.TokenSep:
|
|
if m != 0 {
|
|
continue }
|
|
if c >= t.MaxTokens {
|
|
tl[c] = *(NewInvalidToken(G.ErrInvalidTopicLen,topic[x:]))
|
|
return tl[:c+1], tt | TokenKindInvalid
|
|
}
|
|
tl[c] = t.NewToken(topic[x:i])
|
|
tk := tl[c].Kind()
|
|
if tk == TokenKindRelative {
|
|
tl[c]= NewInvalidToken(G.ErrInvalidToken,topic[x:i])
|
|
} else if tk & allow == 0 {
|
|
tt |= TokenKindInvalid
|
|
}
|
|
tt |= tl[c].Kind()
|
|
x = i + 1
|
|
c++
|
|
}
|
|
}
|
|
|
|
if m != 0 {
|
|
tl[c] = *(NewInvalidToken(G.ErrInvalidTopic,topic[x:]))
|
|
} else {
|
|
tl[c] = t.NewToken(topic[x:])
|
|
}
|
|
|
|
return tl[:c+1], tt | tl[c].Kind()
|
|
|
|
}
|
|
|
|
|
|
// Split
|
|
// Function member -> ATokenizer
|
|
//
|
|
//
|
|
func (t ATokenizer) Split (topic []byte, allow Kind)([][]byte, Kind){
|
|
var ms, ps,x int
|
|
var tk, pk Kind
|
|
ln := len(topic)
|
|
ar := [topicMaxTokens][]byte{}
|
|
|
|
for i:=0; i< ln; i++ {
|
|
switch topic[i] {
|
|
case t.TokenMatchIni: ms++
|
|
case t.TokenMatchEnd : ms--
|
|
case t.TokenSep:
|
|
if ms == 0 {
|
|
ar[ps] = topic[x:i]
|
|
if pk = t.Kind(ar[ps]); pk & allow == 0 {
|
|
return nil, TokenKindInvalid
|
|
}
|
|
tk|= pk
|
|
x = i + 1
|
|
ps++
|
|
}
|
|
}
|
|
}
|
|
ar[ps] = topic[x:]
|
|
if pk = t.Kind(ar[ps]); pk& allow == 0 {
|
|
return nil, TokenKindInvalid
|
|
}
|
|
tk |= pk
|
|
|
|
return ar[:ps+1],tk
|
|
}
|
|
|
|
|
|
|
|
|
|
// NewToken
|
|
// Function member -> ATokenizer
|
|
//
|
|
//
|
|
func (t ATokenizer) NewToken(Token []byte) Token {
|
|
ln := len(Token)
|
|
if ln < t.MinLenToken || ln > t.MaxLenToken {
|
|
return InvalidToken{ error: nil, source: Token } }
|
|
|
|
if ln == 1 {
|
|
switch Token[0] {
|
|
case t.TokenWildcard : return WildcardToken{}
|
|
case t.TokenRelative : return RelativeToken{}
|
|
case t.TokenMatchIni,
|
|
t.TokenMatchEnd : return *(NewInvalidToken(G.ErrInvalidTopicLen,Token)) }
|
|
}
|
|
|
|
if Token[0] != t.TokenMatchIni {
|
|
return LiteralToken(Token) }
|
|
|
|
if Token[ln-1] == t.TokenMatchEnd && ln > t.MinLenToken+ 2 {
|
|
m,e := t.MatchBuilder.Build(Token)
|
|
if e == nil {
|
|
return *(NewMatcherToken(m.Match,Token)) }
|
|
return *(NewInvalidToken(e,Token))
|
|
}
|
|
|
|
return *(NewInvalidToken(matchs.G.ErrInvalidMatcher,Token))
|
|
}
|
|
|
|
|
|
|
|
|
|
// Kind
|
|
// Function member -> ATokenizer
|
|
//
|
|
func (t ATokenizer) Kind (Token []byte)(Kind){
|
|
ln := len(Token)
|
|
if ln < t.MinLenToken || ln > t.MaxLenToken {
|
|
return TokenKindInvalid
|
|
}
|
|
|
|
if ln == 1 {
|
|
switch Token[0] {
|
|
case t.TokenWildcard : return TokenKindWildcard
|
|
case t.TokenRelative : return TokenKindRelative
|
|
case t.TokenMatchIni,
|
|
t.TokenMatchEnd : return TokenKindInvalid
|
|
}
|
|
}
|
|
|
|
if Token[0] == t.TokenMatchIni {
|
|
if Token[ln-1] == t.TokenMatchEnd {
|
|
return TokenKindMatcher }
|
|
return TokenKindInvalid
|
|
}
|
|
return TokenKindLiteral
|
|
}
|
|
|
|
|
|
|