diff options
| author | alex <[email protected]> | 2026-07-28 17:37:06 +0200 |
|---|---|---|
| committer | alex <[email protected]> | 2026-07-28 17:37:06 +0200 |
| commit | c91c3afd8700138f8eb8a25a4ae422c2c9e46cea (patch) | |
| tree | fd9b5ea4f3b821581b2ceb4e7fa8afc1a84c0eeb /core/analyzer/analyzer.go | |
| parent | 5858d34b8ab573e18a4f6ddc5ee962b24ea269dc (diff) | |
| download | search-engine-c91c3afd8700138f8eb8a25a4ae422c2c9e46cea.tar.xz search-engine-c91c3afd8700138f8eb8a25a4ae422c2c9e46cea.zip | |
refactored entire code base to be in the package core so i can later implement interfaces to interact with the engine
Diffstat (limited to 'core/analyzer/analyzer.go')
| -rw-r--r-- | core/analyzer/analyzer.go | 38 |
1 files changed, 38 insertions, 0 deletions
diff --git a/core/analyzer/analyzer.go b/core/analyzer/analyzer.go new file mode 100644 index 0000000..6a02541 --- /dev/null +++ b/core/analyzer/analyzer.go @@ -0,0 +1,38 @@ +package analyzer + +import ( + "strings" + "unicode" +) + +var stopWords = map[string]struct{}{ + "the": {}, "a": {}, "an": {}, "and": {}, "or": {}, "but": {}, + "is": {}, "are": {}, "was": {}, "were": {}, "in": {}, "on": {}, + "at": {}, "to": {}, "from": {}, "by": {}, "of": {}, +} + +func ProcessText(text string) []string { + text = strings.ToLower(text) + text = SanitizeText(text) + tokens := strings.Fields(text) + var cleanTokens []string + for _, token := range tokens { + _, exists := stopWords[token] + if !exists { + cleanTokens = append(cleanTokens, token) + } + } + + return cleanTokens +} + +func SanitizeText(s string) string { + var builder strings.Builder + + for _, char := range s { + if unicode.IsLetter(char) || unicode.IsNumber(char) || unicode.IsSpace(char) { + builder.WriteRune(char) + } + } + return builder.String() +} |
