From 5e3b3e051b70ff801d17fc193a56f3c1a1e24f29 Mon Sep 17 00:00:00 2001 From: GordonT0110 Date: Fri, 27 Feb 2026 20:35:36 +1100 Subject: [PATCH] feat:avoid on the fly regex compile and move them during initializaiton --- pkg/channels/telegram.go | 38 +++++++++++++++++++------------ pkg/providers/error_classifier.go | 12 +++++----- pkg/skills/loader.go | 12 ++++++---- pkg/tools/shell.go | 5 ++-- pkg/tools/web.go | 36 ++++++++++++++++------------- 5 files changed, 60 insertions(+), 43 deletions(-) diff --git a/pkg/channels/telegram.go b/pkg/channels/telegram.go index 524494849..2638c77ac 100644 --- a/pkg/channels/telegram.go +++ b/pkg/channels/telegram.go @@ -23,6 +23,19 @@ import ( "github.com/sipeed/picoclaw/pkg/voice" ) +var ( + reHeading = regexp.MustCompile(`^#{1,6}\s+(.+)$`) + reBlockquote = regexp.MustCompile(`^>\s*(.*)$`) + reLink = regexp.MustCompile(`\[([^\]]+)\]\(([^)]+)\)`) + reBold1 = regexp.MustCompile(`\*\*(.+?)\*\*`) + reBold2 = regexp.MustCompile(`__(.+?)__`) + reItalic = regexp.MustCompile(`_([^_]+)_`) + reStrike = regexp.MustCompile(`~~(.+?)~~`) + reBullet = regexp.MustCompile(`^[-*]\s+`) + reCodeBlock = regexp.MustCompile("```[\\w]*\\n?([\\s\\S]*?)```") + reInlineCode = regexp.MustCompile("`([^`]+)`") +) + type TelegramChannel struct { *BaseChannel bot *telego.Bot @@ -431,19 +444,18 @@ func markdownToTelegramHTML(text string) string { inlineCodes := extractInlineCodes(text) text = inlineCodes.text - text = regexp.MustCompile(`^#{1,6}\s+(.+)$`).ReplaceAllString(text, "$1") + text = reHeading.ReplaceAllString(text, "$1") - text = regexp.MustCompile(`^>\s*(.*)$`).ReplaceAllString(text, "$1") + text = reBlockquote.ReplaceAllString(text, "$1") text = escapeHTML(text) - text = regexp.MustCompile(`\[([^\]]+)\]\(([^)]+)\)`).ReplaceAllString(text, `$1`) + text = reLink.ReplaceAllString(text, `$1`) - text = regexp.MustCompile(`\*\*(.+?)\*\*`).ReplaceAllString(text, "$1") + text = reBold1.ReplaceAllString(text, "$1") - text = regexp.MustCompile(`__(.+?)__`).ReplaceAllString(text, "$1") + text = reBold2.ReplaceAllString(text, "$1") - reItalic := regexp.MustCompile(`_([^_]+)_`) text = reItalic.ReplaceAllStringFunc(text, func(s string) string { match := reItalic.FindStringSubmatch(s) if len(match) < 2 { @@ -452,9 +464,9 @@ func markdownToTelegramHTML(text string) string { return "" + match[1] + "" }) - text = regexp.MustCompile(`~~(.+?)~~`).ReplaceAllString(text, "$1") + text = reStrike.ReplaceAllString(text, "$1") - text = regexp.MustCompile(`^[-*]\s+`).ReplaceAllString(text, "• ") + text = reBullet.ReplaceAllString(text, "• ") for i, code := range inlineCodes.codes { escaped := escapeHTML(code) @@ -479,8 +491,7 @@ type codeBlockMatch struct { } func extractCodeBlocks(text string) codeBlockMatch { - re := regexp.MustCompile("```[\\w]*\\n?([\\s\\S]*?)```") - matches := re.FindAllStringSubmatch(text, -1) + matches := reCodeBlock.FindAllStringSubmatch(text, -1) codes := make([]string, 0, len(matches)) for _, match := range matches { @@ -488,7 +499,7 @@ func extractCodeBlocks(text string) codeBlockMatch { } i := 0 - text = re.ReplaceAllStringFunc(text, func(m string) string { + text = reCodeBlock.ReplaceAllStringFunc(text, func(m string) string { placeholder := fmt.Sprintf("\x00CB%d\x00", i) i++ return placeholder @@ -503,8 +514,7 @@ type inlineCodeMatch struct { } func extractInlineCodes(text string) inlineCodeMatch { - re := regexp.MustCompile("`([^`]+)`") - matches := re.FindAllStringSubmatch(text, -1) + matches := reInlineCode.FindAllStringSubmatch(text, -1) codes := make([]string, 0, len(matches)) for _, match := range matches { @@ -512,7 +522,7 @@ func extractInlineCodes(text string) inlineCodeMatch { } i := 0 - text = re.ReplaceAllStringFunc(text, func(m string) string { + text = reInlineCode.ReplaceAllStringFunc(text, func(m string) string { placeholder := fmt.Sprintf("\x00IC%d\x00", i) i++ return placeholder diff --git a/pkg/providers/error_classifier.go b/pkg/providers/error_classifier.go index a0f003006..69b0a29e7 100644 --- a/pkg/providers/error_classifier.go +++ b/pkg/providers/error_classifier.go @@ -15,6 +15,11 @@ type errorPattern struct { func substr(s string) errorPattern { return errorPattern{substring: s} } func rxp(r string) errorPattern { return errorPattern{regex: regexp.MustCompile("(?i)" + r)} } +var ( + reHTTPStatus = regexp.MustCompile(`status[:\s]+(\d{3})`) + reHTTPStatusLine = regexp.MustCompile(`HTTP[/\s]+\d*\.?\d*\s+(\d{3})`) +) + // Error patterns organized by FailoverReason, matching OpenClaw production (~40 patterns). var ( rateLimitPatterns = []errorPattern{ @@ -201,12 +206,7 @@ func classifyByMessage(msg string) FailoverReason { // Looks for patterns like "status: 429", "status 429", "HTTP 429", or standalone "429". func extractHTTPStatus(msg string) int { // Common patterns in Go HTTP error messages - patterns := []*regexp.Regexp{ - regexp.MustCompile(`status[:\s]+(\d{3})`), - regexp.MustCompile(`HTTP[/\s]+\d*\.?\d*\s+(\d{3})`), - } - - for _, p := range patterns { + for _, p := range []*regexp.Regexp{reHTTPStatus, reHTTPStatusLine} { if m := p.FindStringSubmatch(msg); len(m) > 1 { return parseDigits(m[1]) } diff --git a/pkg/skills/loader.go b/pkg/skills/loader.go index 5749d8983..2d3509ad8 100644 --- a/pkg/skills/loader.go +++ b/pkg/skills/loader.go @@ -13,7 +13,11 @@ import ( "github.com/sipeed/picoclaw/pkg/logger" ) -var namePattern = regexp.MustCompile(`^[a-zA-Z0-9]+(-[a-zA-Z0-9]+)*$`) +var ( + namePattern = regexp.MustCompile(`^[a-zA-Z0-9]+(-[a-zA-Z0-9]+)*$`) + reFrontmatterExtract = regexp.MustCompile(`(?s)^---(?:\r\n|\n|\r)(.*?)(?:\r\n|\n|\r)---`) + reFrontmatterStrip = regexp.MustCompile(`(?s)^---(?:\r\n|\n|\r)(.*?)(?:\r\n|\n|\r)---(?:\r\n|\n|\r)*`) +) const ( MaxNameLength = 64 @@ -259,8 +263,7 @@ func (sl *SkillsLoader) extractFrontmatter(content string) string { // Support \n (Unix), \r\n (Windows), and \r (classic Mac) line endings for frontmatter blocks // (?s) enables DOTALL so . matches newlines; // ^--- at start, then ... --- at start of line, honoring all three line ending types - re := regexp.MustCompile(`(?s)^---(?:\r\n|\n|\r)(.*?)(?:\r\n|\n|\r)---`) - match := re.FindStringSubmatch(content) + match := reFrontmatterExtract.FindStringSubmatch(content) if len(match) > 1 { return match[1] } @@ -272,8 +275,7 @@ func (sl *SkillsLoader) stripFrontmatter(content string) string { // (?s) enables DOTALL so . matches newlines; // ^--- at start, then ... --- at start of line, honoring all three line ending types // Match zero or more trailing line endings after closing --- (handles both with and without blank lines) - re := regexp.MustCompile(`(?s)^---(?:\r\n|\n|\r)(.*?)(?:\r\n|\n|\r)---(?:\r\n|\n|\r)*`) - return re.ReplaceAllString(content, "") + return reFrontmatterStrip.ReplaceAllString(content, "") } func escapeXML(s string) string { diff --git a/pkg/tools/shell.go b/pkg/tools/shell.go index ad1664b5b..0dc5e19a4 100644 --- a/pkg/tools/shell.go +++ b/pkg/tools/shell.go @@ -24,6 +24,8 @@ type ExecTool struct { restrictToWorkspace bool } +var rePathPattern = regexp.MustCompile(`[A-Za-z]:\\[^\\\"']+|/[^\s\"']+`) + var defaultDenyPatterns = []*regexp.Regexp{ regexp.MustCompile(`\brm\s+-[rf]{1,2}\b`), regexp.MustCompile(`\bdel\s+/[fq]\b`), @@ -288,8 +290,7 @@ func (t *ExecTool) guardCommand(command, cwd string) string { return "" } - pathPattern := regexp.MustCompile(`[A-Za-z]:\\[^\\\"']+|/[^\s\"']+`) - matches := pathPattern.FindAllString(cmd, -1) + matches := rePathPattern.FindAllString(cmd, -1) for _, raw := range matches { p, err := filepath.Abs(raw) diff --git a/pkg/tools/web.go b/pkg/tools/web.go index 44df28215..ccfdad963 100644 --- a/pkg/tools/web.go +++ b/pkg/tools/web.go @@ -13,6 +13,18 @@ import ( "time" ) +var ( + reDDGLink = regexp.MustCompile(`]*class="[^"]*result__a[^"]*"[^>]*href="([^"]+)"[^>]*>([\s\S]*?)`) + reDDGSnippet = regexp.MustCompile(`([\s\S]*?)`) + reStripTags = regexp.MustCompile(`<[^>]+>`) + + reExtractScript = regexp.MustCompile(``) + reExtractStyle = regexp.MustCompile(``) + reExtractTags = regexp.MustCompile(`<[^>]+>`) + reExtractSpaces = regexp.MustCompile(`[^\S\n]+`) + reExtractNewlines = regexp.MustCompile(`\n{3,}`) +) + const ( userAgent = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36" ) @@ -251,8 +263,7 @@ func (p *DuckDuckGoSearchProvider) extractResults(html string, count int, query // Try finding the result links directly first, as they are the most critical // Pattern: Title // The previous regex was a bit strict. Let's make it more flexible for attributes order/content - reLink := regexp.MustCompile(`]*class="[^"]*result__a[^"]*"[^>]*href="([^"]+)"[^>]*>([\s\S]*?)`) - matches := reLink.FindAllStringSubmatch(html, count+5) + matches := reDDGLink.FindAllStringSubmatch(html, count+5) if len(matches) == 0 { return fmt.Sprintf("No results found or extraction failed. Query: %s", query), nil @@ -269,8 +280,7 @@ func (p *DuckDuckGoSearchProvider) extractResults(html string, count int, query // A better regex approach: iterate through text and find matches in order // But for now, let's grab all snippets too - reSnippet := regexp.MustCompile(`([\s\S]*?)`) - snippetMatches := reSnippet.FindAllStringSubmatch(html, count+5) + snippetMatches := reDDGSnippet.FindAllStringSubmatch(html, count+5) maxItems := min(len(matches), count) @@ -305,8 +315,7 @@ func (p *DuckDuckGoSearchProvider) extractResults(html string, count int, query } func stripTags(content string) string { - re := regexp.MustCompile(`<[^>]+>`) - return re.ReplaceAllString(content, "") + return reStripTags.ReplaceAllString(content, "") } type PerplexitySearchProvider struct { @@ -654,19 +663,14 @@ func (t *WebFetchTool) Execute(ctx context.Context, args map[string]any) *ToolRe } func (t *WebFetchTool) extractText(htmlContent string) string { - re := regexp.MustCompile(``) - result := re.ReplaceAllLiteralString(htmlContent, "") - re = regexp.MustCompile(``) - result = re.ReplaceAllLiteralString(result, "") - re = regexp.MustCompile(`<[^>]+>`) - result = re.ReplaceAllLiteralString(result, "") + result := reExtractScript.ReplaceAllLiteralString(htmlContent, "") + result = reExtractStyle.ReplaceAllLiteralString(result, "") + result = reExtractTags.ReplaceAllLiteralString(result, "") result = strings.TrimSpace(result) - re = regexp.MustCompile(`[^\S\n]+`) - result = re.ReplaceAllString(result, " ") - re = regexp.MustCompile(`\n{3,}`) - result = re.ReplaceAllString(result, "\n\n") + result = reExtractSpaces.ReplaceAllString(result, " ") + result = reExtractNewlines.ReplaceAllString(result, "\n\n") lines := strings.Split(result, "\n") var cleanLines []string