diff --git a/pkg/channels/discord.go b/pkg/channels/discord.go index ea62217d3..ae3a2d90b 100644 --- a/pkg/channels/discord.go +++ b/pkg/channels/discord.go @@ -123,7 +123,7 @@ func (c *DiscordChannel) Send(ctx context.Context, msg bus.OutboundMessage) erro limit := c.config.MaxMessageLength if limit <= 0 { - limit = 1900 + limit = 1900 // Discord limit is 2000 chars } chunks := utils.SplitMessage(msg.Content, limit) diff --git a/pkg/channels/line.go b/pkg/channels/line.go index 0473c1b2c..e570b04af 100644 --- a/pkg/channels/line.go +++ b/pkg/channels/line.go @@ -512,7 +512,7 @@ func (c *LINEChannel) Send(ctx context.Context, msg bus.OutboundMessage) error { limit := c.config.MaxMessageLength if limit <= 0 { - limit = 4000 + limit = 4500 // LINE individual message bubbles are capped at 5,000 chars } chunks := utils.SplitMessage(msg.Content, limit) @@ -525,7 +525,7 @@ func (c *LINEChannel) Send(ctx context.Context, msg bus.OutboundMessage) error { if i == 0 && hasReplyToken { if err := c.sendReply(ctx, tokenEntry.token, chunk, currentQuoteToken); err == nil { - logger.DebugCF("line", "Message chunk sent via Reply API", map[string]interface{}{ + logger.DebugCF("line", "Message chunk sent via Reply API", map[string]any{ "chat_id": msg.ChatID, "quoted": currentQuoteToken != "", "chunk": i + 1, diff --git a/pkg/channels/slack.go b/pkg/channels/slack.go index 0ac0885d0..82eb78e10 100644 --- a/pkg/channels/slack.go +++ b/pkg/channels/slack.go @@ -121,7 +121,7 @@ func (c *SlackChannel) Send(ctx context.Context, msg bus.OutboundMessage) error limit := c.config.MaxMessageLength if limit <= 0 { - limit = 3000 + limit = 3500 // Slack has a 4,000 char limit (individual blocks may be 3k) } chunks := utils.SplitMessage(msg.Content, limit) diff --git a/pkg/channels/telegram.go b/pkg/channels/telegram.go index 314ca07cc..bfb78a7f6 100644 --- a/pkg/channels/telegram.go +++ b/pkg/channels/telegram.go @@ -170,10 +170,11 @@ func (c *TelegramChannel) Send(ctx context.Context, msg bus.OutboundMessage) err } // Telegram HTML tags (like
) can expand content significantly.
- // We use a safe headroom for the markdown-based split.
+ // We use a safe headroom for the markdown-based split, but never exceed
+ // the configured limit.
effectiveMarkdownLimit := limit - 500
- if effectiveMarkdownLimit < 500 {
- effectiveMarkdownLimit = 500
+ if effectiveMarkdownLimit < 1 {
+ effectiveMarkdownLimit = 1
}
chunks := utils.SplitMessage(msg.Content, effectiveMarkdownLimit)
@@ -199,8 +200,12 @@ func (c *TelegramChannel) Send(ctx context.Context, msg bus.OutboundMessage) err
tgMsg.ParseMode = telego.ModeHTML
if _, err = c.bot.SendMessage(ctx, tgMsg); err != nil {
- logger.ErrorCF("telegram", "HTML parse failed, falling back to plain text", map[string]any{
- "error": err.Error(),
+ logger.ErrorCF("telegram", "failed to send message in HTML mode, falling back to plain text", map[string]any{
+ "error": err.Error(),
+ "chat_id": msg.ChatID,
+ "chunk_index": i,
+ "chunk_total": len(chunks),
+ "chunk_length": len(chunk),
})
tgMsg.ParseMode = ""
tgMsg.Text = chunk // Use raw chunk if HTML fails
diff --git a/pkg/utils/message.go b/pkg/utils/message.go
index 45195fdd0..0c91d99f9 100644
--- a/pkg/utils/message.go
+++ b/pkg/utils/message.go
@@ -28,10 +28,12 @@ func SplitMessage(content string, maxLen int) []string {
}
runes := []rune(content)
+ startIndex := 0
- for len(runes) > 0 {
- if len(runes) <= maxLen {
- messages = append(messages, string(runes))
+ for startIndex < len(runes) {
+ remainingRunes := runes[startIndex:]
+ if len(remainingRunes) <= maxLen {
+ messages = append(messages, string(remainingRunes))
break
}
@@ -41,113 +43,124 @@ func SplitMessage(content string, maxLen int) []string {
effectiveLimit = maxLen / 2
}
- // Find natural split point within the effective limit
- // We pass the full slice and a limit so findLastSentenceBoundaryRunes can look ahead.
- msgEnd := findLastSentenceBoundaryRunes(runes, effectiveLimit, 300)
- if msgEnd <= 0 {
- msgEnd = findLastNewlineRunes(runes[:effectiveLimit], 200)
+ // Find natural split point within the effective limit from the current startIndex
+ // We pass the full slice so findLastSentenceBoundaryRunes can look ahead past the window
+ msgEndOffset := findLastSentenceBoundaryRunes(runes, startIndex+effectiveLimit, 300)
+ if msgEndOffset <= startIndex {
+ msgEndOffset = findLastNewlineRunes(remainingRunes[:effectiveLimit], 200)
+ if msgEndOffset >= 0 {
+ msgEndOffset += startIndex
+ }
}
- if msgEnd <= 0 {
- msgEnd = findLastSpaceRunes(runes[:effectiveLimit], 100)
+ if msgEndOffset <= startIndex {
+ msgEndOffset = findLastSpaceRunes(remainingRunes[:effectiveLimit], 100)
+ if msgEndOffset >= 0 {
+ msgEndOffset += startIndex
+ }
}
- if msgEnd <= 0 {
- msgEnd = effectiveLimit
+ if msgEndOffset <= startIndex {
+ msgEndOffset = startIndex + effectiveLimit
}
// Check if this would end with an incomplete code block
- candidate := runes[:msgEnd]
- unclosedIdx := findLastUnclosedCodeBlockRunes(candidate)
+ candidateRunes := runes[startIndex:msgEndOffset]
+ unclosedIdx := findLastUnclosedCodeBlockRunes(candidateRunes)
if unclosedIdx >= 0 {
- // Message would end with incomplete code block
+ // Absolute index of the unclosed fence
+ absUnclosedIdx := startIndex + unclosedIdx
+
// Try to extend up to maxLen to include the closing ```
- if len(runes) > msgEnd {
- closingIdx := findNextClosingCodeBlockRunes(runes, msgEnd)
- if closingIdx > 0 && closingIdx <= maxLen {
- // Extend to include the closing ```
- msgEnd = closingIdx
+ closingIdx := findNextClosingCodeBlockRunes(runes, msgEndOffset)
+ if closingIdx > 0 && closingIdx <= startIndex+maxLen {
+ msgEndOffset = closingIdx
+ } else {
+ // Find first newline after opening fence to extract header
+ headerEnd := -1
+ for i := absUnclosedIdx; i < len(runes); i++ {
+ if runes[i] == '\n' {
+ headerEnd = i
+ break
+ }
+ }
+ if headerEnd == -1 {
+ headerEnd = absUnclosedIdx + 3
} else {
- // Code block is too long to fit in one chunk or missing closing fence.
- // Try to split inside by injecting closing and reopening fences.
-
- // Find the header end (first newline after the opening ```)
- headerEnd := -1
- for i := unclosedIdx; i < len(runes); i++ {
- if runes[i] == '\n' {
- headerEnd = i
- break
+ headerEnd++ // include newline
+ }
+ header := strings.TrimSpace(string(runes[absUnclosedIdx:headerEnd]))
+
+ if msgEndOffset > headerEnd+20 {
+ innerLimit := maxLen - 5
+ betterEnd := findLastSentenceBoundaryRunes(runes, startIndex+innerLimit, 300)
+ if betterEnd <= headerEnd {
+ betterEnd = findLastNewlineRunes(runes[startIndex:startIndex+innerLimit], 200)
+ if betterEnd >= 0 {
+ betterEnd += startIndex
}
}
-
- if headerEnd == -1 {
- headerEnd = unclosedIdx + 3
+
+ if betterEnd > headerEnd {
+ msgEndOffset = betterEnd
} else {
- // include newline
- headerEnd++
+ msgEndOffset = startIndex + innerLimit
}
- header := strings.TrimSpace(string(runes[unclosedIdx:headerEnd]))
+ chunkStr := string(runes[startIndex:msgEndOffset])
+ messages = append(messages, strings.TrimRight(chunkStr, " \t\n\r")+"\n```")
- // If we have a reasonable amount of content after the header, split inside
- if msgEnd > headerEnd+20 {
- // Find a better split point closer to maxLen
- innerLimit := maxLen - 5 // Leave room for "\n```"
- betterEnd := findLastSentenceBoundaryRunes(runes, innerLimit, 300)
- if betterEnd <= headerEnd {
- betterEnd = findLastNewlineRunes(runes[:innerLimit], 200)
- }
-
- if betterEnd > headerEnd {
- msgEnd = betterEnd
- } else {
- msgEnd = innerLimit
- }
-
- chunkStr := string(runes[:msgEnd])
- messages = append(messages, strings.TrimRight(chunkStr, " \t\n\r")+"\n```")
-
- nextChunkStart := string(runes[msgEnd:])
- content = strings.TrimSpace(header + "\n" + nextChunkStart)
- runes = []rune(content)
- continue
- }
+ // Move startIndex to msgEndOffset but "inject" the header for the next iteration.
+ // We prepend the header and a newline.
+ injectedHeader := header + "\n"
+ nextRunes := append([]rune(injectedHeader), runes[msgEndOffset:]...)
+ runes = append(runes[:0], nextRunes...) // Reuse capacity
+ startIndex = 0
+ continue
+ }
- // Otherwise, try to split before the code block starts
- newEnd := findLastSentenceBoundaryRunes(runes, unclosedIdx, 300)
- if newEnd <= 0 {
- newEnd = findLastNewlineRunes(runes[:unclosedIdx], 200)
+ // Try to split before the code block
+ newEnd := findLastSentenceBoundaryRunes(runes, absUnclosedIdx, 300)
+ if newEnd <= startIndex {
+ newEnd = findLastNewlineRunes(runes[startIndex:absUnclosedIdx], 200)
+ if newEnd >= 0 {
+ newEnd += startIndex
}
- if newEnd <= 0 {
- newEnd = findLastSpaceRunes(runes[:unclosedIdx], 100)
- }
- if newEnd > 0 {
- msgEnd = newEnd
- } else {
- // If we can't split before, we MUST split inside (last resort)
- if unclosedIdx > 20 {
- msgEnd = unclosedIdx
- } else {
- msgEnd = maxLen - 5
- chunkStr := string(runes[:msgEnd])
- messages = append(messages, strings.TrimRight(chunkStr, " \t\n\r")+"\n```")
-
- nextChunkStart := string(runes[msgEnd:])
- content = strings.TrimSpace(header + "\n" + nextChunkStart)
- runes = []rune(content)
- continue
- }
+ }
+ if newEnd <= startIndex {
+ newEnd = findLastSpaceRunes(runes[startIndex:absUnclosedIdx], 100)
+ if newEnd >= 0 {
+ newEnd += startIndex
}
}
+
+ if newEnd > startIndex {
+ msgEndOffset = newEnd
+ } else {
+ // Hard split inside (last resort)
+ msgEndOffset = startIndex + maxLen - 5
+ chunkStr := string(runes[startIndex:msgEndOffset])
+ messages = append(messages, strings.TrimRight(chunkStr, " \t\n\r")+"\n```")
+
+ injectedHeader := header + "\n"
+ nextRunes := append([]rune(injectedHeader), runes[msgEndOffset:]...)
+ runes = append(runes[:0], nextRunes...)
+ startIndex = 0
+ continue
+ }
}
}
- if msgEnd <= 0 {
- msgEnd = effectiveLimit
+ if msgEndOffset <= startIndex {
+ msgEndOffset = startIndex + effectiveLimit
}
- messages = append(messages, string(runes[:msgEnd]))
- nextContent := strings.TrimSpace(string(runes[msgEnd:]))
- runes = []rune(nextContent)
+ messages = append(messages, string(runes[startIndex:msgEndOffset]))
+
+ // Advance startIndex and skip leading whitespace for next chunk
+ startIndex = msgEndOffset
+ for startIndex < len(runes) && (runes[startIndex] == ' ' || runes[startIndex] == '\n' || runes[startIndex] == '\t' || runes[startIndex] == '\r') {
+ startIndex++
+ }
}
return messages
diff --git a/pkg/utils/message_test.go b/pkg/utils/message_test.go
index 686d92d03..911fcbc51 100644
--- a/pkg/utils/message_test.go
+++ b/pkg/utils/message_test.go
@@ -99,6 +99,22 @@ func TestSplitMessage(t *testing.T) {
}
},
},
+ {
+ name: "Prefer sentence boundary",
+ // Content is: 1700 'a's, then ". ", then 500 'b's.
+ // Effective limit with maxLen=2000 is 1800. 1700 is well within it.
+ content: strings.Repeat("a", 1700) + ". " + strings.Repeat("b", 500),
+ maxLen: 2000,
+ expectChunks: 2,
+ checkContent: func(t *testing.T, chunks []string) {
+ if len([]rune(chunks[0])) != 1701 { // 1700 'a's + '.'
+ t.Errorf("Expected chunk 0 to be 1701 runes (split at period), got %d %q", len([]rune(chunks[0])), chunks[0])
+ }
+ if !strings.HasSuffix(chunks[0], ".") {
+ t.Errorf("Chunk 0 should end with a period, got suffix: %q", chunks[0][len(chunks[0])-5:])
+ }
+ },
+ },
}
for _, tc := range tests {