feat: telegram use parse mode ModeMarkdownV2 instead of ModeHTML (#1018)
* feat: telegram use parse mode ModeMarkdownV2 instead of ModeHTML * handle expandable block quotation starts, add test for all md2 formats * fix: linter issue * feat: added flag use_markdown_v2, corrected config, updated documentation * move parseChatID to parser_markdown_to_html * fix: tests and linter issues * fix: case with ~ * test: fixed Test_markdownToTelegramMarkdownV2 * fix: regex block-quote line > * fix: linter issues * fix: send chunk param mismatched, in edit msg use HTML parse mode too * fix: remove from .gitignore redundant comment
This commit is contained in:
parent
3e9b7ce9c1
commit
12f4029610
12 changed files with 517 additions and 139 deletions
3
.gitignore
vendored
3
.gitignore
vendored
|
|
@ -52,6 +52,9 @@ dist/
|
||||||
# Windows Application Icon/Resource
|
# Windows Application Icon/Resource
|
||||||
*.syso
|
*.syso
|
||||||
|
|
||||||
|
# Test telegram integration
|
||||||
|
cmd/telegram/
|
||||||
|
|
||||||
# Keep embedded backend dist directory placeholder in VCS
|
# Keep embedded backend dist directory placeholder in VCS
|
||||||
!web/backend/dist/
|
!web/backend/dist/
|
||||||
web/backend/dist/*
|
web/backend/dist/*
|
||||||
|
|
|
||||||
|
|
@ -78,9 +78,8 @@
|
||||||
"token": "YOUR_TELEGRAM_BOT_TOKEN",
|
"token": "YOUR_TELEGRAM_BOT_TOKEN",
|
||||||
"base_url": "",
|
"base_url": "",
|
||||||
"proxy": "",
|
"proxy": "",
|
||||||
"allow_from": [
|
"allow_from": ["YOUR_USER_ID"],
|
||||||
"YOUR_USER_ID"
|
"use_markdown_v2": false,
|
||||||
],
|
|
||||||
"reasoning_channel_id": ""
|
"reasoning_channel_id": ""
|
||||||
},
|
},
|
||||||
"discord": {
|
"discord": {
|
||||||
|
|
|
||||||
|
|
@ -42,7 +42,8 @@ Talk to your picoclaw through Telegram, Discord, WhatsApp, Matrix, QQ, DingTalk,
|
||||||
"telegram": {
|
"telegram": {
|
||||||
"enabled": true,
|
"enabled": true,
|
||||||
"token": "YOUR_BOT_TOKEN",
|
"token": "YOUR_BOT_TOKEN",
|
||||||
"allow_from": ["YOUR_USER_ID"]
|
"allow_from": ["YOUR_USER_ID"],
|
||||||
|
"use_markdown_v2": false,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
@ -63,6 +64,9 @@ Telegram command menu registration remains channel-local discovery UX; generic c
|
||||||
|
|
||||||
If command registration fails (network/API transient errors), the channel still starts and PicoClaw retries registration in the background.
|
If command registration fails (network/API transient errors), the channel still starts and PicoClaw retries registration in the background.
|
||||||
|
|
||||||
|
**4. Advanced Formatting**
|
||||||
|
You can set use_markdown_v2: true to enable enhanced formatting options. This allows the bot to utilize the full range of Telegram MarkdownV2 features, including nested styles, spoilers, and custom fixed-width blocks.
|
||||||
|
|
||||||
</details>
|
</details>
|
||||||
|
|
||||||
<details>
|
<details>
|
||||||
|
|
|
||||||
197
pkg/channels/telegram/parse_markdown_to_md_v2.go
Normal file
197
pkg/channels/telegram/parse_markdown_to_md_v2.go
Normal file
|
|
@ -0,0 +1,197 @@
|
||||||
|
package telegram
|
||||||
|
|
||||||
|
import (
|
||||||
|
"regexp"
|
||||||
|
"strings"
|
||||||
|
)
|
||||||
|
|
||||||
|
// mdV2SpecialChars are all characters that must be escaped in Telegram MarkdownV2
|
||||||
|
var mdV2SpecialChars = map[rune]bool{
|
||||||
|
'*': true,
|
||||||
|
'_': true,
|
||||||
|
'[': true,
|
||||||
|
']': true,
|
||||||
|
'(': true,
|
||||||
|
')': true,
|
||||||
|
'~': true,
|
||||||
|
'`': true,
|
||||||
|
'>': true,
|
||||||
|
'<': true,
|
||||||
|
'#': true,
|
||||||
|
'+': true,
|
||||||
|
'-': true,
|
||||||
|
'=': true,
|
||||||
|
'|': true,
|
||||||
|
'{': true,
|
||||||
|
'}': true,
|
||||||
|
'.': true,
|
||||||
|
'!': true,
|
||||||
|
'\\': true,
|
||||||
|
}
|
||||||
|
|
||||||
|
// entityPattern describes one Telegram MarkdownV2 inline entity type.
|
||||||
|
type entityPattern struct {
|
||||||
|
re *regexp.Regexp
|
||||||
|
open string
|
||||||
|
close string
|
||||||
|
}
|
||||||
|
|
||||||
|
// allEntityPatterns lists every recognized entity in priority order
|
||||||
|
// (longer / more-specific delimiters first so they win over shorter ones).
|
||||||
|
// Each entry's regex is anchored to find the first occurrence in a string.
|
||||||
|
var allEntityPatterns = []entityPattern{
|
||||||
|
// fenced code block — content is completely verbatim
|
||||||
|
{re: regexp.MustCompile("(?s)```(?:[\\w]*\\n)?[\\s\\S]*?```"), open: "```", close: "```"},
|
||||||
|
// inline code — content is completely verbatim
|
||||||
|
{re: regexp.MustCompile("`(?:[^`\\\n]|\\\\.)*`"), open: "`", close: "`"},
|
||||||
|
// expandable block-quote opener **>…
|
||||||
|
{re: regexp.MustCompile(`(?m)\*\*>(?:[^\n]*)`), open: "**>", close: ""},
|
||||||
|
// block-quote line >…
|
||||||
|
{re: regexp.MustCompile(`(?m)^>(?:[^\n]*)`), open: ">", close: ""},
|
||||||
|
// custom emoji / timestamp  — must come before plain link
|
||||||
|
{re: regexp.MustCompile(`!\[[^\]]*\]\([^)]*\)`), open: "!", close: ""},
|
||||||
|
// inline URL / user mention […](…)
|
||||||
|
{re: regexp.MustCompile(`\[[^\]]*\]\([^)]*\)`), open: "[", close: ""},
|
||||||
|
// spoiler ||…|| — before single | so it wins
|
||||||
|
{re: regexp.MustCompile(`\|\|(?:[^|\\\n]|\\.)*\|\|`), open: "||", close: "||"},
|
||||||
|
// underline __…__ — before single _ so it wins
|
||||||
|
{re: regexp.MustCompile(`__(?:[^_\\\n]|\\.)*__`), open: "__", close: "__"},
|
||||||
|
// bold *…*
|
||||||
|
{re: regexp.MustCompile(`\*(?:[^*\\\n]|\\.)*\*`), open: "*", close: "*"},
|
||||||
|
// italic _…_
|
||||||
|
{re: regexp.MustCompile(`_(?:[^_\\\n]|\\.)*_`), open: "_", close: "_"},
|
||||||
|
// strikethrough ~…~
|
||||||
|
{re: regexp.MustCompile(`~(?:[^~\\\n]|\\.)*~`), open: "~", close: "~"},
|
||||||
|
}
|
||||||
|
|
||||||
|
// verbatimEntities are entity types whose inner content must never be
|
||||||
|
// touched (code blocks, URLs, quotes, custom emoji).
|
||||||
|
// Their content is passed through completely unchanged.
|
||||||
|
var verbatimEntities = map[string]bool{
|
||||||
|
"```": true,
|
||||||
|
"`": true,
|
||||||
|
"**>": true,
|
||||||
|
">": true,
|
||||||
|
"!": true,
|
||||||
|
"[": true,
|
||||||
|
}
|
||||||
|
|
||||||
|
// markdownToTelegramMarkdownV2 converts a Markdown string into a string safe
|
||||||
|
// for sending with Telegram's MarkdownV2 parse mode.
|
||||||
|
//
|
||||||
|
// Rules:
|
||||||
|
// - Markdown headings (# … ######) are converted to *bold*.
|
||||||
|
// - **bold** Markdown syntax is converted to *bold*.
|
||||||
|
// - Recognized Telegram MarkdownV2 entity spans are preserved; their inner
|
||||||
|
// content is processed recursively so that nested valid entities are kept
|
||||||
|
// intact while stray special characters are escaped.
|
||||||
|
// - All plain-text segments have their MarkdownV2 special characters escaped.
|
||||||
|
//
|
||||||
|
// Reference: https://core.telegram.org/bots/api#formatting-options
|
||||||
|
func markdownToTelegramMarkdownV2(text string) string {
|
||||||
|
// 1. Convert Markdown headings → *escaped heading text*
|
||||||
|
text = reHeading.ReplaceAllStringFunc(text, func(match string) string {
|
||||||
|
sub := reHeading.FindStringSubmatch(match)
|
||||||
|
if len(sub) < 2 {
|
||||||
|
return match
|
||||||
|
}
|
||||||
|
// The heading content is fresh plain text — escape everything
|
||||||
|
// including * so the resulting *…* bold span stays valid.
|
||||||
|
return "*" + escapeMarkdownV2(sub[1]) + "*"
|
||||||
|
})
|
||||||
|
|
||||||
|
// 2. Convert **bold** → *bold*
|
||||||
|
text = reBoldStar.ReplaceAllString(text, "*$1*")
|
||||||
|
|
||||||
|
// 3. Recursively escape the full string.
|
||||||
|
return processText(text)
|
||||||
|
}
|
||||||
|
|
||||||
|
// processText walks `text`, finds the leftmost / longest matching entity,
|
||||||
|
// escapes the gap before it, processes the entity (recursing into its inner
|
||||||
|
// content when appropriate), then continues with the remainder.
|
||||||
|
func processText(text string) string {
|
||||||
|
if text == "" {
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
|
||||||
|
// Find the leftmost match among all entity patterns.
|
||||||
|
bestStart := -1
|
||||||
|
bestEnd := -1
|
||||||
|
var bestPat *entityPattern
|
||||||
|
|
||||||
|
for i := range allEntityPatterns {
|
||||||
|
p := &allEntityPatterns[i]
|
||||||
|
loc := p.re.FindStringIndex(text)
|
||||||
|
if loc == nil {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if bestStart == -1 || loc[0] < bestStart ||
|
||||||
|
(loc[0] == bestStart && (loc[1]-loc[0]) > (bestEnd-bestStart)) {
|
||||||
|
bestStart = loc[0]
|
||||||
|
bestEnd = loc[1]
|
||||||
|
bestPat = p
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if bestPat == nil {
|
||||||
|
// No entity found — escape everything.
|
||||||
|
return escapeMarkdownV2(text)
|
||||||
|
}
|
||||||
|
|
||||||
|
var b strings.Builder
|
||||||
|
|
||||||
|
// Plain text before the entity.
|
||||||
|
if bestStart > 0 {
|
||||||
|
b.WriteString(escapeMarkdownV2(text[:bestStart]))
|
||||||
|
}
|
||||||
|
|
||||||
|
// The matched entity span.
|
||||||
|
matched := text[bestStart:bestEnd]
|
||||||
|
|
||||||
|
if verbatimEntities[bestPat.open] {
|
||||||
|
// Code blocks, URLs, quotes: pass through completely untouched.
|
||||||
|
b.WriteString(matched)
|
||||||
|
} else {
|
||||||
|
// Inline formatting (bold, italic, underline, strikethrough, spoiler):
|
||||||
|
// keep the delimiters and recursively process the inner content so that
|
||||||
|
// nested entities survive but stray specials get escaped.
|
||||||
|
openLen := len(bestPat.open)
|
||||||
|
closeLen := len(bestPat.close)
|
||||||
|
inner := matched[openLen : len(matched)-closeLen]
|
||||||
|
|
||||||
|
b.WriteString(bestPat.open)
|
||||||
|
b.WriteString(processText(inner))
|
||||||
|
b.WriteString(bestPat.close)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Continue with the remainder of the string.
|
||||||
|
b.WriteString(processText(text[bestEnd:]))
|
||||||
|
|
||||||
|
return b.String()
|
||||||
|
}
|
||||||
|
|
||||||
|
// escapeMarkdownV2 escapes every MarkdownV2 special character in a plain-text
|
||||||
|
// segment (i.e. a segment that is not part of any recognized entity).
|
||||||
|
// Already-escaped sequences (backslash + char) are forwarded verbatim to avoid
|
||||||
|
// double-escaping.
|
||||||
|
func escapeMarkdownV2(s string) string {
|
||||||
|
var b strings.Builder
|
||||||
|
b.Grow(len(s) + 8)
|
||||||
|
runes := []rune(s)
|
||||||
|
for i := 0; i < len(runes); i++ {
|
||||||
|
ch := runes[i]
|
||||||
|
// Forward an existing escape sequence verbatim.
|
||||||
|
if ch == '\\' && i+1 < len(runes) {
|
||||||
|
b.WriteRune(ch)
|
||||||
|
b.WriteRune(runes[i+1])
|
||||||
|
i++
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if mdV2SpecialChars[ch] {
|
||||||
|
b.WriteByte('\\')
|
||||||
|
}
|
||||||
|
b.WriteRune(ch)
|
||||||
|
}
|
||||||
|
return b.String()
|
||||||
|
}
|
||||||
68
pkg/channels/telegram/parse_markdown_to_md_v2_test.go
Normal file
68
pkg/channels/telegram/parse_markdown_to_md_v2_test.go
Normal file
|
|
@ -0,0 +1,68 @@
|
||||||
|
package telegram
|
||||||
|
|
||||||
|
import (
|
||||||
|
_ "embed"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"github.com/stretchr/testify/require"
|
||||||
|
)
|
||||||
|
|
||||||
|
//go:embed testdata/md2_all_formats.txt
|
||||||
|
var md2AllFormats string
|
||||||
|
|
||||||
|
func Test_markdownToTelegramMarkdownV2(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
input string
|
||||||
|
expected string
|
||||||
|
}{
|
||||||
|
{
|
||||||
|
name: "heading -> bolding",
|
||||||
|
input: `## HeadingH2 #`,
|
||||||
|
expected: "*HeadingH2 \\#*",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "strikethrough",
|
||||||
|
input: "~strikethroughMD~",
|
||||||
|
expected: "~strikethroughMD~",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "inline URL",
|
||||||
|
input: "[inline URL](http://www.example.com/)",
|
||||||
|
expected: "[inline URL](http://www.example.com/)",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "all telegram formats",
|
||||||
|
input: md2AllFormats,
|
||||||
|
expected: md2AllFormats,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "empty",
|
||||||
|
input: "",
|
||||||
|
expected: "",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "one letter",
|
||||||
|
input: "o",
|
||||||
|
expected: "o",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "",
|
||||||
|
input: "*Last update: ~10 24h*",
|
||||||
|
expected: "*Last update: \\~10 24h*",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "",
|
||||||
|
input: "<Market Capitalization>",
|
||||||
|
expected: "\\<Market Capitalization\\>",
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, tc := range cases {
|
||||||
|
t.Run(tc.name, func(t *testing.T) {
|
||||||
|
actual := markdownToTelegramMarkdownV2(tc.input)
|
||||||
|
|
||||||
|
require.EqualValues(t, tc.expected, actual)
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
111
pkg/channels/telegram/parser_markdown_to_html.go
Normal file
111
pkg/channels/telegram/parser_markdown_to_html.go
Normal file
|
|
@ -0,0 +1,111 @@
|
||||||
|
package telegram
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"strings"
|
||||||
|
)
|
||||||
|
|
||||||
|
func markdownToTelegramHTML(text string) string {
|
||||||
|
if text == "" {
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
|
||||||
|
codeBlocks := extractCodeBlocks(text)
|
||||||
|
text = codeBlocks.text
|
||||||
|
|
||||||
|
inlineCodes := extractInlineCodes(text)
|
||||||
|
text = inlineCodes.text
|
||||||
|
|
||||||
|
text = reHeading.ReplaceAllString(text, "$1")
|
||||||
|
|
||||||
|
text = reBlockquote.ReplaceAllString(text, "$1")
|
||||||
|
|
||||||
|
text = escapeHTML(text)
|
||||||
|
|
||||||
|
text = reLink.ReplaceAllString(text, `<a href="$2">$1</a>`)
|
||||||
|
|
||||||
|
text = reBoldStar.ReplaceAllString(text, "<b>$1</b>")
|
||||||
|
|
||||||
|
text = reBoldUnder.ReplaceAllString(text, "<b>$1</b>")
|
||||||
|
|
||||||
|
text = reItalic.ReplaceAllStringFunc(text, func(s string) string {
|
||||||
|
match := reItalic.FindStringSubmatch(s)
|
||||||
|
if len(match) < 2 {
|
||||||
|
return s
|
||||||
|
}
|
||||||
|
return "<i>" + match[1] + "</i>"
|
||||||
|
})
|
||||||
|
|
||||||
|
text = reStrike.ReplaceAllString(text, "<s>$1</s>")
|
||||||
|
|
||||||
|
text = reListItem.ReplaceAllString(text, "• ")
|
||||||
|
|
||||||
|
for i, code := range inlineCodes.codes {
|
||||||
|
escaped := escapeHTML(code)
|
||||||
|
text = strings.ReplaceAll(text, fmt.Sprintf("\x00IC%d\x00", i), fmt.Sprintf("<code>%s</code>", escaped))
|
||||||
|
}
|
||||||
|
|
||||||
|
for i, code := range codeBlocks.codes {
|
||||||
|
escaped := escapeHTML(code)
|
||||||
|
text = strings.ReplaceAll(
|
||||||
|
text,
|
||||||
|
fmt.Sprintf("\x00CB%d\x00", i),
|
||||||
|
fmt.Sprintf("<pre><code>%s</code></pre>", escaped),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
return text
|
||||||
|
}
|
||||||
|
|
||||||
|
type codeBlockMatch struct {
|
||||||
|
text string
|
||||||
|
codes []string
|
||||||
|
}
|
||||||
|
|
||||||
|
func extractCodeBlocks(text string) codeBlockMatch {
|
||||||
|
matches := reCodeBlock.FindAllStringSubmatch(text, -1)
|
||||||
|
|
||||||
|
codes := make([]string, 0, len(matches))
|
||||||
|
for _, match := range matches {
|
||||||
|
codes = append(codes, match[1])
|
||||||
|
}
|
||||||
|
|
||||||
|
i := 0
|
||||||
|
text = reCodeBlock.ReplaceAllStringFunc(text, func(m string) string {
|
||||||
|
placeholder := fmt.Sprintf("\x00CB%d\x00", i)
|
||||||
|
i++
|
||||||
|
return placeholder
|
||||||
|
})
|
||||||
|
|
||||||
|
return codeBlockMatch{text: text, codes: codes}
|
||||||
|
}
|
||||||
|
|
||||||
|
type inlineCodeMatch struct {
|
||||||
|
text string
|
||||||
|
codes []string
|
||||||
|
}
|
||||||
|
|
||||||
|
func extractInlineCodes(text string) inlineCodeMatch {
|
||||||
|
matches := reInlineCode.FindAllStringSubmatch(text, -1)
|
||||||
|
|
||||||
|
codes := make([]string, 0, len(matches))
|
||||||
|
for _, match := range matches {
|
||||||
|
codes = append(codes, match[1])
|
||||||
|
}
|
||||||
|
|
||||||
|
i := 0
|
||||||
|
text = reInlineCode.ReplaceAllStringFunc(text, func(m string) string {
|
||||||
|
placeholder := fmt.Sprintf("\x00IC%d\x00", i)
|
||||||
|
i++
|
||||||
|
return placeholder
|
||||||
|
})
|
||||||
|
|
||||||
|
return inlineCodeMatch{text: text, codes: codes}
|
||||||
|
}
|
||||||
|
|
||||||
|
func escapeHTML(text string) string {
|
||||||
|
text = strings.ReplaceAll(text, "&", "&")
|
||||||
|
text = strings.ReplaceAll(text, "<", "<")
|
||||||
|
text = strings.ReplaceAll(text, ">", ">")
|
||||||
|
return text
|
||||||
|
}
|
||||||
|
|
@ -27,7 +27,7 @@ import (
|
||||||
)
|
)
|
||||||
|
|
||||||
var (
|
var (
|
||||||
reHeading = regexp.MustCompile(`^#{1,6}\s+(.+)$`)
|
reHeading = regexp.MustCompile(`(?m)^#{1,6}\s+([^\n]+)`)
|
||||||
reBlockquote = regexp.MustCompile(`^>\s*(.*)$`)
|
reBlockquote = regexp.MustCompile(`^>\s*(.*)$`)
|
||||||
reLink = regexp.MustCompile(`\[([^\]]+)\]\(([^)]+)\)`)
|
reLink = regexp.MustCompile(`\[([^\]]+)\]\(([^)]+)\)`)
|
||||||
reBoldStar = regexp.MustCompile(`\*\*(.+?)\*\*`)
|
reBoldStar = regexp.MustCompile(`\*\*(.+?)\*\*`)
|
||||||
|
|
@ -170,6 +170,8 @@ func (c *TelegramChannel) Send(ctx context.Context, msg bus.OutboundMessage) err
|
||||||
return channels.ErrNotRunning
|
return channels.ErrNotRunning
|
||||||
}
|
}
|
||||||
|
|
||||||
|
useMarkdownV2 := c.config.Channels.Telegram.UseMarkdownV2
|
||||||
|
|
||||||
chatID, threadID, err := parseTelegramChatID(msg.ChatID)
|
chatID, threadID, err := parseTelegramChatID(msg.ChatID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("invalid chat ID %s: %w", msg.ChatID, channels.ErrSendFailed)
|
return fmt.Errorf("invalid chat ID %s: %w", msg.ChatID, channels.ErrSendFailed)
|
||||||
|
|
@ -188,11 +190,11 @@ func (c *TelegramChannel) Send(ctx context.Context, msg bus.OutboundMessage) err
|
||||||
chunk := queue[0]
|
chunk := queue[0]
|
||||||
queue = queue[1:]
|
queue = queue[1:]
|
||||||
|
|
||||||
htmlContent := markdownToTelegramHTML(chunk)
|
content := parseContent(chunk, useMarkdownV2)
|
||||||
|
|
||||||
if len([]rune(htmlContent)) > 4096 {
|
if len([]rune(content)) > 4096 {
|
||||||
runeChunk := []rune(chunk)
|
runeChunk := []rune(chunk)
|
||||||
ratio := float64(len(runeChunk)) / float64(len([]rune(htmlContent)))
|
ratio := float64(len(runeChunk)) / float64(len([]rune(content)))
|
||||||
smallerLen := int(float64(4096) * ratio * 0.95) // 5% safety margin
|
smallerLen := int(float64(4096) * ratio * 0.95) // 5% safety margin
|
||||||
|
|
||||||
// Guarantee progress: if estimated length is >= chunk length, force it smaller
|
// Guarantee progress: if estimated length is >= chunk length, force it smaller
|
||||||
|
|
@ -201,7 +203,14 @@ func (c *TelegramChannel) Send(ctx context.Context, msg bus.OutboundMessage) err
|
||||||
}
|
}
|
||||||
|
|
||||||
if smallerLen <= 0 {
|
if smallerLen <= 0 {
|
||||||
if err := c.sendHTMLChunk(ctx, chatID, threadID, htmlContent, chunk, replyToID); err != nil {
|
if err := c.sendChunk(ctx, sendChunkParams{
|
||||||
|
chatID: chatID,
|
||||||
|
threadID: threadID,
|
||||||
|
content: content,
|
||||||
|
replyToID: replyToID,
|
||||||
|
mdFallback: chunk,
|
||||||
|
useMarkdownV2: useMarkdownV2,
|
||||||
|
}); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
replyToID = ""
|
replyToID = ""
|
||||||
|
|
@ -232,7 +241,14 @@ func (c *TelegramChannel) Send(ctx context.Context, msg bus.OutboundMessage) err
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
|
|
||||||
if err := c.sendHTMLChunk(ctx, chatID, threadID, htmlContent, chunk, replyToID); err != nil {
|
if err := c.sendChunk(ctx, sendChunkParams{
|
||||||
|
chatID: chatID,
|
||||||
|
threadID: threadID,
|
||||||
|
content: content,
|
||||||
|
replyToID: replyToID,
|
||||||
|
mdFallback: chunk,
|
||||||
|
useMarkdownV2: useMarkdownV2,
|
||||||
|
}); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
// Only the first chunk should be a reply; subsequent chunks are normal messages.
|
// Only the first chunk should be a reply; subsequent chunks are normal messages.
|
||||||
|
|
@ -242,17 +258,31 @@ func (c *TelegramChannel) Send(ctx context.Context, msg bus.OutboundMessage) err
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// sendHTMLChunk sends a single HTML message, falling back to the original
|
type sendChunkParams struct {
|
||||||
// markdown as plain text on parse failure so users never see raw HTML tags.
|
chatID int64
|
||||||
func (c *TelegramChannel) sendHTMLChunk(
|
threadID int
|
||||||
ctx context.Context, chatID int64, threadID int, htmlContent, mdFallback string, replyToID string,
|
content string
|
||||||
) error {
|
replyToID string
|
||||||
tgMsg := tu.Message(tu.ID(chatID), htmlContent)
|
mdFallback string
|
||||||
tgMsg.ParseMode = telego.ModeHTML
|
useMarkdownV2 bool
|
||||||
tgMsg.MessageThreadID = threadID
|
}
|
||||||
|
|
||||||
if replyToID != "" {
|
// sendChunk sends a single HTML/MarkdownV2 message, falling back to the original
|
||||||
if mid, parseErr := strconv.Atoi(replyToID); parseErr == nil {
|
// markdown as plain text on parse failure so users never see raw HTML/MarkdownV2 tags.
|
||||||
|
func (c *TelegramChannel) sendChunk(
|
||||||
|
ctx context.Context,
|
||||||
|
params sendChunkParams,
|
||||||
|
) error {
|
||||||
|
tgMsg := tu.Message(tu.ID(params.chatID), params.content)
|
||||||
|
tgMsg.MessageThreadID = params.threadID
|
||||||
|
if params.useMarkdownV2 {
|
||||||
|
tgMsg.WithParseMode(telego.ModeMarkdownV2)
|
||||||
|
} else {
|
||||||
|
tgMsg.WithParseMode(telego.ModeHTML)
|
||||||
|
}
|
||||||
|
|
||||||
|
if params.replyToID != "" {
|
||||||
|
if mid, parseErr := strconv.Atoi(params.replyToID); parseErr == nil {
|
||||||
tgMsg.ReplyParameters = &telego.ReplyParameters{
|
tgMsg.ReplyParameters = &telego.ReplyParameters{
|
||||||
MessageID: mid,
|
MessageID: mid,
|
||||||
}
|
}
|
||||||
|
|
@ -260,15 +290,15 @@ func (c *TelegramChannel) sendHTMLChunk(
|
||||||
}
|
}
|
||||||
|
|
||||||
if _, err := c.bot.SendMessage(ctx, tgMsg); err != nil {
|
if _, err := c.bot.SendMessage(ctx, tgMsg); err != nil {
|
||||||
logger.ErrorCF("telegram", "HTML parse failed, falling back to plain text", map[string]any{
|
logParseFailed(err, params.useMarkdownV2)
|
||||||
"error": err.Error(),
|
|
||||||
})
|
tgMsg.Text = params.mdFallback
|
||||||
tgMsg.Text = mdFallback
|
|
||||||
tgMsg.ParseMode = ""
|
tgMsg.ParseMode = ""
|
||||||
if _, err = c.bot.SendMessage(ctx, tgMsg); err != nil {
|
if _, err = c.bot.SendMessage(ctx, tgMsg); err != nil {
|
||||||
return fmt.Errorf("telegram send: %w", channels.ErrTemporary)
|
return fmt.Errorf("telegram send: %w", channels.ErrTemporary)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -309,6 +339,7 @@ func (c *TelegramChannel) StartTyping(ctx context.Context, chatID string) (func(
|
||||||
|
|
||||||
// EditMessage implements channels.MessageEditor.
|
// EditMessage implements channels.MessageEditor.
|
||||||
func (c *TelegramChannel) EditMessage(ctx context.Context, chatID string, messageID string, content string) error {
|
func (c *TelegramChannel) EditMessage(ctx context.Context, chatID string, messageID string, content string) error {
|
||||||
|
useMarkdownV2 := c.config.Channels.Telegram.UseMarkdownV2
|
||||||
cid, _, err := parseTelegramChatID(chatID)
|
cid, _, err := parseTelegramChatID(chatID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
|
|
@ -317,10 +348,19 @@ func (c *TelegramChannel) EditMessage(ctx context.Context, chatID string, messag
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
htmlContent := markdownToTelegramHTML(content)
|
parsedContent := parseContent(content, useMarkdownV2)
|
||||||
editMsg := tu.EditMessageText(tu.ID(cid), mid, htmlContent)
|
editMsg := tu.EditMessageText(tu.ID(cid), mid, parsedContent)
|
||||||
editMsg.ParseMode = telego.ModeHTML
|
if useMarkdownV2 {
|
||||||
|
editMsg.WithParseMode(telego.ModeMarkdownV2)
|
||||||
|
} else {
|
||||||
|
editMsg.WithParseMode(telego.ModeHTML)
|
||||||
|
}
|
||||||
_, err = c.bot.EditMessageText(ctx, editMsg)
|
_, err = c.bot.EditMessageText(ctx, editMsg)
|
||||||
|
if err != nil {
|
||||||
|
logParseFailed(err, useMarkdownV2)
|
||||||
|
_, err = c.bot.EditMessageText(ctx, tu.EditMessageText(tu.ID(cid), mid, content))
|
||||||
|
}
|
||||||
|
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -668,6 +708,14 @@ func (c *TelegramChannel) downloadFile(ctx context.Context, fileID, ext string)
|
||||||
return c.downloadFileWithInfo(file, ext)
|
return c.downloadFileWithInfo(file, ext)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func parseContent(text string, useMarkdownV2 bool) string {
|
||||||
|
if useMarkdownV2 {
|
||||||
|
return markdownToTelegramMarkdownV2(text)
|
||||||
|
}
|
||||||
|
|
||||||
|
return markdownToTelegramHTML(text)
|
||||||
|
}
|
||||||
|
|
||||||
// parseTelegramChatID splits "chatID/threadID" into its components.
|
// parseTelegramChatID splits "chatID/threadID" into its components.
|
||||||
// Returns threadID=0 when no "/" is present (non-forum messages).
|
// Returns threadID=0 when no "/" is present (non-forum messages).
|
||||||
func parseTelegramChatID(chatID string) (int64, int, error) {
|
func parseTelegramChatID(chatID string) (int64, int, error) {
|
||||||
|
|
@ -687,109 +735,18 @@ func parseTelegramChatID(chatID string) (int64, int, error) {
|
||||||
return cid, tid, nil
|
return cid, tid, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func markdownToTelegramHTML(text string) string {
|
func logParseFailed(err error, useMarkdownV2 bool) {
|
||||||
if text == "" {
|
parsingName := "HTML"
|
||||||
return ""
|
if useMarkdownV2 {
|
||||||
|
parsingName = "MarkdownV2"
|
||||||
}
|
}
|
||||||
|
|
||||||
codeBlocks := extractCodeBlocks(text)
|
logger.ErrorCF("telegram",
|
||||||
text = codeBlocks.text
|
fmt.Sprintf("%s parse failed, falling back to plain text", parsingName),
|
||||||
|
map[string]any{
|
||||||
inlineCodes := extractInlineCodes(text)
|
"error": err.Error(),
|
||||||
text = inlineCodes.text
|
},
|
||||||
|
)
|
||||||
text = reHeading.ReplaceAllString(text, "$1")
|
|
||||||
|
|
||||||
text = reBlockquote.ReplaceAllString(text, "$1")
|
|
||||||
|
|
||||||
text = escapeHTML(text)
|
|
||||||
|
|
||||||
text = reLink.ReplaceAllString(text, `<a href="$2">$1</a>`)
|
|
||||||
|
|
||||||
text = reBoldStar.ReplaceAllString(text, "<b>$1</b>")
|
|
||||||
|
|
||||||
text = reBoldUnder.ReplaceAllString(text, "<b>$1</b>")
|
|
||||||
|
|
||||||
text = reItalic.ReplaceAllStringFunc(text, func(s string) string {
|
|
||||||
match := reItalic.FindStringSubmatch(s)
|
|
||||||
if len(match) < 2 {
|
|
||||||
return s
|
|
||||||
}
|
|
||||||
return "<i>" + match[1] + "</i>"
|
|
||||||
})
|
|
||||||
|
|
||||||
text = reStrike.ReplaceAllString(text, "<s>$1</s>")
|
|
||||||
|
|
||||||
text = reListItem.ReplaceAllString(text, "• ")
|
|
||||||
|
|
||||||
for i, code := range inlineCodes.codes {
|
|
||||||
escaped := escapeHTML(code)
|
|
||||||
text = strings.ReplaceAll(text, fmt.Sprintf("\x00IC%d\x00", i), fmt.Sprintf("<code>%s</code>", escaped))
|
|
||||||
}
|
|
||||||
|
|
||||||
for i, code := range codeBlocks.codes {
|
|
||||||
escaped := escapeHTML(code)
|
|
||||||
text = strings.ReplaceAll(
|
|
||||||
text,
|
|
||||||
fmt.Sprintf("\x00CB%d\x00", i),
|
|
||||||
fmt.Sprintf("<pre><code>%s</code></pre>", escaped),
|
|
||||||
)
|
|
||||||
}
|
|
||||||
|
|
||||||
return text
|
|
||||||
}
|
|
||||||
|
|
||||||
type codeBlockMatch struct {
|
|
||||||
text string
|
|
||||||
codes []string
|
|
||||||
}
|
|
||||||
|
|
||||||
func extractCodeBlocks(text string) codeBlockMatch {
|
|
||||||
matches := reCodeBlock.FindAllStringSubmatch(text, -1)
|
|
||||||
|
|
||||||
codes := make([]string, 0, len(matches))
|
|
||||||
for _, match := range matches {
|
|
||||||
codes = append(codes, match[1])
|
|
||||||
}
|
|
||||||
|
|
||||||
i := 0
|
|
||||||
text = reCodeBlock.ReplaceAllStringFunc(text, func(m string) string {
|
|
||||||
placeholder := fmt.Sprintf("\x00CB%d\x00", i)
|
|
||||||
i++
|
|
||||||
return placeholder
|
|
||||||
})
|
|
||||||
|
|
||||||
return codeBlockMatch{text: text, codes: codes}
|
|
||||||
}
|
|
||||||
|
|
||||||
type inlineCodeMatch struct {
|
|
||||||
text string
|
|
||||||
codes []string
|
|
||||||
}
|
|
||||||
|
|
||||||
func extractInlineCodes(text string) inlineCodeMatch {
|
|
||||||
matches := reInlineCode.FindAllStringSubmatch(text, -1)
|
|
||||||
|
|
||||||
codes := make([]string, 0, len(matches))
|
|
||||||
for _, match := range matches {
|
|
||||||
codes = append(codes, match[1])
|
|
||||||
}
|
|
||||||
|
|
||||||
i := 0
|
|
||||||
text = reInlineCode.ReplaceAllStringFunc(text, func(m string) string {
|
|
||||||
placeholder := fmt.Sprintf("\x00IC%d\x00", i)
|
|
||||||
i++
|
|
||||||
return placeholder
|
|
||||||
})
|
|
||||||
|
|
||||||
return inlineCodeMatch{text: text, codes: codes}
|
|
||||||
}
|
|
||||||
|
|
||||||
func escapeHTML(text string) string {
|
|
||||||
text = strings.ReplaceAll(text, "&", "&")
|
|
||||||
text = strings.ReplaceAll(text, "<", "<")
|
|
||||||
text = strings.ReplaceAll(text, ">", ">")
|
|
||||||
return text
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// isBotMentioned checks if the bot is mentioned in the message via entities.
|
// isBotMentioned checks if the bot is mentioned in the message via entities.
|
||||||
|
|
|
||||||
|
|
@ -17,6 +17,7 @@ import (
|
||||||
|
|
||||||
"github.com/sipeed/picoclaw/pkg/bus"
|
"github.com/sipeed/picoclaw/pkg/bus"
|
||||||
"github.com/sipeed/picoclaw/pkg/channels"
|
"github.com/sipeed/picoclaw/pkg/channels"
|
||||||
|
"github.com/sipeed/picoclaw/pkg/config"
|
||||||
"github.com/sipeed/picoclaw/pkg/media"
|
"github.com/sipeed/picoclaw/pkg/media"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
@ -131,6 +132,7 @@ func newTestChannelWithConstructor(
|
||||||
BaseChannel: base,
|
BaseChannel: base,
|
||||||
bot: bot,
|
bot: bot,
|
||||||
chatIDs: make(map[string]int64),
|
chatIDs: make(map[string]int64),
|
||||||
|
config: config.DefaultConfig(),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
|
||||||
31
pkg/channels/telegram/testdata/md2_all_formats.txt
vendored
Normal file
31
pkg/channels/telegram/testdata/md2_all_formats.txt
vendored
Normal file
|
|
@ -0,0 +1,31 @@
|
||||||
|
*bold \*text*
|
||||||
|
_italic \*text_
|
||||||
|
__underline__
|
||||||
|
~strikethrough~
|
||||||
|
||spoiler||
|
||||||
|
*bold _italic bold ~italic bold strikethrough ||italic bold strikethrough spoiler||~ __underline italic bold___ bold*
|
||||||
|
[inline URL](http://www.example.com/)
|
||||||
|
[inline mention of a user](tg://user?id=123456789)
|
||||||
|

|
||||||
|

|
||||||
|

|
||||||
|

|
||||||
|

|
||||||
|
`inline fixed-width code`
|
||||||
|
```
|
||||||
|
pre-formatted fixed-width code block
|
||||||
|
```
|
||||||
|
```python
|
||||||
|
pre-formatted fixed-width code block written in the Python programming language
|
||||||
|
```
|
||||||
|
>Block quotation started
|
||||||
|
>Block quotation continued
|
||||||
|
>Block quotation continued
|
||||||
|
>Block quotation continued
|
||||||
|
>The last line of the block quotation
|
||||||
|
**>The expandable block quotation started right after the previous block quotation
|
||||||
|
>It is separated from the previous block quotation by an empty bold entity
|
||||||
|
>Expandable block quotation continued
|
||||||
|
>Hidden by default part of the expandable block quotation started
|
||||||
|
>Expandable block quotation continued
|
||||||
|
>The last line of the expandable block quotation with the expandability mark||
|
||||||
|
|
@ -311,6 +311,7 @@ type TelegramConfig struct {
|
||||||
Typing TypingConfig `json:"typing,omitempty"`
|
Typing TypingConfig `json:"typing,omitempty"`
|
||||||
Placeholder PlaceholderConfig `json:"placeholder,omitempty"`
|
Placeholder PlaceholderConfig `json:"placeholder,omitempty"`
|
||||||
ReasoningChannelID string `json:"reasoning_channel_id" env:"PICOCLAW_CHANNELS_TELEGRAM_REASONING_CHANNEL_ID"`
|
ReasoningChannelID string `json:"reasoning_channel_id" env:"PICOCLAW_CHANNELS_TELEGRAM_REASONING_CHANNEL_ID"`
|
||||||
|
UseMarkdownV2 bool `json:"use_markdown_v2" env:"PICOCLAW_CHANNELS_TELEGRAM_USE_MARKDOWN_V2"`
|
||||||
}
|
}
|
||||||
|
|
||||||
type FeishuConfig struct {
|
type FeishuConfig struct {
|
||||||
|
|
|
||||||
|
|
@ -58,6 +58,7 @@ func DefaultConfig() *Config {
|
||||||
Enabled: true,
|
Enabled: true,
|
||||||
Text: "Thinking... 💭",
|
Text: "Thinking... 💭",
|
||||||
},
|
},
|
||||||
|
UseMarkdownV2: false,
|
||||||
},
|
},
|
||||||
Feishu: FeishuConfig{
|
Feishu: FeishuConfig{
|
||||||
Enabled: false,
|
Enabled: false,
|
||||||
|
|
|
||||||
|
|
@ -132,11 +132,12 @@ type OpenClawChannels struct {
|
||||||
}
|
}
|
||||||
|
|
||||||
type OpenClawTelegramConfig struct {
|
type OpenClawTelegramConfig struct {
|
||||||
BotToken *string `json:"botToken"`
|
BotToken *string `json:"botToken"`
|
||||||
AllowFrom []string `json:"allowFrom"`
|
AllowFrom []string `json:"allowFrom"`
|
||||||
GroupPolicy *string `json:"groupPolicy"`
|
GroupPolicy *string `json:"groupPolicy"`
|
||||||
DmPolicy *string `json:"dmPolicy"`
|
DmPolicy *string `json:"dmPolicy"`
|
||||||
Enabled *bool `json:"enabled"`
|
Enabled *bool `json:"enabled"`
|
||||||
|
UseMarkdownV2 *bool `json:"useMarkdownV2"`
|
||||||
}
|
}
|
||||||
|
|
||||||
type OpenClawDiscordConfig struct {
|
type OpenClawDiscordConfig struct {
|
||||||
|
|
@ -645,10 +646,11 @@ type WhatsAppConfig struct {
|
||||||
}
|
}
|
||||||
|
|
||||||
type TelegramConfig struct {
|
type TelegramConfig struct {
|
||||||
Enabled bool `json:"enabled"`
|
Enabled bool `json:"enabled"`
|
||||||
Token string `json:"token"`
|
Token string `json:"token"`
|
||||||
Proxy string `json:"proxy"`
|
Proxy string `json:"proxy"`
|
||||||
AllowFrom []string `json:"allow_from"`
|
AllowFrom []string `json:"allow_from"`
|
||||||
|
UseMarkdownV2 bool `json:"use_markdown_v2"`
|
||||||
}
|
}
|
||||||
|
|
||||||
type FeishuConfig struct {
|
type FeishuConfig struct {
|
||||||
|
|
@ -777,9 +779,11 @@ func (c *OpenClawConfig) convertChannels(warnings *[]string) ChannelsConfig {
|
||||||
|
|
||||||
if c.Channels.Telegram != nil {
|
if c.Channels.Telegram != nil {
|
||||||
enabled := c.Channels.Telegram.Enabled == nil || *c.Channels.Telegram.Enabled
|
enabled := c.Channels.Telegram.Enabled == nil || *c.Channels.Telegram.Enabled
|
||||||
|
useMarkdownV2 := c.Channels.Telegram.UseMarkdownV2 != nil && *c.Channels.Telegram.UseMarkdownV2
|
||||||
channels.Telegram = TelegramConfig{
|
channels.Telegram = TelegramConfig{
|
||||||
Enabled: enabled,
|
Enabled: enabled,
|
||||||
AllowFrom: c.Channels.Telegram.AllowFrom,
|
AllowFrom: c.Channels.Telegram.AllowFrom,
|
||||||
|
UseMarkdownV2: useMarkdownV2,
|
||||||
}
|
}
|
||||||
if c.Channels.Telegram.BotToken != nil {
|
if c.Channels.Telegram.BotToken != nil {
|
||||||
channels.Telegram.Token = *c.Channels.Telegram.BotToken
|
channels.Telegram.Token = *c.Channels.Telegram.BotToken
|
||||||
|
|
|
||||||
Loading…
Add table
Reference in a new issue