Implement function to prune string keeping HTML closing tags (#1870)

* Implement function to prune string keeping HTML closing tags

Fixes #1587

* change const name

remove unneeded comment

* move pruneHTML to separated file

* move const back to telegram.go

* Add unit tests for string array manipulation and HTML pruning

Introduce comprehensive test cases for stringArr methods (Push, Pop, Unshift, Shift, String) to ensure correct behavior and state management. Additionally, add tests for HTML pruning functions (pruneHTML, pruneStringToWord) to validate handling of length constraints and formatting scenarios.

* Improve behavior

* Fix pruneHTML to count visible text only, add parent text pruning

- Fix bug where HTML tags were counted toward the character limit
  instead of only visible text content
- Add pruning for parent comment text in Telegram notifications
- Simplify pruneStringToWord using strings.LastIndex
- Remove unused stringArr type and its tests
- Consolidate and simplify test cases

---------

Co-authored-by: Umputun <umputun@gmail.com>
Co-authored-by: Dmitry Verkhoturov <paskal.07@gmail.com>
This commit is contained in:
Alik Send
2025-12-04 11:11:10 -06:00
committed by GitHub
co-authored by Umputun Dmitry Verkhoturov
parent 564e8ff316
commit d01b738741
4 changed files with 137 additions and 2 deletions
+77
View File
@@ -0,0 +1,77 @@
package notify
import (
"fmt"
"strings"
"golang.org/x/net/html"
)
// pruneHTML prunes string keeping HTML closing tags.
// maxLength applies to visible text only, not HTML tags.
func pruneHTML(htmlText string, maxLength int) string {
var result strings.Builder
var endTokens []string
visibleLen := 0
suffix := "..."
suffixLen := len(suffix)
tokenizer := html.NewTokenizer(strings.NewReader(htmlText))
for {
if tokenizer.Next() == html.ErrorToken {
return result.String()
}
token := tokenizer.Token()
switch token.Type {
case html.CommentToken, html.DoctypeToken:
continue
case html.StartTagToken:
endTokens = append([]string{fmt.Sprintf("</%s>", token.Data)}, endTokens...)
result.WriteString(token.String())
case html.EndTagToken:
if len(endTokens) > 0 {
endTokens = endTokens[1:]
}
result.WriteString(token.String())
case html.SelfClosingTagToken:
result.WriteString(token.String())
case html.TextToken:
text := token.String()
if visibleLen+len(text)+suffixLen > maxLength {
remaining := maxLength - visibleLen - suffixLen
text = pruneStringToWord(text, remaining)
result.WriteString(text)
result.WriteString(suffix)
for _, endTag := range endTokens {
result.WriteString(endTag)
}
return result.String()
}
visibleLen += len(text)
result.WriteString(text)
}
}
}
// pruneStringToWord prunes string to specified length respecting word boundaries
func pruneStringToWord(text string, maxLength int) string {
if maxLength <= 0 {
return ""
}
if len(text) <= maxLength {
return text
}
// find last space at or before maxLength to cut at word boundary
lastSpace := strings.LastIndex(text[:maxLength+1], " ")
if lastSpace <= 0 {
return ""
}
return text[:lastSpace]
}
+47
View File
@@ -0,0 +1,47 @@
package notify
import (
"testing"
"github.com/stretchr/testify/assert"
)
func TestPruneHTML(t *testing.T) {
tests := []struct {
name string
html string
maxLength int
expected string
}{
{"within limit", "<p>Hello</p>", 20, "<p>Hello</p>"},
{"exceeds limit", "<p>Hello world, this is a long text</p>", 15, "<p>Hello world,...</p>"},
{"nested tags", "<div><p>Hello world</p><p>More text</p></div>", 20, "<div><p>Hello world</p><p>More...</p></div>"},
{"html comment stripped", "<!-- comment --><p>Hello</p>", 20, "<p>Hello</p>"},
{"self-closing tag", "<p>Hello<br/>World</p>", 8, "<p>Hello<br/>...</p>"},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
assert.Equal(t, tt.expected, pruneHTML(tt.html, tt.maxLength))
})
}
}
func TestPruneStringToWord(t *testing.T) {
tests := []struct {
name string
text string
maxLength int
expected string
}{
{"within limit", "hello world", 15, "hello world"},
{"cut at word boundary", "hello world and more", 11, "hello world"},
{"zero length", "hello", 0, ""},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
assert.Equal(t, tt.expected, pruneStringToWord(tt.text, tt.maxLength))
})
}
}
+4 -2
View File
@@ -10,6 +10,8 @@ import (
"github.com/hashicorp/go-multierror"
)
const commentTextLengthLimit = 100
// TelegramParams contain settings for telegram notifications
type TelegramParams struct {
AdminChannelID string // unique identifier for the target chat or username of the target channel (in the format @channelusername)
@@ -85,10 +87,10 @@ func (t *Telegram) buildMessage(req Request) string {
msg += fmt.Sprintf(" -> <a href=%q>%s</a>", commentURLPrefix+req.parent.ID, ntf.EscapeTelegramText(req.parent.User.Name))
}
msg += fmt.Sprintf("\n\n%s", ntf.TelegramSupportedHTML(req.Comment.Text))
msg += fmt.Sprintf("\n\n%s", pruneHTML(ntf.TelegramSupportedHTML(req.Comment.Text), commentTextLengthLimit))
if req.Comment.ParentID != "" {
msg += fmt.Sprintf("\n\n\"<i>%s</i>\"", ntf.TelegramSupportedHTML(req.parent.Text))
msg += fmt.Sprintf("\n\n\"<i>%s</i>\"", pruneHTML(ntf.TelegramSupportedHTML(req.parent.Text), commentTextLengthLimit))
}
if req.Comment.PostTitle != "" {
+9
View File
@@ -53,6 +53,15 @@ some text
<b>Hello</b><i><b>World</b></i>`,
res)
// prune string keeping HTML closing tags
c = store.Comment{
Text: "<b>Lorem ipsum <i>dolor sit amet</i>, consectetur adipiscing <code>elit, sed do eiusmod tempor incididunt</code> ut labore et dolore magna aliqua.</b>",
}
res = tb.buildMessage(Request{Comment: c})
assert.Equal(t, `<a href="#remark42__comment-"></a>
<b>Lorem ipsum <i>dolor sit amet</i>, consectetur adipiscing <code>elit, sed do eiusmod tempor incididunt</code> ut...</b>`, res)
}
func TestTelegram_SendVerification(t *testing.T) {