Files
weatherfeeder/internal/providers/spc/discussion.go
Eric Rakestraw a990da957b
All checks were successful
ci/woodpecker/push/build-image Pipeline was successful
Finalize SPC outlook feature addition and clean up implemented roadmap documentation
2026-06-10 19:54:39 -05:00

189 lines
5.2 KiB
Go

package spc
import (
"fmt"
"html"
"regexp"
"strings"
"time"
)
var (
scriptBlockRE = regexp.MustCompile(`(?is)<script\b[^>]*>.*?</script>`)
preBlockRE = regexp.MustCompile(`(?is)<pre\b[^>]*>(.*?)</pre>`)
tagRE = regexp.MustCompile(`(?is)<[^>]+>`)
updatedRE = regexp.MustCompile(`(?im)^\s*Updated:\s*(.+?)\s*$`)
pageUpdatedRE = regexp.MustCompile(`(?i)\bUpdated:\s*((?:\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}Z)|(?:[A-Z][a-z]{2}\s+[A-Z][a-z]{2}\s+\d{1,2}\s+\d{2}:\d{2}:\d{2}\s+UTC\s+\d{4})|(?:\d{4}\s+UTC\s+[A-Z][a-z]{2}\s+[A-Z][a-z]{2}\s+\d{1,2}\s+\d{4})|(?:\d{4}Z\s+[A-Z][a-z]{2}\s+[A-Z][a-z]{2}\s+\d{1,2}\s+\d{4}))`)
productCodeRE = regexp.MustCompile(`(?i)^SPC\s+AC\s+\d+\s*$`)
sectionRE = regexp.MustCompile(`^\s*\.\.\.[A-Z0-9 /-]+\.{3}\s*$`)
)
// DiscussionText contains parsed text from an SPC print page.
type DiscussionText struct {
ProductTitle string
Headline string
Summary string
Discussion string
UpdatedAt *time.Time
}
// ExtractProductText extracts and cleans the first useful preformatted SPC
// product text block from a print page.
func ExtractProductText(rawHTML string) (string, error) {
matches := preBlockRE.FindAllStringSubmatch(rawHTML, -1)
for _, match := range matches {
if len(match) < 2 {
continue
}
text := cleanHTMLText(match[1])
if strings.TrimSpace(text) != "" {
return text, nil
}
}
return "", fmt.Errorf("no useful pre block found")
}
// ParseDiscussionHTML extracts SPC product text and page-level metadata from a
// print-page HTML document.
func ParseDiscussionHTML(rawHTML string) (DiscussionText, error) {
text, err := ExtractProductText(rawHTML)
if err != nil {
return DiscussionText{}, err
}
parsed := ParseDiscussionText(text)
if updatedAt := ParsePageUpdatedTimestamp(rawHTML); updatedAt != nil {
parsed.UpdatedAt = updatedAt
}
return parsed, nil
}
// ParseDiscussionText extracts common SPC narrative metadata from cleaned
// product text.
func ParseDiscussionText(text string) DiscussionText {
text = trimBlankLines(normalizeNewlines(text))
title := ParseProductTitle(text)
return DiscussionText{
ProductTitle: title,
Headline: title,
Summary: ExtractSummary(text),
Discussion: text,
UpdatedAt: ParseUpdatedTimestamp(text),
}
}
// ParsePageUpdatedTimestamp parses the page-level Updated row from an SPC print
// page. SPC currently places this outside the product <pre> block.
func ParsePageUpdatedTimestamp(rawHTML string) *time.Time {
text := cleanHTMLText(rawHTML)
text = strings.ReplaceAll(text, "\u00a0", " ")
text = strings.Join(strings.Fields(text), " ")
match := pageUpdatedRE.FindStringSubmatch(text)
if len(match) != 2 {
return nil
}
return parseUpdatedValue(match[1])
}
// ParseUpdatedTimestamp parses an SPC print-page Updated line when present.
func ParseUpdatedTimestamp(text string) *time.Time {
match := updatedRE.FindStringSubmatch(normalizeNewlines(text))
if len(match) != 2 {
return nil
}
return parseUpdatedValue(match[1])
}
// ParseProductTitle returns the first non-empty product line from cleaned text.
func ParseProductTitle(text string) string {
for _, line := range strings.Split(normalizeNewlines(text), "\n") {
line = strings.TrimSpace(line)
if line == "" || strings.HasPrefix(line, "Updated:") || productCodeRE.MatchString(line) {
continue
}
return line
}
return ""
}
// ParseHeadline returns the human-facing headline from cleaned text.
func ParseHeadline(text string) string {
return ParseProductTitle(text)
}
// ExtractSummary returns text under the ...SUMMARY... section through the next
// SPC section heading.
func ExtractSummary(text string) string {
lines := strings.Split(normalizeNewlines(text), "\n")
start := -1
for i, line := range lines {
if strings.EqualFold(strings.TrimSpace(line), "...SUMMARY...") {
start = i + 1
break
}
}
if start < 0 {
return ""
}
var out []string
for _, line := range lines[start:] {
if sectionRE.MatchString(line) {
break
}
out = append(out, line)
}
return trimBlankLines(strings.Join(out, "\n"))
}
func cleanHTMLText(raw string) string {
raw = scriptBlockRE.ReplaceAllString(raw, "")
raw = tagRE.ReplaceAllString(raw, "")
raw = html.UnescapeString(raw)
raw = normalizeNewlines(raw)
return trimBlankLines(raw)
}
func normalizeNewlines(text string) string {
text = strings.ReplaceAll(text, "\r\n", "\n")
text = strings.ReplaceAll(text, "\r", "\n")
return text
}
func trimBlankLines(text string) string {
lines := strings.Split(normalizeNewlines(text), "\n")
start := 0
for start < len(lines) && strings.TrimSpace(lines[start]) == "" {
start++
}
end := len(lines)
for end > start && strings.TrimSpace(lines[end-1]) == "" {
end--
}
return strings.Join(lines[start:end], "\n")
}
func parseUpdatedValue(value string) *time.Time {
value = strings.TrimSpace(value)
if value == "" {
return nil
}
if t := parseOptionalISOTimestamp(value); t != nil {
return t
}
for _, layout := range []string{
"Mon Jan 2 15:04:05 UTC 2006",
"1504 UTC Mon Jan 2 2006",
"1504Z Mon Jan 2 2006",
"3:04 PM UTC Mon Jan 2 2006",
time.RFC1123,
time.RFC1123Z,
} {
t, err := time.Parse(layout, value)
if err == nil {
tt := t.UTC()
return &tt
}
}
return nil
}