2026-05-25 12:02:18 -04:00
|
|
|
package news
|
|
|
|
|
|
|
|
|
|
import (
|
|
|
|
|
"context"
|
|
|
|
|
"encoding/xml"
|
|
|
|
|
"fmt"
|
|
|
|
|
"io"
|
|
|
|
|
"net/http"
|
|
|
|
|
"net/url"
|
|
|
|
|
"sort"
|
|
|
|
|
"strings"
|
|
|
|
|
"time"
|
2026-05-25 15:16:00 -04:00
|
|
|
|
|
|
|
|
"golang.org/x/net/html"
|
2026-05-25 12:02:18 -04:00
|
|
|
)
|
|
|
|
|
|
|
|
|
|
const UserAgent = "box-box/phase-19b-rss-spike"
|
|
|
|
|
|
|
|
|
|
// Source describes a feed that can be fetched and normalized.
|
|
|
|
|
type Source struct {
|
2026-05-25 15:16:00 -04:00
|
|
|
ID string
|
|
|
|
|
Name string
|
|
|
|
|
URL string
|
|
|
|
|
Category string
|
|
|
|
|
SkipSummary bool // omit summary for sources where it's useless (e.g. YouTube)
|
2026-05-25 12:02:18 -04:00
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Item is the normalized shape consumed by storage and API layers.
|
|
|
|
|
type Item struct {
|
2026-05-25 15:16:00 -04:00
|
|
|
Source string
|
|
|
|
|
Title string
|
|
|
|
|
URL string
|
|
|
|
|
PublishedAt time.Time
|
|
|
|
|
Summary string
|
|
|
|
|
Category string
|
|
|
|
|
FetchedAt time.Time
|
|
|
|
|
OGImageURL string
|
|
|
|
|
OGDescription string
|
2026-05-25 12:02:18 -04:00
|
|
|
}
|
|
|
|
|
|
2026-05-25 15:16:00 -04:00
|
|
|
// DefaultSources are free RSS/Atom feeds for the Paddock Briefing.
|
2026-05-25 12:02:18 -04:00
|
|
|
var DefaultSources = []Source{
|
|
|
|
|
{ID: "fia", Name: "FIA", URL: "https://www.fia.com/rss/news", Category: "official"},
|
|
|
|
|
{ID: "bbc-f1", Name: "BBC Sport F1", URL: "https://feeds.bbci.co.uk/sport/formula1", Category: "news"},
|
|
|
|
|
{ID: "autosport-f1", Name: "Autosport F1", URL: "https://www.autosport.com/rss/f1/news/", Category: "news"},
|
|
|
|
|
{ID: "racefans-f1", Name: "RaceFans F1", URL: "https://www.racefans.net/category/f1-news/feed/", Category: "news"},
|
|
|
|
|
{ID: "guardian-f1", Name: "Guardian Formula One", URL: "https://www.theguardian.com/sport/formulaone/rss", Category: "news"},
|
|
|
|
|
{ID: "racer-f1", Name: "RACER F1", URL: "https://racer.com/f1/feed", Category: "news"},
|
2026-05-25 15:16:00 -04:00
|
|
|
{ID: "f1-youtube", Name: "Formula 1 YouTube", URL: "https://www.youtube.com/feeds/videos.xml?channel_id=UCB_qr75-ydFVKSF9Dmo6izg", Category: "video", SkipSummary: true},
|
2026-05-25 12:02:18 -04:00
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Fetch retrieves and parses one RSS or Atom feed with the provided HTTP client.
|
|
|
|
|
func Fetch(ctx context.Context, client *http.Client, source Source) ([]Item, error) {
|
|
|
|
|
if client == nil {
|
|
|
|
|
client = &http.Client{Timeout: 10 * time.Second}
|
|
|
|
|
}
|
|
|
|
|
req, err := http.NewRequestWithContext(ctx, http.MethodGet, source.URL, nil)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
req.Header.Set("User-Agent", UserAgent)
|
|
|
|
|
|
|
|
|
|
resp, err := client.Do(req)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
defer resp.Body.Close()
|
|
|
|
|
if resp.StatusCode < 200 || resp.StatusCode >= 300 {
|
|
|
|
|
return nil, fmt.Errorf("fetch %s: status %d", source.ID, resp.StatusCode)
|
|
|
|
|
}
|
|
|
|
|
return Parse(source, resp.Body, time.Now().UTC())
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Parse normalizes RSS 2.0 or Atom XML into Items.
|
|
|
|
|
func Parse(source Source, r io.Reader, fetchedAt time.Time) ([]Item, error) {
|
|
|
|
|
payload, err := io.ReadAll(io.LimitReader(r, 2<<20))
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
var rss rssFeed
|
|
|
|
|
if err := xml.Unmarshal(payload, &rss); err == nil && len(rss.Channel.Items) > 0 {
|
|
|
|
|
return normalizeRSS(source, rss.Channel.Items, fetchedAt), nil
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
var atom atomFeed
|
|
|
|
|
if err := xml.Unmarshal(payload, &atom); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
if len(atom.Entries) == 0 {
|
|
|
|
|
return nil, fmt.Errorf("parse %s: no RSS items or Atom entries found", source.ID)
|
|
|
|
|
}
|
|
|
|
|
return normalizeAtom(source, atom.Entries, fetchedAt), nil
|
|
|
|
|
}
|
|
|
|
|
|
2026-05-25 15:16:00 -04:00
|
|
|
// FetchOGMeta fetches only the <head> of a page and extracts og:image and og:description.
|
|
|
|
|
// It reads at most 64 KB to avoid full-page downloads.
|
|
|
|
|
func FetchOGMeta(ctx context.Context, client *http.Client, rawURL string) (imageURL, description string, err error) {
|
|
|
|
|
if client == nil {
|
|
|
|
|
client = &http.Client{Timeout: 8 * time.Second}
|
|
|
|
|
}
|
|
|
|
|
req, err := http.NewRequestWithContext(ctx, http.MethodGet, rawURL, nil)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return "", "", err
|
|
|
|
|
}
|
|
|
|
|
req.Header.Set("User-Agent", UserAgent)
|
|
|
|
|
|
|
|
|
|
resp, err := client.Do(req)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return "", "", err
|
|
|
|
|
}
|
|
|
|
|
defer resp.Body.Close()
|
|
|
|
|
if resp.StatusCode < 200 || resp.StatusCode >= 300 {
|
|
|
|
|
return "", "", fmt.Errorf("og fetch %s: status %d", rawURL, resp.StatusCode)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
limited := io.LimitReader(resp.Body, 64<<10)
|
|
|
|
|
tokenizer := html.NewTokenizer(limited)
|
|
|
|
|
|
|
|
|
|
for {
|
|
|
|
|
tt := tokenizer.Next()
|
|
|
|
|
if tt == html.ErrorToken {
|
|
|
|
|
break
|
|
|
|
|
}
|
|
|
|
|
if tt == html.EndTagToken {
|
|
|
|
|
tag, _ := tokenizer.TagName()
|
|
|
|
|
if string(tag) == "head" {
|
|
|
|
|
break
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
if tt != html.SelfClosingTagToken && tt != html.StartTagToken {
|
|
|
|
|
continue
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
tag, hasAttr := tokenizer.TagName()
|
|
|
|
|
if !hasAttr || string(tag) != "meta" {
|
|
|
|
|
continue
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
var property, name, content string
|
|
|
|
|
for {
|
|
|
|
|
k, v, more := tokenizer.TagAttr()
|
|
|
|
|
switch string(k) {
|
|
|
|
|
case "property":
|
|
|
|
|
property = string(v)
|
|
|
|
|
case "name":
|
|
|
|
|
name = string(v)
|
|
|
|
|
case "content":
|
|
|
|
|
content = string(v)
|
|
|
|
|
}
|
|
|
|
|
if !more {
|
|
|
|
|
break
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
switch {
|
|
|
|
|
case property == "og:image" && imageURL == "":
|
|
|
|
|
imageURL = strings.TrimSpace(content)
|
|
|
|
|
case property == "og:description" && description == "":
|
|
|
|
|
description = stripHTML(content)
|
|
|
|
|
case name == "description" && description == "":
|
|
|
|
|
description = stripHTML(content)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if imageURL != "" && description != "" {
|
|
|
|
|
break
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
return imageURL, description, nil
|
|
|
|
|
}
|
|
|
|
|
|
2026-05-25 12:02:18 -04:00
|
|
|
// DeduplicateByURL keeps the newest instance of each canonical URL.
|
|
|
|
|
func DeduplicateByURL(items []Item) []Item {
|
|
|
|
|
byURL := make(map[string]Item, len(items))
|
|
|
|
|
for _, item := range items {
|
|
|
|
|
key := canonicalURL(item.URL)
|
|
|
|
|
if key == "" {
|
|
|
|
|
continue
|
|
|
|
|
}
|
|
|
|
|
item.URL = key
|
|
|
|
|
if existing, ok := byURL[key]; !ok || item.PublishedAt.After(existing.PublishedAt) {
|
|
|
|
|
byURL[key] = item
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
out := make([]Item, 0, len(byURL))
|
|
|
|
|
for _, item := range byURL {
|
|
|
|
|
out = append(out, item)
|
|
|
|
|
}
|
|
|
|
|
sort.Slice(out, func(i, j int) bool {
|
|
|
|
|
return out[i].PublishedAt.After(out[j].PublishedAt)
|
|
|
|
|
})
|
|
|
|
|
return out
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
type rssFeed struct {
|
|
|
|
|
Channel struct {
|
|
|
|
|
Items []rssItem `xml:"item"`
|
|
|
|
|
} `xml:"channel"`
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
type rssItem struct {
|
|
|
|
|
Title string `xml:"title"`
|
|
|
|
|
Link string `xml:"link"`
|
|
|
|
|
GUID string `xml:"guid"`
|
|
|
|
|
PubDate string `xml:"pubDate"`
|
|
|
|
|
Description string `xml:"description"`
|
|
|
|
|
Categories []string `xml:"category"`
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
type atomFeed struct {
|
|
|
|
|
Entries []atomEntry `xml:"entry"`
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
type atomEntry struct {
|
|
|
|
|
Title string `xml:"title"`
|
|
|
|
|
ID string `xml:"id"`
|
|
|
|
|
Updated string `xml:"updated"`
|
|
|
|
|
Published string `xml:"published"`
|
|
|
|
|
Summary string `xml:"summary"`
|
|
|
|
|
Content string `xml:"content"`
|
|
|
|
|
Links []atomLink `xml:"link"`
|
|
|
|
|
Categories []struct {
|
|
|
|
|
Term string `xml:"term,attr"`
|
|
|
|
|
Label string `xml:"label,attr"`
|
|
|
|
|
} `xml:"category"`
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
type atomLink struct {
|
|
|
|
|
Href string `xml:"href,attr"`
|
|
|
|
|
Rel string `xml:"rel,attr"`
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
func normalizeRSS(source Source, raw []rssItem, fetchedAt time.Time) []Item {
|
|
|
|
|
items := make([]Item, 0, len(raw))
|
|
|
|
|
for _, entry := range raw {
|
|
|
|
|
link := strings.TrimSpace(entry.Link)
|
|
|
|
|
if link == "" {
|
|
|
|
|
link = strings.TrimSpace(entry.GUID)
|
|
|
|
|
}
|
2026-05-25 15:16:00 -04:00
|
|
|
summary := ""
|
|
|
|
|
if !source.SkipSummary {
|
|
|
|
|
summary = stripHTML(entry.Description)
|
|
|
|
|
}
|
2026-05-25 12:02:18 -04:00
|
|
|
items = append(items, Item{
|
|
|
|
|
Source: source.ID,
|
|
|
|
|
Title: cleanText(entry.Title),
|
|
|
|
|
URL: link,
|
|
|
|
|
PublishedAt: parseFeedTime(entry.PubDate),
|
2026-05-25 15:16:00 -04:00
|
|
|
Summary: summary,
|
2026-05-25 12:02:18 -04:00
|
|
|
Category: firstNonEmpty(entry.Categories, source.Category),
|
|
|
|
|
FetchedAt: fetchedAt,
|
|
|
|
|
})
|
|
|
|
|
}
|
|
|
|
|
return DeduplicateByURL(items)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
func normalizeAtom(source Source, raw []atomEntry, fetchedAt time.Time) []Item {
|
|
|
|
|
items := make([]Item, 0, len(raw))
|
|
|
|
|
for _, entry := range raw {
|
|
|
|
|
category := source.Category
|
|
|
|
|
if len(entry.Categories) > 0 {
|
|
|
|
|
category = firstNonEmpty([]string{entry.Categories[0].Label, entry.Categories[0].Term}, source.Category)
|
|
|
|
|
}
|
2026-05-25 15:16:00 -04:00
|
|
|
summary := ""
|
|
|
|
|
if !source.SkipSummary {
|
|
|
|
|
raw := firstNonEmpty([]string{entry.Summary, entry.Content}, "")
|
|
|
|
|
summary = stripHTML(raw)
|
|
|
|
|
}
|
2026-05-25 12:02:18 -04:00
|
|
|
items = append(items, Item{
|
|
|
|
|
Source: source.ID,
|
|
|
|
|
Title: cleanText(entry.Title),
|
|
|
|
|
URL: atomEntryURL(entry),
|
|
|
|
|
PublishedAt: parseFeedTime(firstNonEmpty([]string{entry.Published, entry.Updated}, "")),
|
2026-05-25 15:16:00 -04:00
|
|
|
Summary: summary,
|
2026-05-25 12:02:18 -04:00
|
|
|
Category: category,
|
|
|
|
|
FetchedAt: fetchedAt,
|
|
|
|
|
})
|
|
|
|
|
}
|
|
|
|
|
return DeduplicateByURL(items)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
func atomEntryURL(entry atomEntry) string {
|
|
|
|
|
for _, link := range entry.Links {
|
|
|
|
|
if link.Rel == "" || link.Rel == "alternate" {
|
|
|
|
|
return strings.TrimSpace(link.Href)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
if len(entry.Links) > 0 {
|
|
|
|
|
return strings.TrimSpace(entry.Links[0].Href)
|
|
|
|
|
}
|
|
|
|
|
return strings.TrimSpace(entry.ID)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
func parseFeedTime(value string) time.Time {
|
|
|
|
|
value = strings.TrimSpace(value)
|
|
|
|
|
if value == "" {
|
|
|
|
|
return time.Time{}
|
|
|
|
|
}
|
|
|
|
|
layouts := []string{
|
|
|
|
|
time.RFC1123Z,
|
|
|
|
|
time.RFC1123,
|
|
|
|
|
time.RFC3339,
|
|
|
|
|
time.RFC3339Nano,
|
|
|
|
|
"Mon, 02 Jan 2006 15:04:05 -0700",
|
|
|
|
|
"Mon, 2 Jan 2006 15:04:05 -0700",
|
|
|
|
|
}
|
|
|
|
|
for _, layout := range layouts {
|
|
|
|
|
if ts, err := time.Parse(layout, value); err == nil {
|
|
|
|
|
return ts.UTC()
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
return time.Time{}
|
|
|
|
|
}
|
|
|
|
|
|
2026-05-25 15:16:00 -04:00
|
|
|
// stripHTML removes HTML tags and decodes entities, inserting line breaks for
|
|
|
|
|
// block-level elements so the resulting text remains readable as prose.
|
|
|
|
|
func stripHTML(value string) string {
|
|
|
|
|
if value == "" {
|
|
|
|
|
return ""
|
|
|
|
|
}
|
|
|
|
|
tokenizer := html.NewTokenizer(strings.NewReader(value))
|
|
|
|
|
var b strings.Builder
|
|
|
|
|
for {
|
|
|
|
|
tt := tokenizer.Next()
|
|
|
|
|
if tt == html.ErrorToken {
|
|
|
|
|
break
|
|
|
|
|
}
|
|
|
|
|
switch tt {
|
|
|
|
|
case html.TextToken:
|
|
|
|
|
b.WriteString(tokenizer.Token().Data)
|
|
|
|
|
case html.StartTagToken, html.EndTagToken, html.SelfClosingTagToken:
|
|
|
|
|
tag, _ := tokenizer.TagName()
|
|
|
|
|
switch string(tag) {
|
|
|
|
|
case "p", "br", "li", "h1", "h2", "h3", "h4", "h5", "h6", "div", "blockquote", "tr":
|
|
|
|
|
b.WriteByte('\n')
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
return cleanText(b.String())
|
|
|
|
|
}
|
|
|
|
|
|
2026-05-25 12:02:18 -04:00
|
|
|
func cleanText(value string) string {
|
|
|
|
|
value = strings.TrimSpace(value)
|
|
|
|
|
value = strings.ReplaceAll(value, "\n", " ")
|
|
|
|
|
value = strings.ReplaceAll(value, "\t", " ")
|
|
|
|
|
return strings.Join(strings.Fields(value), " ")
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
func firstNonEmpty(values []string, fallback string) string {
|
|
|
|
|
for _, value := range values {
|
|
|
|
|
if strings.TrimSpace(value) != "" {
|
|
|
|
|
return strings.TrimSpace(value)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
return fallback
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
func canonicalURL(raw string) string {
|
|
|
|
|
raw = strings.TrimSpace(raw)
|
|
|
|
|
if raw == "" {
|
|
|
|
|
return ""
|
|
|
|
|
}
|
|
|
|
|
parsed, err := url.Parse(raw)
|
|
|
|
|
if err != nil || parsed.Scheme == "" || parsed.Host == "" {
|
|
|
|
|
return raw
|
|
|
|
|
}
|
|
|
|
|
parsed.Fragment = ""
|
|
|
|
|
q := parsed.Query()
|
|
|
|
|
for key := range q {
|
|
|
|
|
if strings.HasPrefix(strings.ToLower(key), "utm_") {
|
|
|
|
|
q.Del(key)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
parsed.RawQuery = q.Encode()
|
|
|
|
|
return parsed.String()
|
|
|
|
|
}
|