Files
tweetdistributor/main.go
sirrow d794f65673 make links in bluesky posts clickable
Bluesky linkifies nothing by itself. A URL in the text of a record is
plain text unless a richtext facet says which bytes of the post are a
link and where they point, so our posts carried URLs no one could press.

Each link now travels on the Post as an output.Link with the offsets of
the text it occupies, and the bluesky output turns those into
app.bsky.richtext.facet#link. The offsets are UTF-8 byte offsets, not
character counts: a facet measured in characters slides off the URL as
soon as any Japanese text precedes it, and underlines the wrong words.

Scanning moved from preview.go to findLinks in main.go, which now
reports every link with its offsets rather than just the first one; the
preview card still goes to the first, which is the one a reader meets
first. X is unaffected, as it linkifies URLs itself.

The facet feature only gains its lexicon type when marshalled, so the
test checks the encoded record rather than the struct.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-03 09:35:59 +09:00

339 lines
9.5 KiB
Go

package main
import (
"bytes"
"fmt"
"image"
_ "image/gif"
"image/jpeg"
_ "image/png"
"io"
"mime"
"net/http"
"net/url"
"os"
"path"
"regexp"
"strings"
"time"
"tweetdistributor/discord"
"tweetdistributor/output"
"tweetdistributor/store"
"unicode/utf8"
"github.com/bwmarrin/discordgo"
"golang.org/x/image/draw"
_ "golang.org/x/image/webp"
)
const maxTweetLength = 140
const maxImagesPerPost = 4
// Bluesky rejects blobs over 2,000,000 bytes; Twitter allows up to 5MB.
const maxImageBytes = 2_000_000
// urlLength is what a link costs against maxTweetLength: Twitter rewrites
// every URL to a t.co address of this fixed length, however long the original
// was, so counting the raw characters would reject messages it would accept.
const urlLength = 23
// urlPattern picks candidate links out of message text; each one is parsed
// properly before it is judged to be a YouTube link.
var urlPattern = regexp.MustCompile(`https?://[^\s<>"']+`)
// trimURL drops the punctuation that ends the sentence rather than the URL.
func trimURL(match string) string {
return strings.TrimRight(match, ".,!?、。)]}>")
}
// findLinks returns every link in content, each with the UTF-8 byte offsets
// of the text it occupies.
func findLinks(content string) []output.Link {
var links []output.Link
for _, span := range urlPattern.FindAllStringIndex(content, -1) {
raw := trimURL(content[span[0]:span[1]])
if u, err := url.Parse(raw); err != nil || u.Host == "" {
continue
}
links = append(links, output.Link{
URL: raw,
ByteStart: span[0],
ByteEnd: span[0] + len(raw),
})
}
return links
}
// tweetLength counts a message the way Twitter does, charging every link a
// fixed length instead of its actual one.
func tweetLength(content string) int {
length := utf8.RuneCountInString(content)
for _, match := range urlPattern.FindAllString(content, -1) {
// The trimmed punctuation is still ordinary text and keeps counting.
length += urlLength - utf8.RuneCountInString(trimURL(match))
}
return length
}
// maxDownloadBytes caps an image download. Discord's own attachment limit is
// well below it, so a truncated image means something else served us a body
// far larger than any picture we would want to post.
const maxDownloadBytes = 32 << 20
// userAgent names the bot to the sites whose preview tags it reads; some of
// them serve those tags only to a client that identifies itself.
const userAgent = "tweetdistributor/1.0 (link preview)"
var httpClient = &http.Client{Timeout: 30 * time.Second}
// fetched is a downloaded document together with what the response said
// about it.
type fetched struct {
body []byte
// mediaType is the Content-Type without its parameters, e.g. "text/html".
mediaType string
// contentType is the header as sent, parameters and all, which is what
// tells a decoder the character encoding.
contentType string
// url is where the body actually came from, after any redirects, and is
// what relative links in it resolve against.
url *url.URL
}
// shrinkImage re-encodes (and if necessary downscales) an image until it
// fits within maxImageBytes. Images already small enough pass through
// untouched.
func shrinkImage(img output.Image) (output.Image, error) {
if len(img.Data) <= maxImageBytes {
return img, nil
}
src, _, err := image.Decode(bytes.NewReader(img.Data))
if err != nil {
return output.Image{}, fmt.Errorf("decoding %s: %w", img.Filename, err)
}
for scale := 1.0; scale > 0.05; scale *= 0.7 {
width := int(float64(src.Bounds().Dx()) * scale)
height := int(float64(src.Bounds().Dy()) * scale)
if width < 1 || height < 1 {
break
}
scaled := image.NewRGBA(image.Rect(0, 0, width, height))
draw.CatmullRom.Scale(scaled, scaled.Bounds(), src, src.Bounds(), draw.Src, nil)
var buf bytes.Buffer
if err := jpeg.Encode(&buf, scaled, &jpeg.Options{Quality: 85}); err != nil {
return output.Image{}, fmt.Errorf("encoding %s: %w", img.Filename, err)
}
if buf.Len() <= maxImageBytes {
filename := strings.TrimSuffix(img.Filename, path.Ext(img.Filename)) + ".jpg"
return output.Image{
Data: buf.Bytes(),
ContentType: "image/jpeg",
Filename: filename,
}, nil
}
}
return output.Image{}, fmt.Errorf("%s could not be shrunk below %d bytes", img.Filename, maxImageBytes)
}
// fetch GETs rawurl, reading at most limit bytes of the body. Callers that
// only need the beginning of a document pass a small limit and treat the
// truncation as normal.
func fetch(rawurl string, limit int64) (*fetched, error) {
req, err := http.NewRequest(http.MethodGet, rawurl, nil)
if err != nil {
return nil, err
}
req.Header.Set("User-Agent", userAgent)
resp, err := httpClient.Do(req)
if err != nil {
return nil, err
}
defer resp.Body.Close()
data, err := io.ReadAll(io.LimitReader(resp.Body, limit))
if err != nil {
return nil, err
}
if resp.StatusCode != http.StatusOK {
return nil, fmt.Errorf("status %s", resp.Status)
}
contenttype := resp.Header.Get("Content-Type")
mediatype, _, err := mime.ParseMediaType(contenttype)
if err != nil {
mediatype = ""
}
return &fetched{
body: data,
mediaType: mediatype,
contentType: contenttype,
url: resp.Request.URL,
}, nil
}
// downloadImage fetches an image and shrinks it to a postable size.
func downloadImage(url, filename, contentType string) (output.Image, error) {
got, err := fetch(url, maxDownloadBytes)
if err != nil {
return output.Image{}, fmt.Errorf("downloading %s: %w", filename, err)
}
return shrinkImage(output.Image{
Data: got.body,
ContentType: contentType,
Filename: filename,
})
}
func downloadImages(attachments []*discordgo.MessageAttachment) ([]output.Image, error) {
var images []output.Image
for _, attachment := range attachments {
if !strings.HasPrefix(attachment.ContentType, "image/") {
continue
}
img, err := downloadImage(attachment.URL, attachment.Filename, attachment.ContentType)
if err != nil {
return nil, err
}
images = append(images, img)
}
return images, nil
}
// distributor mirrors what happens on the Discord channel to every output.
type distributor struct {
d *discord.Client
outputs []output.OutputInterface
store *store.Store
}
// reportf logs an error and echoes it back into the Discord channel.
func (dist *distributor) reportf(format string, args ...any) {
errstr := fmt.Sprintf(format, args...)
fmt.Fprintln(os.Stderr, errstr)
dist.d.Write(errstr)
}
// created posts a new Discord message to every output, as a reply when the
// Discord message itself was a reply to something we already distributed.
func (dist *distributor) created(event discord.Event) {
if length := tweetLength(event.Content); length > maxTweetLength {
dist.reportf("Error: message is %d characters counting each link as %d, exceeding the %d character limit; not posted", length, urlLength, maxTweetLength)
return
}
images, err := downloadImages(event.Attachments)
if err != nil {
dist.reportf("Error: %s; not posted", err)
return
}
if len(images) > maxImagesPerPost {
dist.reportf("Error: %d images attached, exceeding the limit of %d; not posted", len(images), maxImagesPerPost)
return
}
// The card goes to the first link, the one a reader meets first.
links := findLinks(event.Content)
var preview *output.Preview
if len(links) > 0 {
preview, err = linkPreview(links[0].URL)
if err != nil {
// The post is still worth making without its card.
fmt.Fprintln(os.Stderr, err)
}
}
// A reply to a message we never distributed becomes a top level post.
var parents store.Refs
if event.ReplyToID != "" {
parents, _ = dist.store.Get(event.ReplyToID)
}
refs := store.Refs{}
for _, out := range dist.outputs {
post := output.Post{
Text: event.Content,
Images: images,
Links: links,
Preview: preview,
}
if parent, ok := parents[out.GetName()]; ok && !parent.IsZero() {
post.ReplyTo = &parent
}
ref, err := out.Write(post)
if err != nil {
dist.reportf("%s Error: %s", out.GetName(), err)
continue
}
refs[out.GetName()] = ref
}
if len(refs) == 0 {
return
}
if err := dist.store.Put(event.MessageID, refs); err != nil {
dist.reportf("Error: could not remember the posts for this message: %s", err)
}
}
// deleted removes the posts that a now deleted Discord message produced.
func (dist *distributor) deleted(event discord.Event) {
refs, ok := dist.store.Get(event.MessageID)
if !ok {
return
}
for _, out := range dist.outputs {
ref, ok := refs[out.GetName()]
if !ok {
continue
}
if err := out.Delete(ref); err != nil {
dist.reportf("%s Error: could not delete the post: %s", out.GetName(), err)
}
}
if err := dist.store.Delete(event.MessageID); err != nil {
dist.reportf("Error: could not forget the posts for this message: %s", err)
}
}
func main() {
posts, err := store.New(os.Getenv("POST_STORE"))
if err != nil {
fmt.Fprintln(os.Stderr, err)
os.Exit(1)
}
d := discord.Discord(os.Getenv("DISCORD_TOKEN"), os.Getenv("DISCORD_CHANNEL"))
eventchannel := make(chan discord.Event, 1)
d.BeginRead(eventchannel)
d.Write("Tweetdistributor Started")
var outputs []output.OutputInterface
outputs = append(outputs, output.StdOutput())
outputs = append(outputs, output.TwitterOutput(os.Getenv("TW_ACCESS_TOKEN"), os.Getenv("TW_ACCESS_SECRET")))
outputs = append(outputs, output.BlueskyOutput(os.Getenv("BSKY_IDENTIFIER"), os.Getenv("BSKY_PASSWORD")))
dist := &distributor{d: d, outputs: outputs, store: posts}
for event := range eventchannel {
switch event.Kind {
case discord.MessageCreated:
dist.created(event)
case discord.MessageDeleted:
dist.deleted(event)
}
}
}