social-posting/social-feed.py
2026-07-28 09:57:31 -03:00

379 lines
No EOL
15 KiB
Python

#!/usr/bin/env python3
"""
Social Feed Posting
Date: 2026-07-28
Website: https://git.nas.jeffmackinnon.com/jeff/social-posting
Description: This script parses RSS feeds to a database and
then posts new articles to your Mastodon account.
"""
__author__ = "Jeff MacKinnon"
__license__ = "MIT"
__copyright__ = "Copyright 2026, Jeff MacKinnon"
import datetime
import io
import logging
from urllib.parse import urlparse
from bs4 import BeautifulSoup
import feedparser
from mastodon import Mastodon
import requests
from peewee import *
from playhouse.sqlite_ext import JSONField
import io
import time
import tomllib
from pathlib import Path
# --- Configuration ---
# Load a TOML file into a standard dictionary
with open("config.toml", "rb") as f:
config = tomllib.load(f)
age_limit = datetime.datetime.now() - datetime.timedelta(days=config["post_age_limit"])
MASTODON_API_BASE_URL = config["mastodon"]["base_url"]
MASTODON_ACCESS_TOKEN = config["mastodon"]["api_token"]
logging.basicConfig(level=logging.INFO, format="%(asctime)s - %(levelname)s - %(message)s")
db = SqliteDatabase('rss_reader.db')
class BaseModel(Model):
class Meta:
database = db
class Feed(BaseModel):
title = CharField()
site_url = CharField(unique=True)
feed_url = CharField(unique=True)
last_checked = DateTimeField(null=True)
class Article(BaseModel):
feed = ForeignKeyField(Feed, backref='articles')
title = CharField()
link = CharField(unique=True)
published_date = DateTimeField(null=True)
author = CharField(null=True)
summary = TextField(null=True)
content = TextField(null=True)
og_image = CharField(max_length=500, null=True)
metadata_json = JSONField(null=True)
date_added = DateTimeField(default=datetime.datetime.now)
posted_to_mastodon = BooleanField(default=False)
posted_to_instagram = BooleanField(default=False)
posted_to_bluesky = BooleanField(default=False)
db.connect()
# Run create_tables again safely to automatically add the new column if using SQLite
db.create_tables([Feed, Article], safe=True)
# --- Mastodon Integration ---
def get_mastodon_client():
"""Initializes the Mastodon client."""
return Mastodon(
access_token=MASTODON_ACCESS_TOKEN,
api_base_url=MASTODON_API_BASE_URL
)
def post_article_to_mastodon(mastodon_client, article: Article):
"""Formats an article and posts it to Mastodon with line-by-line feedback."""
try:
logging.info(f"--- Starting Mastodon Process for Article #{article.id} ---")
# 1. Format text
extra = article.metadata_json or {}
description = extra.get('og_description') or article.summary or ""
if len(description) > 200:
description = description[:197] + "..."
raw_tags = extra.get('keywords', '') or ",".join(extra.get('feed_tags', []))
hashtag_list = []
if raw_tags:
cleaned_tags = [t.strip().replace(" ", "").replace("-", "") for t in raw_tags.split(",") if t.strip()]
hashtag_list = [f"#{tag}" for tag in cleaned_tags[:5]]
hashtags_str = " ".join(hashtag_list)
status_text = f" {article.title}\n\n"
if description:
status_text += f"{description}\n\n"
status_text += f"🔗 {article.link}\n\n"
if hashtags_str:
status_text += f"{hashtags_str}"
# 2. Upload OpenGraph image safely with line-by-line feedback
media_ids = []
if article.og_image:
logging.info(f"[Step 1/4] Starting image download request: {article.og_image}")
try:
headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
# Enforce a strict 5-second connection & data read timeout
img_response = requests.get(article.og_image, headers=headers, timeout=(5, 5))
logging.info(f"[Step 2/4] Image request finished. HTTP Status Code: {img_response.status_code}")
if img_response.status_code == 200:
img_data = img_response.content
logging.info(f"[Step 3/4] Successfully read {len(img_data)} image bytes into local RAM.")
mime_type = "image/jpeg" if article.og_image.lower().endswith(('.jpg', '.jpeg')) else "image/png"
logging.info("[Step 4/4] Sending bytes to Mastodon API via client.media_post()... (Hangs here if API fails)")
media_meta = mastodon_client.media_post(
media_file=io.BytesIO(img_data),
mime_type=mime_type,
file_name=f"thumbnail_{article.id}.jpg",
description=f"Thumbnail for {article.title}",
synchronous=True # Forces the script to wait until upload completes or times out natively
)
media_ids.append(media_meta['id'])
logging.info(f"-> Mastodon image upload accepted! Received Media ID: {media_meta['id']}")
else:
logging.warning(f"-> Skipping image upload: Remote server returned status {img_response.status_code}")
except requests.exceptions.Timeout:
logging.error("-> Image download skipped: Connection timed out after 5 seconds.")
except Exception as img_err:
logging.error(f"-> Image handling system failed: {img_err}. Proceeding with text-only post.")
# 3. Publish the text payload
logging.info(f"Sending final status text payload to Mastodon timeline... (Media IDs: {media_ids})")
mastodon_client.status_post(status=status_text, media_ids=media_ids if media_ids else None)
logging.info(f"-> Successfully posted to Mastodon: {article.title}")
# 4. Save DB state
article.posted_to_mastodon = True
article.save()
logging.info(f"--- Finished Article #{article.id} ---\n")
except Exception as e:
logging.error(f"Failed to post article {article.id} to Mastodon: {e}")
# --- Core Pipeline Logic ---
def fetch_opengraph_data(url: str) -> dict:
metadata = {"og_image": None, "extra": {}}
headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
try:
response = requests.get(url, headers=headers, timeout=10)
if response.status_code != 200:
return metadata
soup = BeautifulSoup(response.text, 'html.parser')
for tag in soup.find_all('meta'):
property_attr = tag.get('property', '')
name_attr = tag.get('name', '')
content_attr = tag.get('content', '')
if not content_attr:
continue
if property_attr.startswith('og:'):
key = property_attr[3:]
if key == 'image':
metadata['og_image'] = content_attr
else:
metadata['extra'][f"og_{key}"] = content_attr
elif name_attr in ['description', 'keywords', 'author']:
metadata['extra'][name_attr] = content_attr
except Exception as e:
logging.error(f"Error scraping metadata from {url}: {e}")
return metadata
def process_rss_feed(feed_url: str):
logging.info(f"Parsing feed: {feed_url}")
headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
try:
response = requests.get(feed_url, headers=headers, timeout=15)
if response.status_code != 200:
return
parsed_feed = feedparser.parse(response.text)
except Exception as e:
return
if not parsed_feed.entries:
return
feed_title = parsed_feed.feed.get('title', 'Unknown Feed')
site_url = parsed_feed.feed.get('link', f"{urlparse(feed_url).scheme}://{urlparse(feed_url).netloc}")
feed_record, _ = Feed.get_or_create(
feed_url=feed_url,
defaults={'title': feed_title, 'site_url': site_url}
)
one_month_ago = datetime.datetime.now() - datetime.timedelta(days=30)
new_articles = []
with db.atomic():
for entry in parsed_feed.entries:
link = entry.get('link')
if not link or Article.select().where(Article.link == link).exists():
continue
logging.info(f"New article: {link}. Scraping OpenGraph...")
# 1. Safely extract the article publication date
pub_date = None
time_struct = entry.get('published_parsed') or entry.get('updated_parsed') or entry.get('created_parsed')
if time_struct:
try:
pub_date = datetime.datetime(*time_struct[:6])
except (ValueError, TypeError):
pub_date = datetime.datetime.now()
else:
pub_date = datetime.datetime.now()
# 2. Check the article's age BEFORE processing
# If the article is older than 30 days, we treat it as already posted
is_too_old = pub_date < age_limit
if is_too_old:
logging.info(f"-> Article is older than 1 month ({pub_date.date()}). Skipping social queues.")
# We flip the flags to True so social media functions completely ignore it
already_posted_state = True
else:
already_posted_state = False
# Scrape web page metadata
web_metadata = fetch_opengraph_data(link)
author = entry.get('author') or web_metadata['extra'].get('author')
summary = entry.get('summary') or web_metadata['extra'].get('og_description')
extra_metadata = web_metadata['extra']
if 'tags' in entry:
extra_metadata['feed_tags'] = [tag.get('term') for tag in entry.tags]
# 3. Save to the database with conditional flags
art = Article.create(
feed=feed_record,
title=entry.get('title', 'Untitled'),
link=link,
published_date=pub_date,
author=author,
summary=summary,
content=entry.get('description'),
og_image=web_metadata['og_image'],
metadata_json=extra_metadata,
# FIXED: Old posts are born marked as True, new posts are born False
posted_to_mastodon=already_posted_state,
posted_to_instagram=already_posted_state,
posted_to_bluesky=already_posted_state
)
# Only add to the execution list if it actually needs to be published
if not already_posted_state:
new_articles.append(art)
feed_record.last_checked = datetime.datetime.now()
feed_record.save()
# Return any newly added records so the runner knows what needs posting
return new_articles
def retry_failed_mastodon_posts(mastodon_client):
"""
Finds articles younger than 30 days that failed to post,
and publishes them with a 15-minute delay between posts.
"""
logging.info("Checking database for older articles that failed to post to Mastodon...")
# Calculate the 30-day lookback threshold window
one_month_ago = datetime.datetime.now() - datetime.timedelta(days=30)
# Query database for matching failed items, ordered oldest to newest
failed_articles = (Article
.select()
.where(
(Article.posted_to_mastodon == False) &
(Article.date_added >= one_month_ago)
)
.order_by(Article.date_added.asc()))
count = failed_articles.count()
if count == 0:
logging.info("No failed posts found within the 1-month window.")
return
logging.info(f"Found {count} articles waiting to be retried. Processing backlog...")
for idx, article in enumerate(failed_articles, start=1):
logging.info(f"Retrying backlog item ({idx}/{count}): {article.title}")
# Attempt to publish the post
post_article_to_mastodon(mastodon_client, article)
# Refresh row to see if it successfully flipped to True
article_refresh = Article.get_by_id(article.id)
# Only enforce the 15-minute delay if the post was successful AND there are more items remaining
if article_refresh.posted_to_mastodon and idx < count:
logging.info("Post successful. Enforcing a 15-minute rate limit cooldown step...")
# 15 minutes = 15 * 60 seconds = 900 seconds
time.sleep(900)
'''
# --- Execution Example ---
if __name__ == "__main__":
target_feeds = [
"https://jeffmackinnon.com/feeds/all.rss.xml"
]
# Initialize Mastodon API connection
m_client = get_mastodon_client()
for feed_url in target_feeds:
try:
# 1. Parse and scrape the feeds
new_posts = process_rss_feed(feed_url)
# 2. Cycle through only the newly added posts and cross-post them
if new_posts:
logging.info(f"Found {len(new_posts)} new items to share to Mastodon.")
for article in new_posts:
post_article_to_mastodon(m_client, article)
except Exception as e:
logging.error(f"Error processing loop for {feed_url}: {e}")
'''
if __name__ == "__main__":
target_feeds = config["rss_feeds"]
m_client = get_mastodon_client()
# Track if the primary crawl cycle ran smoothly
feed_crawl_successful = True
for feed_url in target_feeds:
try:
new_posts = process_rss_feed(feed_url)
if new_posts:
logging.info(f"Found {len(new_posts)} new items to share.")
for article in new_posts:
post_article_to_mastodon(m_client, article)
except Exception as e:
logging.error(f"Critical pipeline failure on {feed_url}: {e}")
feed_crawl_successful = False
# Trigger the retry backlog recovery routine if a crawl failed
# OR run it unconditionally every time to clean up historical failures
if not feed_crawl_successful:
logging.warning("Pipeline encountered errors during the crawl. Launching recovery runner...")
retry_failed_mastodon_posts(m_client)
else:
# Optional: Run it anyway just in case old network drops left orphan records behind
retry_failed_mastodon_posts(m_client)