#!/usr/bin/env python3 """ Social Feed Posting Date: 2026-07-28 Website: https://git.nas.jeffmackinnon.com/jeff/social-posting Description: This script parses RSS feeds to a database and then posts new articles to your Mastodon account. """ __author__ = "Jeff MacKinnon" __license__ = "MIT" __copyright__ = "Copyright 2026, Jeff MacKinnon" import datetime import io import logging from urllib.parse import urlparse from bs4 import BeautifulSoup import feedparser from mastodon import Mastodon import requests from peewee import * from playhouse.sqlite_ext import JSONField import io import time import tomllib from pathlib import Path # --- Configuration --- # Load a TOML file into a standard dictionary with open("config.toml", "rb") as f: config = tomllib.load(f) age_limit = datetime.datetime.now() - datetime.timedelta(days=config["post_age_limit"]) MASTODON_API_BASE_URL = config["mastodon"]["base_url"] MASTODON_ACCESS_TOKEN = config["mastodon"]["api_token"] logging.basicConfig(level=logging.INFO, format="%(asctime)s - %(levelname)s - %(message)s") db = SqliteDatabase('rss_reader.db') class BaseModel(Model): class Meta: database = db class Feed(BaseModel): title = CharField() site_url = CharField(unique=True) feed_url = CharField(unique=True) last_checked = DateTimeField(null=True) class Article(BaseModel): feed = ForeignKeyField(Feed, backref='articles') title = CharField() link = CharField(unique=True) published_date = DateTimeField(null=True) author = CharField(null=True) summary = TextField(null=True) content = TextField(null=True) og_image = CharField(max_length=500, null=True) metadata_json = JSONField(null=True) date_added = DateTimeField(default=datetime.datetime.now) posted_to_mastodon = BooleanField(default=False) posted_to_instagram = BooleanField(default=False) posted_to_bluesky = BooleanField(default=False) db.connect() # Run create_tables again safely to automatically add the new column if using SQLite db.create_tables([Feed, Article], safe=True) # --- Mastodon Integration --- def get_mastodon_client(): """Initializes the Mastodon client.""" return Mastodon( access_token=MASTODON_ACCESS_TOKEN, api_base_url=MASTODON_API_BASE_URL ) def post_article_to_mastodon(mastodon_client, article: Article): """Formats an article and posts it to Mastodon with line-by-line feedback.""" try: logging.info(f"--- Starting Mastodon Process for Article #{article.id} ---") # 1. Format text extra = article.metadata_json or {} description = extra.get('og_description') or article.summary or "" if len(description) > 200: description = description[:197] + "..." raw_tags = extra.get('keywords', '') or ",".join(extra.get('feed_tags', [])) hashtag_list = [] if raw_tags: cleaned_tags = [t.strip().replace(" ", "").replace("-", "") for t in raw_tags.split(",") if t.strip()] hashtag_list = [f"#{tag}" for tag in cleaned_tags[:5]] hashtags_str = " ".join(hashtag_list) status_text = f" {article.title}\n\n" if description: status_text += f"{description}\n\n" status_text += f"🔗 {article.link}\n\n" if hashtags_str: status_text += f"{hashtags_str}" # 2. Upload OpenGraph image safely with line-by-line feedback media_ids = [] if article.og_image: logging.info(f"[Step 1/4] Starting image download request: {article.og_image}") try: headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"} # Enforce a strict 5-second connection & data read timeout img_response = requests.get(article.og_image, headers=headers, timeout=(5, 5)) logging.info(f"[Step 2/4] Image request finished. HTTP Status Code: {img_response.status_code}") if img_response.status_code == 200: img_data = img_response.content logging.info(f"[Step 3/4] Successfully read {len(img_data)} image bytes into local RAM.") mime_type = "image/jpeg" if article.og_image.lower().endswith(('.jpg', '.jpeg')) else "image/png" logging.info("[Step 4/4] Sending bytes to Mastodon API via client.media_post()... (Hangs here if API fails)") media_meta = mastodon_client.media_post( media_file=io.BytesIO(img_data), mime_type=mime_type, file_name=f"thumbnail_{article.id}.jpg", description=f"Thumbnail for {article.title}", synchronous=True # Forces the script to wait until upload completes or times out natively ) media_ids.append(media_meta['id']) logging.info(f"-> Mastodon image upload accepted! Received Media ID: {media_meta['id']}") else: logging.warning(f"-> Skipping image upload: Remote server returned status {img_response.status_code}") except requests.exceptions.Timeout: logging.error("-> Image download skipped: Connection timed out after 5 seconds.") except Exception as img_err: logging.error(f"-> Image handling system failed: {img_err}. Proceeding with text-only post.") # 3. Publish the text payload logging.info(f"Sending final status text payload to Mastodon timeline... (Media IDs: {media_ids})") mastodon_client.status_post(status=status_text, media_ids=media_ids if media_ids else None) logging.info(f"-> Successfully posted to Mastodon: {article.title}") # 4. Save DB state article.posted_to_mastodon = True article.save() logging.info(f"--- Finished Article #{article.id} ---\n") except Exception as e: logging.error(f"Failed to post article {article.id} to Mastodon: {e}") # --- Core Pipeline Logic --- def fetch_opengraph_data(url: str) -> dict: metadata = {"og_image": None, "extra": {}} headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"} try: response = requests.get(url, headers=headers, timeout=10) if response.status_code != 200: return metadata soup = BeautifulSoup(response.text, 'html.parser') for tag in soup.find_all('meta'): property_attr = tag.get('property', '') name_attr = tag.get('name', '') content_attr = tag.get('content', '') if not content_attr: continue if property_attr.startswith('og:'): key = property_attr[3:] if key == 'image': metadata['og_image'] = content_attr else: metadata['extra'][f"og_{key}"] = content_attr elif name_attr in ['description', 'keywords', 'author']: metadata['extra'][name_attr] = content_attr except Exception as e: logging.error(f"Error scraping metadata from {url}: {e}") return metadata def process_rss_feed(feed_url: str): logging.info(f"Parsing feed: {feed_url}") headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"} try: response = requests.get(feed_url, headers=headers, timeout=15) if response.status_code != 200: return parsed_feed = feedparser.parse(response.text) except Exception as e: return if not parsed_feed.entries: return feed_title = parsed_feed.feed.get('title', 'Unknown Feed') site_url = parsed_feed.feed.get('link', f"{urlparse(feed_url).scheme}://{urlparse(feed_url).netloc}") feed_record, _ = Feed.get_or_create( feed_url=feed_url, defaults={'title': feed_title, 'site_url': site_url} ) one_month_ago = datetime.datetime.now() - datetime.timedelta(days=30) new_articles = [] with db.atomic(): for entry in parsed_feed.entries: link = entry.get('link') if not link or Article.select().where(Article.link == link).exists(): continue logging.info(f"New article: {link}. Scraping OpenGraph...") # 1. Safely extract the article publication date pub_date = None time_struct = entry.get('published_parsed') or entry.get('updated_parsed') or entry.get('created_parsed') if time_struct: try: pub_date = datetime.datetime(*time_struct[:6]) except (ValueError, TypeError): pub_date = datetime.datetime.now() else: pub_date = datetime.datetime.now() # 2. Check the article's age BEFORE processing # If the article is older than 30 days, we treat it as already posted is_too_old = pub_date < age_limit if is_too_old: logging.info(f"-> Article is older than 1 month ({pub_date.date()}). Skipping social queues.") # We flip the flags to True so social media functions completely ignore it already_posted_state = True else: already_posted_state = False # Scrape web page metadata web_metadata = fetch_opengraph_data(link) author = entry.get('author') or web_metadata['extra'].get('author') summary = entry.get('summary') or web_metadata['extra'].get('og_description') extra_metadata = web_metadata['extra'] if 'tags' in entry: extra_metadata['feed_tags'] = [tag.get('term') for tag in entry.tags] # 3. Save to the database with conditional flags art = Article.create( feed=feed_record, title=entry.get('title', 'Untitled'), link=link, published_date=pub_date, author=author, summary=summary, content=entry.get('description'), og_image=web_metadata['og_image'], metadata_json=extra_metadata, # FIXED: Old posts are born marked as True, new posts are born False posted_to_mastodon=already_posted_state, posted_to_instagram=already_posted_state, posted_to_bluesky=already_posted_state ) # Only add to the execution list if it actually needs to be published if not already_posted_state: new_articles.append(art) feed_record.last_checked = datetime.datetime.now() feed_record.save() # Return any newly added records so the runner knows what needs posting return new_articles def retry_failed_mastodon_posts(mastodon_client): """ Finds articles younger than 30 days that failed to post, and publishes them with a 15-minute delay between posts. """ logging.info("Checking database for older articles that failed to post to Mastodon...") # Calculate the 30-day lookback threshold window one_month_ago = datetime.datetime.now() - datetime.timedelta(days=30) # Query database for matching failed items, ordered oldest to newest failed_articles = (Article .select() .where( (Article.posted_to_mastodon == False) & (Article.date_added >= one_month_ago) ) .order_by(Article.date_added.asc())) count = failed_articles.count() if count == 0: logging.info("No failed posts found within the 1-month window.") return logging.info(f"Found {count} articles waiting to be retried. Processing backlog...") for idx, article in enumerate(failed_articles, start=1): logging.info(f"Retrying backlog item ({idx}/{count}): {article.title}") # Attempt to publish the post post_article_to_mastodon(mastodon_client, article) # Refresh row to see if it successfully flipped to True article_refresh = Article.get_by_id(article.id) # Only enforce the 15-minute delay if the post was successful AND there are more items remaining if article_refresh.posted_to_mastodon and idx < count: logging.info("Post successful. Enforcing a 15-minute rate limit cooldown step...") # 15 minutes = 15 * 60 seconds = 900 seconds time.sleep(900) ''' # --- Execution Example --- if __name__ == "__main__": target_feeds = [ "https://jeffmackinnon.com/feeds/all.rss.xml" ] # Initialize Mastodon API connection m_client = get_mastodon_client() for feed_url in target_feeds: try: # 1. Parse and scrape the feeds new_posts = process_rss_feed(feed_url) # 2. Cycle through only the newly added posts and cross-post them if new_posts: logging.info(f"Found {len(new_posts)} new items to share to Mastodon.") for article in new_posts: post_article_to_mastodon(m_client, article) except Exception as e: logging.error(f"Error processing loop for {feed_url}: {e}") ''' if __name__ == "__main__": target_feeds = config["rss_feeds"] m_client = get_mastodon_client() # Track if the primary crawl cycle ran smoothly feed_crawl_successful = True for feed_url in target_feeds: try: new_posts = process_rss_feed(feed_url) if new_posts: logging.info(f"Found {len(new_posts)} new items to share.") for article in new_posts: post_article_to_mastodon(m_client, article) except Exception as e: logging.error(f"Critical pipeline failure on {feed_url}: {e}") feed_crawl_successful = False # Trigger the retry backlog recovery routine if a crawl failed # OR run it unconditionally every time to clean up historical failures if not feed_crawl_successful: logging.warning("Pipeline encountered errors during the crawl. Launching recovery runner...") retry_failed_mastodon_posts(m_client) else: # Optional: Run it anyway just in case old network drops left orphan records behind retry_failed_mastodon_posts(m_client)