#!/usr/bin/env python3 from flask import Flask, request, render_template, abort from flask_caching import Cache from urllib.parse import quote, unquote import requests import re import os from datetime import datetime, timezone from jinja2 import Environment, FileSystemLoader from orgpython import to_html app = Flask(__name__) # Get cache timeouts from environment variables CACHE_TIMEOUT = int(os.getenv("CACHE_TIMEOUT", "30")) CACHE_FILE_TIMEOUT = int(os.getenv("CACHE_FILE_TIMEOUT", "30")) # Configure Flask-Caching app.config["CACHE_TYPE"] = "SimpleCache" app.config["CACHE_DEFAULT_TIMEOUT"] = CACHE_TIMEOUT cache = Cache(app) class OrgSocialParser: def __init__(self): self.metadata = {} self.posts = [] def parse_content(self, content): """Parse the org social content and extract metadata and posts""" self.metadata = {} self.posts = [] # Extract global metadata self._extract_metadata(content) # Extract posts self._extract_posts(content) return self.posts def _extract_metadata(self, content): """Extract global metadata from the org file""" # Keywords are case-insensitive in org-mode (#+title: is as valid as # #+TITLE:). Use [ \t]* after the colon instead of \s* so an empty # value does not swallow the next line (\s matches newlines). metadata_patterns = { "TITLE": r"^\s*\#\+TITLE:[ \t]*(\S.*)$", "NICK": r"^\s*\#\+NICK:[ \t]*(\S.*)$", "DESCRIPTION": r"^\s*\#\+DESCRIPTION:[ \t]*(\S.*)$", "AVATAR": r"^\s*\#\+AVATAR:[ \t]*(\S.*)$", "PINNED": r"^\s*\#\+PINNED:[ \t]*(\S.*)$", } for key, pattern in metadata_patterns.items(): match = re.search(pattern, content, re.MULTILINE | re.IGNORECASE) if match: self.metadata[key] = match.group(1).strip() def _extract_posts(self, content): """Extract all posts from the org file""" # Find the Posts section (heading case varies across feeds: "* posts") posts_pattern = r"^\*\s+Posts\s*$" posts_section_match = re.search( posts_pattern, content, re.MULTILINE | re.IGNORECASE ) if not posts_section_match: print("Posts section not found") return posts_content = content[posts_section_match.end() :] # Ranges of #+BEGIN_.../#+END_... blocks: a "**" line inside them is # content (e.g. an org example in a src block), not a post header. block_ranges = [ (m.start(), m.end()) for m in re.finditer( r"^[ \t]*#\+BEGIN_(\w+)\b.*?^[ \t]*#\+END_\1[ \t]*$", posts_content, re.MULTILINE | re.DOTALL | re.IGNORECASE, ) ] def in_block(pos): return any(start <= pos < end for start, end in block_ranges) # Find all ** headers (posts) - support both formats: # Format 1: ** (ID in properties) # Format 2: ** (ID in header) post_pattern = r"^\*\*(?:\s+(.+?))?$" post_matches = [] for match in re.finditer(post_pattern, posts_content, re.MULTILINE): if in_block(match.start()): continue header_id = match.group(1).strip() if match.group(1) else None post_matches.append( {"start": match.start(), "end": match.end(), "header_id": header_id} ) if not post_matches: print("No headers found in Posts section") return print(f"Found {len(post_matches)} headers") # Extract content between ** headers for i, post_match in enumerate(post_matches): # Find the end of this post (next ** or end of content) if i + 1 < len(post_matches): end_pos = post_matches[i + 1]["start"] else: end_pos = len(posts_content) # Extract the block starting after the header line block_start = post_match["end"] block = posts_content[block_start:end_pos].strip() # Parse the post block with the header ID post = self._parse_post_block(block, post_match["header_id"]) if post and post.get("ID"): self.posts.append(post) print(f"Post added with ID: {post.get('ID')}") def _parse_post_block(self, block, header_id=None): """Parse a single post block Args: block: The post content block header_id: Optional ID from the header (takes priority over properties) """ post = {} # Extract properties properties_match = re.search( r":PROPERTIES:\s*\n(.*?)\n:END:", block, re.DOTALL | re.IGNORECASE ) if properties_match: properties_content = properties_match.group(1) # Parse each property using simple string operations for line in properties_content.split("\n"): line = line.strip() if line and line.startswith(":") and line.count(":") >= 2: # Find the second colon first_colon = line.find(":", 1) if first_colon != -1: # Property names are case-insensitive in org-mode; # normalize so lookups like post["ID"] always work. key = line[1:first_colon].strip().upper() value = line[first_colon + 1 :].strip() if key: post[key] = value # Header ID takes priority over properties ID (per specification) if header_id: post["ID"] = header_id # Extract post content: anything after the :PROPERTIES: drawer, or the # whole block if there is no drawer. A drawer-only post yields "". if properties_match: post["content"] = block[properties_match.end() :].strip() else: post["content"] = block.strip() return post def find_post_by_id(self, post_id): """Find a specific post by ID""" for post in self.posts: if post.get("ID") == post_id: return post return None class PreviewGenerator: def __init__(self, template_dir=".", template_name="template.html"): self.env = Environment(loader=FileSystemLoader(template_dir)) def og_description(value, max_length=120): import re # Replace newlines with spaces text = value.replace("\r\n", " ").replace("\n", " ").replace("\r", " ") # Collapse all whitespace to single spaces text = re.sub(r"\s+", " ", text) # HTML tag filter text = re.sub(r"<[^>]+>", "", text) # Collapse multiple spaces text = re.sub(r" +", " ", text) if len(text) > max_length: text = text[:max_length].rstrip() + "..." return text.strip() self.env.filters["og_description"] = og_description self.template = self.env.get_template(template_name) def generate_preview(self, post, metadata, feed_url=""): """Generate HTML preview for a single post""" context = self._prepare_context(post, metadata, feed_url) return self.template.render(**context) def _prepare_context(self, post, metadata, feed_url): """Prepare context data for template rendering""" post_id = post.get("ID", "") content = post.get("content", "") mood = post.get("MOOD", "") lang = post.get("LANG", "es") tags = post.get("TAGS", "") reply_to = post.get("REPLY_TO", "") client = post.get("CLIENT", "") formatted_content = self._format_content(content, mood, reply_to) nick = metadata.get("NICK", "User") title = metadata.get("TITLE", "social.org") description = metadata.get("DESCRIPTION", "") avatar_url = metadata.get("AVATAR", "") formatted_time = self._format_timestamp(post_id) tags_list = tags.split() if tags else [] post_url = f"{feed_url}#{post_id}" if feed_url and post_id else "" return { "post_id": post_id, "content": content, "formatted_content": formatted_content, "mood": mood, "language": lang, "tags": tags_list, "tags_string": tags, "reply_to": reply_to, "client": client, "is_reply": bool(reply_to), "has_mood": bool(mood), "has_tags": bool(tags), "has_content": bool(content.strip()), "nick": nick, "title": title, "description": description, "avatar_url": avatar_url, "has_avatar": bool(avatar_url), "user_initial": nick[0].upper() if nick else "U", "formatted_time": formatted_time, "timestamp": post_id, "post_url": post_url, } def _format_content(self, content, mood, reply_to): """Format post content from Org Mode to HTML using org-python""" if not content.strip() and mood: return f'{mood}' try: # Pre-process: Extract code blocks and replace with placeholders code_blocks = [] # Language and header args (:results ...) are optional code_block_pattern = ( r"^[ \t]*#\+BEGIN_SRC(?:[ \t]+([\w-]+))?[^\n]*\n" r"(.*?)\n[ \t]*#\+END_SRC[ \t]*$" ) def replace_code_block(match): lang = match.group(1) or "text" code = match.group(2) # HTML escape the code content code_escaped = ( code.replace("&", "&").replace("<", "<").replace(">", ">") ) placeholder = f"___CODE_BLOCK_{len(code_blocks)}___" code_blocks.append( {"lang": lang, "code": code_escaped, "placeholder": placeholder} ) return placeholder # Replace code blocks with placeholders content_processed = re.sub( code_block_pattern, replace_code_block, content, flags=re.MULTILINE | re.DOTALL | re.IGNORECASE, ) # Convert Org Mode to HTML using org-python html = to_html(content_processed, toc=False, highlight=True) # Restore code blocks with proper HTML formatting for block in code_blocks: code_html = f'
{block["code"]}
' html = html.replace(block["placeholder"], code_html) # Custom styling and post-processing # Make images full width html = re.sub( r']*?)src="([^"]+)"([^>]*?)>', r'', html, ) # Style links with our color html = html.replace("]*href="org-social:([^"]+)"[^>]*>@?([^<]+)', r'@\2 ', html, ) # Convert plain text URLs to clickable links or images # Match URLs that are not already part of an href attribute def linkify_urls(text): # Pattern to match URLs not already in href="" or src="" url_pattern = r'(?"]+)' def replace_url(match): url = match.group(0) # Check if URL is an image (check path before query parameters) image_extensions = ( ".jpg", ".jpeg", ".png", ".gif", ".webp", ".svg", ".bmp", ".ico", ) # Extract the path part before query parameters url_path = url.split("?")[0].split("#")[0].lower() if url_path.endswith(image_extensions): return f'Image' else: return f'{url}' return re.sub(url_pattern, replace_url, text) html = linkify_urls(html) return html or "No content" except Exception as e: print(f"Error formatting content with org-python: {e}") # Fallback to simple HTML escaping if org-python fails return content.replace("\n", "
").replace(" ", "  ") def _format_timestamp(self, timestamp): """Format timestamp for display""" try: dt = datetime.fromisoformat(timestamp.replace("Z", "+00:00")) return dt.strftime("%Y-%m-%d") except Exception: return "2024-01-01" def parse_post_url(post_url): """ Parse a post URL to extract the social.org file URL and post ID. Example: https://foo.org/social.org#2025-02-03T23:05:00+0100 Returns: (file_url, post_id) """ if "#" not in post_url: return None, None parts = post_url.split("#", 1) file_url = parts[0] post_id = parts[1] if len(parts) > 1 else None # '+' in timezone offsets (e.g. +0200) is decoded as space by query-string # parsers; restore it so IDs match the .org file entries. if post_id: post_id = post_id.replace(" ", "+") return file_url, post_id @cache.memoize(timeout=CACHE_FILE_TIMEOUT) def fetch_social_org(url): """Fetch a social.org file from a URL""" try: response = requests.get(url, timeout=10) response.raise_for_status() return response.text except Exception as e: print(f"Error fetching {url}: {e}") return None HOST_BASE_URL = "https://host.org-social.org" RELAY_RSS_URL = "https://relay.org-social.org/rss.xml" NICK_PATTERN = re.compile(r"[A-Za-z0-9_.-]{1,64}") def _post_datetime(post_id): try: dt = datetime.fromisoformat(post_id.replace("Z", "+00:00")) if dt.tzinfo is None: dt = dt.replace(tzinfo=timezone.utc) return dt except Exception: return datetime.min.replace(tzinfo=timezone.utc) def _format_long_timestamp(post_id): try: dt = datetime.fromisoformat(post_id.replace("Z", "+00:00")) return dt.strftime("%Y-%m-%d %H:%M") except Exception: return post_id def _build_blog_posts(parser, generator): blog_posts = [] now = datetime.now(timezone.utc) pinned_id = parser.metadata.get("PINNED", "") for raw in parser.posts: post_id = raw.get("ID", "") body = raw.get("content", "") or "" if not post_id: continue if raw.get("REPLY_TO"): continue if not body.strip(): continue # Scheduled posts (future ID) should not be displayed yet (spec 1.6) if _post_datetime(post_id) > now: continue # Mention-only posts are not public; VISIBILITY does not apply to # posts with GROUP (spec 1.5) visibility = (raw.get("VISIBILITY", "") or "").strip().lower() if visibility == "mention" and not raw.get("GROUP"): continue formatted_content = generator._format_content( body, raw.get("MOOD", ""), raw.get("REPLY_TO", "") ) tags_string = raw.get("TAGS", "") or "" blog_posts.append( { "id": post_id, "datetime": _post_datetime(post_id), "formatted_time": _format_long_timestamp(post_id), "formatted_content": formatted_content, "mood": raw.get("MOOD", ""), "lang": raw.get("LANG", "en"), "tags": tags_string.split() if tags_string else [], "client": raw.get("CLIENT", ""), "pinned": post_id == pinned_id, } ) # Pinned post first (spec 1.6), then newest first blog_posts.sort(key=lambda p: (p["pinned"], p["datetime"]), reverse=True) return blog_posts @app.route("/blog/") @cache.cached(timeout=CACHE_TIMEOUT) def blog(nick): """Render a single-page blog with all parent posts of a hosted user.""" if not NICK_PATTERN.fullmatch(nick): abort(400, "Invalid nick") file_url = f"{HOST_BASE_URL}/{nick}/social.org" content = fetch_social_org(file_url) if not content: abort(404, f"social.org not found for nick '{nick}'") parser = OrgSocialParser() parser.parse_content(content) generator = PreviewGenerator(template_dir="templates", template_name="post.html") posts = _build_blog_posts(parser, generator) metadata = parser.metadata display_nick = metadata.get("NICK", nick) return render_template( "blog.html", nick=display_nick, title=metadata.get("TITLE", f"{display_nick}'s journal"), description=metadata.get("DESCRIPTION", ""), avatar_url=metadata.get("AVATAR", ""), has_avatar=bool(metadata.get("AVATAR")), user_initial=(display_nick or "U")[0].upper(), feed_url=file_url, rss_url=f"{RELAY_RSS_URL}?feed={quote(file_url, safe='')}", posts=posts, ) @app.route("/") @cache.cached(timeout=CACHE_TIMEOUT, query_string=True) def preview(): """Main route to display post preview""" post_url = request.args.get("post") if not post_url: domain = os.getenv("DOMAIN", "localhost") port = os.getenv("EXTERNAL_PORT", "8080") protocol = os.getenv("PROTOCOL", "http") debug_mode = os.getenv("FLASK_DEBUG", "False").lower() in ("true", "1", "t") flask_env = os.getenv("FLASK_ENV", "production") # Show port only in debug mode or development show_port = debug_mode or flask_env == "development" return render_template( "welcome.html", domain=domain, port=port, protocol=protocol, show_port=show_port, ) # Decode the URL parameter post_url = unquote(post_url) # Parse the post URL file_url, post_id = parse_post_url(post_url) if not file_url or not post_id: abort( 400, "Invalid post URL format. Expected: https://example.org/social.org#POST_ID", ) # Fetch the social.org file content = fetch_social_org(file_url) if not content: abort(500, f"Could not fetch social.org file from {file_url}") # Parse the content parser = OrgSocialParser() parser.parse_content(content) # Find the specific post post = parser.find_post_by_id(post_id) if not post: abort(404, f"Post with ID {post_id} not found") # Generate preview generator = PreviewGenerator(template_dir="templates", template_name="post.html") html = generator.generate_preview(post, parser.metadata, feed_url=file_url) return html if __name__ == "__main__": debug_mode = os.getenv("FLASK_DEBUG", "False").lower() in ("true", "1", "t") app.run(host="0.0.0.0", port=8080, debug=debug_mode)