Rosie

Public wire desk

Rosies Scraper

Rosie reads public posts with Python. No logins, no private accounts, no walls.

Rosie at the desk
Rosie runs this desk. She only pulls what a stranger could already read.
Source
How many

The desk is clear.

Pick Bluesky, Reddit, Mastodon, Hacker News, or a public RSS feed. Rosie will bring the posts back here.

Python engine

The desk shells out to this script. On a host without Python, the same public reads run in Node instead.

#!/usr/bin/env python3
"""Rosies Scraper — public social feeds only.

Stdin:  {"source": "bluesky"|"reddit"|"mastodon"|"hackernews"|"rss",
         "query": "...", "limit": 12}
Stdout: one JSON object. No logins, no cookies, no private accounts.
"""

from __future__ import annotations

import ipaddress
import json
import re
import socket
import sys
import time
import urllib.error
import urllib.parse
import urllib.request
import xml.etree.ElementTree as ET
from html.parser import HTMLParser

UA_HONEST = "RosiesScraper/1.0 (public feed reader; no login)"
UA_REDDIT = (
    "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
    "(KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36"
)
MAX_BYTES = 1_500_000
TIMEOUT = 12
SOURCES = {"bluesky", "reddit", "mastodon", "hackernews", "rss"}


class AppError(Exception):
    pass


class TextExtractor(HTMLParser):
    def __init__(self) -> None:
        super().__init__(convert_charrefs=True)
        self.parts: list[str] = []

    def handle_data(self, data: str) -> None:
        self.parts.append(data)

    def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
        if tag in {"p", "br", "div", "li", "tr", "h1", "h2", "h3", "h4"}:
            self.parts.append(" ")


def html_to_text(value: str) -> str:
    raw = value or ""
    parser = TextExtractor()
    try:
        parser.feed(raw)
        parser.close()
        text = "".join(parser.parts)
    except Exception:
        text = re.sub(r"<[^>]+>", " ", raw)
    text = re.sub(r"\s+", " ", text).strip()
    return text[:4000]


def local(tag: str) -> str:
    return tag.rsplit("}", 1)[-1] if tag else ""


def direct(el: ET.Element, name: str) -> list[ET.Element]:
    return [child for child in list(el) if local(child.tag) == name]


def child_text(el: ET.Element | None, name: str) -> str:
    if el is None:
        return ""
    for child in direct(el, name):
        return "".join(child.itertext()).strip()
    return ""


def link_href(el: ET.Element) -> str:
    fallback = ""
    for child in direct(el, "link"):
        href = (child.attrib.get("href") or "").strip()
        rel = child.attrib.get("rel") or "alternate"
        if rel in {"alternate", ""} and href:
            return href
        if href and not fallback:
            fallback = href
    return fallback or child_text(el, "link")


def assert_public(url: str) -> urllib.parse.ParseResult:
    parsed = urllib.parse.urlparse(url)
    if parsed.scheme not in {"http", "https"}:
        raise AppError("Only public http and https addresses are allowed.")
    if parsed.username or parsed.password:
        raise AppError("Addresses with embedded passwords are not allowed.")
    host = (parsed.hostname or "").strip(".").lower()
    if not host:
        raise AppError("That address has no host.")
    if parsed.port and parsed.port not in {80, 443}:
        raise AppError("Only standard web ports are allowed.")
    blocked_suffix = (".local", ".localhost", ".internal", ".localdomain")
    if host in {"localhost", "metadata.google.internal", "0.0.0.0"} or host.endswith(blocked_suffix):
        raise AppError("That host is not a public source.")
    literal = None
    try:
        literal = ipaddress.ip_address(host)
    except ValueError:
        literal = None
    addresses: list[str] = []
    if literal is not None:
        addresses.append(str(literal))
    else:
        try:
            infos = socket.getaddrinfo(host, None)
        except socket.gaierror as exc:
            raise AppError("Could not resolve that host.") from exc
        for info in infos:
            addresses.append(info[4][0])
    if not addresses:
        raise AppError("Could not resolve that host.")
    for raw_ip in addresses:
        ip = ipaddress.ip_address(raw_ip)
        if not ip.is_global:
            raise AppError("Only public hosts are allowed.")
    return parsed


class SafeRedirect(urllib.request.HTTPRedirectHandler):
    max_redirections = 4

    def redirect_request(self, req, fp, code, msg, headers, newurl):
        assert_public(newurl)
        return super().redirect_request(req, fp, code, msg, headers, newurl)


OPENER = urllib.request.build_opener(SafeRedirect)


def fetch(url: str, accept: str, user_agent: str) -> tuple[int, str]:
    assert_public(url)
    request = urllib.request.Request(
        url,
        headers={"User-Agent": user_agent, "Accept": accept},
        method="GET",
    )
    try:
        with OPENER.open(request, timeout=TIMEOUT) as response:
            status = getattr(response, "status", 200)
            chunks: list[bytes] = []
            size = 0
            while True:
                piece = response.read(65536)
                if not piece:
                    break
                size += len(piece)
                if size > MAX_BYTES:
                    raise AppError("The source response was too large.")
                chunks.append(piece)
            body = b"".join(chunks).decode("utf-8", "replace")
            return status, body
    except AppError:
        raise
    except urllib.error.HTTPError as exc:
        if exc.code == 404:
            raise AppError("Nothing public lives at that address.") from exc
        if exc.code == 429:
            raise AppError("That source asked Rosies to slow down. Try again in a minute.") from exc
        if exc.code in {401, 403}:
            raise AppError("That source refused an anonymous public read.") from exc
        raise AppError(f"The source returned {exc.code}.") from exc
    except (TimeoutError, socket.timeout):
        raise AppError("The source took too long to answer.")
    except urllib.error.URLError as exc:
        reason = getattr(exc, "reason", None)
        if isinstance(reason, (TimeoutError, socket.timeout)):
            raise AppError("The source took too long to answer.") from exc
        raise AppError("Rosies could not reach that source.") from exc


def fetch_retry(url: str, accept: str, user_agent: str, log: list[dict]) -> str:
    last: AppError | None = None
    for attempt in range(2):
        try:
            status, body = fetch(url, accept, user_agent)
            log.append({"url": url, "status": status})
            return body
        except AppError as exc:
            last = exc
            if attempt == 0 and "slow down" in str(exc):
                time.sleep(1.2)
                continue
            raise
    raise last or AppError("The source did not answer.")


def blank_post() -> dict:
    return {
        "id": "",
        "author": "",
        "handle": "",
        "title": "",
        "text": "",
        "url": "",
        "createdAt": "",
        "likes": None,
        "replies": None,
        "reposts": None,
        "score": None,
    }


def bsky_post_url(handle: str, uri: str) -> str:
    rkey = uri.rsplit("/", 1)[-1]
    return f"https://bsky.app/profile/{urllib.parse.quote(handle)}/post/{urllib.parse.quote(rkey)}"


def clean_handle_bsky(query: str) -> str:
    handle = query.strip()
    handle = re.sub(r"^https?://bsky\.app/profile/", "", handle, flags=re.I)
    handle = handle.strip().strip("/").lstrip("@")
    if not re.fullmatch(r"[A-Za-z0-9._-]{2,80}", handle) or ".." in handle:
        raise AppError("Enter a Bluesky handle, like bsky.app.")
    return handle


def scrape_bluesky(query: str, limit: int, log: list[dict]) -> dict:
    handle = clean_handle_bsky(query)
    quoted = urllib.parse.quote(handle)
    profile_url = f"https://public.api.bsky.app/xrpc/app.bsky.actor.getProfile?actor={quoted}"
    feed_url = (
        "https://public.api.bsky.app/xrpc/app.bsky.feed.getAuthorFeed"
        f"?actor={quoted}&limit={limit}"
    )
    profile_raw = fetch_retry(profile_url, "application/json", UA_HONEST, log)
    feed_raw = fetch_retry(feed_url, "application/json", UA_HONEST, log)
    try:
        profile = json.loads(profile_raw)
        feed = json.loads(feed_raw)
    except json.JSONDecodeError as exc:
        raise AppError("Bluesky returned a response Rosies could not read.") from exc
    posts = []
    for item in (feed.get("feed") or [])[:limit]:
        post = item.get("post") or {}
        author = post.get("author") or {}
        record = post.get("record") or {}
        handle_name = author.get("handle") or handle
        row = blank_post()
        row.update(
            {
                "id": post.get("uri") or "",
                "author": author.get("displayName") or handle_name,
                "handle": handle_name,
                "title": "",
                "text": record.get("text") or "",
                "url": bsky_post_url(handle_name, post.get("uri") or ""),
                "createdAt": record.get("createdAt") or post.get("indexedAt") or "",
                "likes": post.get("likeCount"),
                "replies": post.get("replyCount"),
                "reposts": post.get("repostCount"),
            }
        )
        posts.append(row)
    about = profile.get("description") or ""
    return {
        "profile": {
            "name": profile.get("displayName") or handle,
            "handle": profile.get("handle") or handle,
            "about": about[:500],
            "url": f"https://bsky.app/profile/{urllib.parse.quote(profile.get('handle') or handle)}",
        },
        "posts": posts,
        "note": "" if posts else "The account is public, but no posts came back.",
    }


def clean_reddit(query: str) -> tuple[str, str]:
    raw = query.strip().strip("/")
    lowered = raw.lower()
    if lowered.startswith("u/") or lowered.startswith("user/"):
        name = raw.split("/", 1)[1].strip("/")
        kind = "user"
    elif lowered.startswith("r/"):
        name = raw.split("/", 1)[1].strip("/")
        kind = "subreddit"
    elif re.fullmatch(r"[A-Za-z0-9_]{2,30}", raw):
        name = raw
        kind = "subreddit"
    else:
        raise AppError("Use a subreddit like r/python or a user like u/spez.")
    if not re.fullmatch(r"[A-Za-z0-9_]{2,30}", name):
        raise AppError("That Reddit name looks off. Stick to letters, numbers, and underscores.")
    return kind, name


def scrape_reddit(query: str, limit: int, log: list[dict]) -> dict:
    kind, name = clean_reddit(query)
    if kind == "user":
        url = f"https://www.reddit.com/user/{urllib.parse.quote(name)}/.rss?limit={limit}"
        label = f"u/{name}"
        profile_url = f"https://www.reddit.com/user/{urllib.parse.quote(name)}"
    else:
        url = f"https://www.reddit.com/r/{urllib.parse.quote(name)}/.rss?limit={limit}"
        label = f"r/{name}"
        profile_url = f"https://www.reddit.com/r/{urllib.parse.quote(name)}"
    body = fetch_retry(url, "application/atom+xml, application/rss+xml, */*", UA_REDDIT, log)
    title, posts = parse_feed(body, limit)
    return {
        "profile": {
            "name": title or label,
            "handle": label,
            "about": "Public Atom feed. Reddit blocks anonymous JSON, so Rosies reads the feed.",
            "url": profile_url,
        },
        "posts": posts,
        "note": "" if posts else "The feed answered, but it had no entries.",
    }


def parse_masto(query: str) -> tuple[str, str]:
    raw = query.strip()
    if raw.startswith("http://") or raw.startswith("https://"):
        parsed = urllib.parse.urlparse(raw)
        host = (parsed.hostname or "").lower()
        path = parsed.path.strip("/")
        if path.startswith("@"):
            acct = path[1:].split("/")[0]
        elif path.startswith("users/"):
            bits = path.split("/")
            acct = bits[1] if len(bits) > 1 else ""
        else:
            raise AppError("Use a profile URL or name@instance.")
        if "@" in acct:
            user, host = acct.split("@", 1)
            return host.lower(), user
        if not host or not acct:
            raise AppError("Use a profile URL or name@instance.")
        return host, acct
    raw = raw.lstrip("@")
    if raw.count("@") != 1:
        raise AppError("Use name@instance, for example [email protected].")
    user, host = raw.split("@", 1)
    user = user.strip()
    host = host.strip().lower()
    if not user or not host or "/" in host:
        raise AppError("Use name@instance, for example [email protected].")
    return host, user


def scrape_mastodon(query: str, limit: int, log: list[dict]) -> dict:
    host, user = parse_masto(query)
    assert_public(f"https://{host}/")
    lookup = f"https://{host}/api/v1/accounts/lookup?acct={urllib.parse.quote(user)}"
    raw_account = fetch_retry(lookup, "application/json", UA_HONEST, log)
    try:
        account = json.loads(raw_account)
    except json.JSONDecodeError as exc:
        raise AppError("That instance returned an account Rosies could not read.") from exc
    account_id = str(account.get("id") or "")
    if not account_id or not re.fullmatch(r"[A-Za-z0-9]+", account_id):
        raise AppError("That instance did not return a public account id.")
    statuses_url = (
        f"https://{host}/api/v1/accounts/{account_id}/statuses?limit={limit}"
    )
    raw_statuses = fetch_retry(statuses_url, "application/json", UA_HONEST, log)
    try:
        statuses = json.loads(raw_statuses)
    except json.JSONDecodeError as exc:
        raise AppError("That instance returned posts Rosies could not read.") from exc
    if not isinstance(statuses, list):
        raise AppError("That instance did not return a public timeline.")
    posts = []
    for status in statuses[:limit]:
        if not isinstance(status, dict):
            continue
        warning = (status.get("spoiler_text") or "").strip()
        text = html_to_text(status.get("content") or "")
        if warning:
            text = f"[CW: {warning}] {text}".strip()
        row = blank_post()
        row.update(
            {
                "id": str(status.get("id") or ""),
                "author": account.get("display_name") or account.get("username") or user,
                "handle": f"@{account.get('acct') or user}@{host}",
                "title": "",
                "text": text,
                "url": status.get("url") or "",
                "createdAt": status.get("created_at") or "",
                "likes": status.get("favourites_count"),
                "replies": status.get("replies_count"),
                "reposts": status.get("reblogs_count"),
            }
        )
        posts.append(row)
    return {
        "profile": {
            "name": account.get("display_name") or user,
            "handle": f"@{account.get('acct') or user}@{host}",
            "about": html_to_text(account.get("note") or "")[:500],
            "url": account.get("url") or f"https://{host}/@{user}",
        },
        "posts": posts,
        "note": "" if posts else "The account is public, but no posts came back.",
    }


def scrape_hn(query: str, limit: int, log: list[dict]) -> dict:
    raw = query.strip()
    lowered = raw.lower()
    if lowered.startswith("u:") or lowered.startswith("user:"):
        name = raw.split(":", 1)[1].strip()
        if not re.fullmatch(r"[A-Za-z0-9_-]{2,30}", name):
            raise AppError("Enter a Hacker News username, like u:dang.")
        tag = f"story,author_{name}"
        url = (
            "https://hn.algolia.com/api/v1/search_by_date?"
            f"tags={urllib.parse.quote(tag)}&hitsPerPage={limit}"
        )
        label = f"u:{name}"
        about = f"Public stories by {name}."
    else:
        if len(raw) < 2:
            raise AppError("Enter a topic, or a user like u:dang.")
        url = (
            "https://hn.algolia.com/api/v1/search_by_date?"
            f"query={urllib.parse.quote(raw)}&tags=story&hitsPerPage={limit}"
        )
        label = raw
        about = f"Public stories matching “{raw}”."
    body = fetch_retry(url, "application/json", UA_HONEST, log)
    try:
        payload = json.loads(body)
    except json.JSONDecodeError as exc:
        raise AppError("Hacker News returned a response Rosies could not read.") from exc
    posts = []
    for hit in (payload.get("hits") or [])[:limit]:
        if not isinstance(hit, dict):
            continue
        object_id = str(hit.get("objectID") or "")
        row = blank_post()
        row.update(
            {
                "id": object_id,
                "author": hit.get("author") or "",
                "handle": hit.get("author") or "",
                "title": hit.get("title") or "",
                "text": "",
                "url": hit.get("url") or (
                    f"https://news.ycombinator.com/item?id={object_id}" if object_id else ""
                ),
                "createdAt": hit.get("created_at") or "",
                "replies": hit.get("num_comments"),
                "score": hit.get("points"),
            }
        )
        posts.append(row)
    return {
        "profile": {
            "name": "Hacker News",
            "handle": label,
            "about": about,
            "url": "https://news.ycombinator.com/",
        },
        "posts": posts,
        "note": "" if posts else "No public stories matched.",
    }


def scrape_rss(query: str, limit: int, log: list[dict]) -> dict:
    url = query.strip()
    parsed = assert_public(url)
    if not parsed.path or parsed.path == "/":
        # A bare origin is still a valid feed location sometimes; allow it.
        pass
    body = fetch_retry(url, "application/rss+xml, application/atom+xml, application/xml, */*", UA_HONEST, log)
    title, posts = parse_feed(body, limit)
    host = parsed.hostname or url
    return {
        "profile": {
            "name": title or host,
            "handle": host,
            "about": "Public RSS or Atom feed.",
            "url": url,
        },
        "posts": posts,
        "note": "" if posts else "The feed answered, but it had no entries.",
    }


def parse_feed(xml_text: str, limit: int) -> tuple[str, list[dict]]:
    try:
        root = ET.fromstring(xml_text)
    except ET.ParseError as exc:
        raise AppError("That feed was not valid RSS or Atom.") from exc
    title = ""
    nodes: list[ET.Element] = []
    if local(root.tag) == "rss":
        channel = next((child for child in list(root) if local(child.tag) == "channel"), None)
        if channel is not None:
            title = child_text(channel, "title")
            nodes = direct(channel, "item")
    else:
        title = child_text(root, "title")
        nodes = [node for node in root.iter() if local(node.tag) == "entry"]
        if not nodes:
            nodes = [node for node in root.iter() if local(node.tag) == "item"]
    posts = []
    for node in nodes[:limit]:
        author_el = next(iter(direct(node, "author")), None)
        author = ""
        if author_el is not None:
            author = child_text(author_el, "name") or "".join(author_el.itertext()).strip()
        if not author:
            author = child_text(node, "creator")
        content = (
            child_text(node, "content")
            or child_text(node, "description")
            or child_text(node, "summary")
            or child_text(node, "encoded")
        )
        row = blank_post()
        row.update(
            {
                "id": child_text(node, "id") or child_text(node, "guid") or link_href(node),
                "author": html_to_text(author),
                "handle": html_to_text(author),
                "title": html_to_text(child_text(node, "title")),
                "text": html_to_text(content),
                "url": link_href(node),
                "createdAt": (
                    child_text(node, "updated")
                    or child_text(node, "published")
                    or child_text(node, "pubDate")
                    or child_text(node, "date")
                ),
            }
        )
        posts.append(row)
    return html_to_text(title), posts


def dispatch(source: str, query: str, limit: int) -> dict:
    log: list[dict] = []
    if source == "bluesky":
        result = scrape_bluesky(query, limit, log)
    elif source == "reddit":
        result = scrape_reddit(query, limit, log)
    elif source == "mastodon":
        result = scrape_mastodon(query, limit, log)
    elif source == "hackernews":
        result = scrape_hn(query, limit, log)
    elif source == "rss":
        result = scrape_rss(query, limit, log)
    else:
        raise AppError("Pick a source Rosies knows.")
    result["requests"] = log
    return result


def emit(payload: dict) -> None:
    sys.stdout.write(json.dumps(payload, ensure_ascii=False))
    sys.stdout.write("\n")


def main() -> int:
    raw = sys.stdin.read()
    try:
        incoming = json.loads(raw) if raw.strip() else {}
    except json.JSONDecodeError:
        emit({"ok": False, "error": "The desk sent an unreadable request."})
        return 0
    if not isinstance(incoming, dict):
        emit({"ok": False, "error": "The desk sent an unreadable request."})
        return 0
    source = incoming.get("source")
    query = incoming.get("query")
    limit = incoming.get("limit", 12)
    if source not in SOURCES:
        emit({"ok": False, "error": "Pick a source Rosies knows."})
        return 0
    if not isinstance(query, str) or not query.strip():
        emit({"ok": False, "error": "Enter something to pull."})
        return 0
    if isinstance(limit, bool) or not isinstance(limit, int) or not 1 <= limit <= 25:
        emit({"ok": False, "error": "Ask for between 1 and 25 posts."})
        return 0
    started = time.perf_counter()
    try:
        result = dispatch(source, query.strip(), limit)
    except AppError as exc:
        emit({"ok": False, "error": str(exc)})
        return 0
    except Exception:
        emit({"ok": False, "error": "The source did not answer in a way Rosies could read."})
        return 0
    elapsed = int((time.perf_counter() - started) * 1000)
    emit(
        {
            "ok": True,
            "engine": "python",
            "engineDetail": f"Python {sys.version.split()[0]}",
            "source": source,
            "query": query.strip(),
            "fetchedAt": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
            "elapsedMs": elapsed,
            "profile": result.get("profile"),
            "posts": result.get("posts") or [],
            "requests": result.get("requests") or [],
            "note": result.get("note") or "",
        }
    )
    return 0


if __name__ == "__main__":
    raise SystemExit(main())