
Public wire desk
Rosies Scraper
Rosie reads public posts with Python. No logins, no private accounts, no walls.

The desk is clear.
Pick Bluesky, Reddit, Mastodon, Hacker News, or a public RSS feed. Rosie will bring the posts back here.
Python engine
The desk shells out to this script. On a host without Python, the same public reads run in Node instead.
#!/usr/bin/env python3
"""Rosies Scraper — public social feeds only.
Stdin: {"source": "bluesky"|"reddit"|"mastodon"|"hackernews"|"rss",
"query": "...", "limit": 12}
Stdout: one JSON object. No logins, no cookies, no private accounts.
"""
from __future__ import annotations
import ipaddress
import json
import re
import socket
import sys
import time
import urllib.error
import urllib.parse
import urllib.request
import xml.etree.ElementTree as ET
from html.parser import HTMLParser
UA_HONEST = "RosiesScraper/1.0 (public feed reader; no login)"
UA_REDDIT = (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36"
)
MAX_BYTES = 1_500_000
TIMEOUT = 12
SOURCES = {"bluesky", "reddit", "mastodon", "hackernews", "rss"}
class AppError(Exception):
pass
class TextExtractor(HTMLParser):
def __init__(self) -> None:
super().__init__(convert_charrefs=True)
self.parts: list[str] = []
def handle_data(self, data: str) -> None:
self.parts.append(data)
def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
if tag in {"p", "br", "div", "li", "tr", "h1", "h2", "h3", "h4"}:
self.parts.append(" ")
def html_to_text(value: str) -> str:
raw = value or ""
parser = TextExtractor()
try:
parser.feed(raw)
parser.close()
text = "".join(parser.parts)
except Exception:
text = re.sub(r"<[^>]+>", " ", raw)
text = re.sub(r"\s+", " ", text).strip()
return text[:4000]
def local(tag: str) -> str:
return tag.rsplit("}", 1)[-1] if tag else ""
def direct(el: ET.Element, name: str) -> list[ET.Element]:
return [child for child in list(el) if local(child.tag) == name]
def child_text(el: ET.Element | None, name: str) -> str:
if el is None:
return ""
for child in direct(el, name):
return "".join(child.itertext()).strip()
return ""
def link_href(el: ET.Element) -> str:
fallback = ""
for child in direct(el, "link"):
href = (child.attrib.get("href") or "").strip()
rel = child.attrib.get("rel") or "alternate"
if rel in {"alternate", ""} and href:
return href
if href and not fallback:
fallback = href
return fallback or child_text(el, "link")
def assert_public(url: str) -> urllib.parse.ParseResult:
parsed = urllib.parse.urlparse(url)
if parsed.scheme not in {"http", "https"}:
raise AppError("Only public http and https addresses are allowed.")
if parsed.username or parsed.password:
raise AppError("Addresses with embedded passwords are not allowed.")
host = (parsed.hostname or "").strip(".").lower()
if not host:
raise AppError("That address has no host.")
if parsed.port and parsed.port not in {80, 443}:
raise AppError("Only standard web ports are allowed.")
blocked_suffix = (".local", ".localhost", ".internal", ".localdomain")
if host in {"localhost", "metadata.google.internal", "0.0.0.0"} or host.endswith(blocked_suffix):
raise AppError("That host is not a public source.")
literal = None
try:
literal = ipaddress.ip_address(host)
except ValueError:
literal = None
addresses: list[str] = []
if literal is not None:
addresses.append(str(literal))
else:
try:
infos = socket.getaddrinfo(host, None)
except socket.gaierror as exc:
raise AppError("Could not resolve that host.") from exc
for info in infos:
addresses.append(info[4][0])
if not addresses:
raise AppError("Could not resolve that host.")
for raw_ip in addresses:
ip = ipaddress.ip_address(raw_ip)
if not ip.is_global:
raise AppError("Only public hosts are allowed.")
return parsed
class SafeRedirect(urllib.request.HTTPRedirectHandler):
max_redirections = 4
def redirect_request(self, req, fp, code, msg, headers, newurl):
assert_public(newurl)
return super().redirect_request(req, fp, code, msg, headers, newurl)
OPENER = urllib.request.build_opener(SafeRedirect)
def fetch(url: str, accept: str, user_agent: str) -> tuple[int, str]:
assert_public(url)
request = urllib.request.Request(
url,
headers={"User-Agent": user_agent, "Accept": accept},
method="GET",
)
try:
with OPENER.open(request, timeout=TIMEOUT) as response:
status = getattr(response, "status", 200)
chunks: list[bytes] = []
size = 0
while True:
piece = response.read(65536)
if not piece:
break
size += len(piece)
if size > MAX_BYTES:
raise AppError("The source response was too large.")
chunks.append(piece)
body = b"".join(chunks).decode("utf-8", "replace")
return status, body
except AppError:
raise
except urllib.error.HTTPError as exc:
if exc.code == 404:
raise AppError("Nothing public lives at that address.") from exc
if exc.code == 429:
raise AppError("That source asked Rosies to slow down. Try again in a minute.") from exc
if exc.code in {401, 403}:
raise AppError("That source refused an anonymous public read.") from exc
raise AppError(f"The source returned {exc.code}.") from exc
except (TimeoutError, socket.timeout):
raise AppError("The source took too long to answer.")
except urllib.error.URLError as exc:
reason = getattr(exc, "reason", None)
if isinstance(reason, (TimeoutError, socket.timeout)):
raise AppError("The source took too long to answer.") from exc
raise AppError("Rosies could not reach that source.") from exc
def fetch_retry(url: str, accept: str, user_agent: str, log: list[dict]) -> str:
last: AppError | None = None
for attempt in range(2):
try:
status, body = fetch(url, accept, user_agent)
log.append({"url": url, "status": status})
return body
except AppError as exc:
last = exc
if attempt == 0 and "slow down" in str(exc):
time.sleep(1.2)
continue
raise
raise last or AppError("The source did not answer.")
def blank_post() -> dict:
return {
"id": "",
"author": "",
"handle": "",
"title": "",
"text": "",
"url": "",
"createdAt": "",
"likes": None,
"replies": None,
"reposts": None,
"score": None,
}
def bsky_post_url(handle: str, uri: str) -> str:
rkey = uri.rsplit("/", 1)[-1]
return f"https://bsky.app/profile/{urllib.parse.quote(handle)}/post/{urllib.parse.quote(rkey)}"
def clean_handle_bsky(query: str) -> str:
handle = query.strip()
handle = re.sub(r"^https?://bsky\.app/profile/", "", handle, flags=re.I)
handle = handle.strip().strip("/").lstrip("@")
if not re.fullmatch(r"[A-Za-z0-9._-]{2,80}", handle) or ".." in handle:
raise AppError("Enter a Bluesky handle, like bsky.app.")
return handle
def scrape_bluesky(query: str, limit: int, log: list[dict]) -> dict:
handle = clean_handle_bsky(query)
quoted = urllib.parse.quote(handle)
profile_url = f"https://public.api.bsky.app/xrpc/app.bsky.actor.getProfile?actor={quoted}"
feed_url = (
"https://public.api.bsky.app/xrpc/app.bsky.feed.getAuthorFeed"
f"?actor={quoted}&limit={limit}"
)
profile_raw = fetch_retry(profile_url, "application/json", UA_HONEST, log)
feed_raw = fetch_retry(feed_url, "application/json", UA_HONEST, log)
try:
profile = json.loads(profile_raw)
feed = json.loads(feed_raw)
except json.JSONDecodeError as exc:
raise AppError("Bluesky returned a response Rosies could not read.") from exc
posts = []
for item in (feed.get("feed") or [])[:limit]:
post = item.get("post") or {}
author = post.get("author") or {}
record = post.get("record") or {}
handle_name = author.get("handle") or handle
row = blank_post()
row.update(
{
"id": post.get("uri") or "",
"author": author.get("displayName") or handle_name,
"handle": handle_name,
"title": "",
"text": record.get("text") or "",
"url": bsky_post_url(handle_name, post.get("uri") or ""),
"createdAt": record.get("createdAt") or post.get("indexedAt") or "",
"likes": post.get("likeCount"),
"replies": post.get("replyCount"),
"reposts": post.get("repostCount"),
}
)
posts.append(row)
about = profile.get("description") or ""
return {
"profile": {
"name": profile.get("displayName") or handle,
"handle": profile.get("handle") or handle,
"about": about[:500],
"url": f"https://bsky.app/profile/{urllib.parse.quote(profile.get('handle') or handle)}",
},
"posts": posts,
"note": "" if posts else "The account is public, but no posts came back.",
}
def clean_reddit(query: str) -> tuple[str, str]:
raw = query.strip().strip("/")
lowered = raw.lower()
if lowered.startswith("u/") or lowered.startswith("user/"):
name = raw.split("/", 1)[1].strip("/")
kind = "user"
elif lowered.startswith("r/"):
name = raw.split("/", 1)[1].strip("/")
kind = "subreddit"
elif re.fullmatch(r"[A-Za-z0-9_]{2,30}", raw):
name = raw
kind = "subreddit"
else:
raise AppError("Use a subreddit like r/python or a user like u/spez.")
if not re.fullmatch(r"[A-Za-z0-9_]{2,30}", name):
raise AppError("That Reddit name looks off. Stick to letters, numbers, and underscores.")
return kind, name
def scrape_reddit(query: str, limit: int, log: list[dict]) -> dict:
kind, name = clean_reddit(query)
if kind == "user":
url = f"https://www.reddit.com/user/{urllib.parse.quote(name)}/.rss?limit={limit}"
label = f"u/{name}"
profile_url = f"https://www.reddit.com/user/{urllib.parse.quote(name)}"
else:
url = f"https://www.reddit.com/r/{urllib.parse.quote(name)}/.rss?limit={limit}"
label = f"r/{name}"
profile_url = f"https://www.reddit.com/r/{urllib.parse.quote(name)}"
body = fetch_retry(url, "application/atom+xml, application/rss+xml, */*", UA_REDDIT, log)
title, posts = parse_feed(body, limit)
return {
"profile": {
"name": title or label,
"handle": label,
"about": "Public Atom feed. Reddit blocks anonymous JSON, so Rosies reads the feed.",
"url": profile_url,
},
"posts": posts,
"note": "" if posts else "The feed answered, but it had no entries.",
}
def parse_masto(query: str) -> tuple[str, str]:
raw = query.strip()
if raw.startswith("http://") or raw.startswith("https://"):
parsed = urllib.parse.urlparse(raw)
host = (parsed.hostname or "").lower()
path = parsed.path.strip("/")
if path.startswith("@"):
acct = path[1:].split("/")[0]
elif path.startswith("users/"):
bits = path.split("/")
acct = bits[1] if len(bits) > 1 else ""
else:
raise AppError("Use a profile URL or name@instance.")
if "@" in acct:
user, host = acct.split("@", 1)
return host.lower(), user
if not host or not acct:
raise AppError("Use a profile URL or name@instance.")
return host, acct
raw = raw.lstrip("@")
if raw.count("@") != 1:
raise AppError("Use name@instance, for example [email protected].")
user, host = raw.split("@", 1)
user = user.strip()
host = host.strip().lower()
if not user or not host or "/" in host:
raise AppError("Use name@instance, for example [email protected].")
return host, user
def scrape_mastodon(query: str, limit: int, log: list[dict]) -> dict:
host, user = parse_masto(query)
assert_public(f"https://{host}/")
lookup = f"https://{host}/api/v1/accounts/lookup?acct={urllib.parse.quote(user)}"
raw_account = fetch_retry(lookup, "application/json", UA_HONEST, log)
try:
account = json.loads(raw_account)
except json.JSONDecodeError as exc:
raise AppError("That instance returned an account Rosies could not read.") from exc
account_id = str(account.get("id") or "")
if not account_id or not re.fullmatch(r"[A-Za-z0-9]+", account_id):
raise AppError("That instance did not return a public account id.")
statuses_url = (
f"https://{host}/api/v1/accounts/{account_id}/statuses?limit={limit}"
)
raw_statuses = fetch_retry(statuses_url, "application/json", UA_HONEST, log)
try:
statuses = json.loads(raw_statuses)
except json.JSONDecodeError as exc:
raise AppError("That instance returned posts Rosies could not read.") from exc
if not isinstance(statuses, list):
raise AppError("That instance did not return a public timeline.")
posts = []
for status in statuses[:limit]:
if not isinstance(status, dict):
continue
warning = (status.get("spoiler_text") or "").strip()
text = html_to_text(status.get("content") or "")
if warning:
text = f"[CW: {warning}] {text}".strip()
row = blank_post()
row.update(
{
"id": str(status.get("id") or ""),
"author": account.get("display_name") or account.get("username") or user,
"handle": f"@{account.get('acct') or user}@{host}",
"title": "",
"text": text,
"url": status.get("url") or "",
"createdAt": status.get("created_at") or "",
"likes": status.get("favourites_count"),
"replies": status.get("replies_count"),
"reposts": status.get("reblogs_count"),
}
)
posts.append(row)
return {
"profile": {
"name": account.get("display_name") or user,
"handle": f"@{account.get('acct') or user}@{host}",
"about": html_to_text(account.get("note") or "")[:500],
"url": account.get("url") or f"https://{host}/@{user}",
},
"posts": posts,
"note": "" if posts else "The account is public, but no posts came back.",
}
def scrape_hn(query: str, limit: int, log: list[dict]) -> dict:
raw = query.strip()
lowered = raw.lower()
if lowered.startswith("u:") or lowered.startswith("user:"):
name = raw.split(":", 1)[1].strip()
if not re.fullmatch(r"[A-Za-z0-9_-]{2,30}", name):
raise AppError("Enter a Hacker News username, like u:dang.")
tag = f"story,author_{name}"
url = (
"https://hn.algolia.com/api/v1/search_by_date?"
f"tags={urllib.parse.quote(tag)}&hitsPerPage={limit}"
)
label = f"u:{name}"
about = f"Public stories by {name}."
else:
if len(raw) < 2:
raise AppError("Enter a topic, or a user like u:dang.")
url = (
"https://hn.algolia.com/api/v1/search_by_date?"
f"query={urllib.parse.quote(raw)}&tags=story&hitsPerPage={limit}"
)
label = raw
about = f"Public stories matching “{raw}”."
body = fetch_retry(url, "application/json", UA_HONEST, log)
try:
payload = json.loads(body)
except json.JSONDecodeError as exc:
raise AppError("Hacker News returned a response Rosies could not read.") from exc
posts = []
for hit in (payload.get("hits") or [])[:limit]:
if not isinstance(hit, dict):
continue
object_id = str(hit.get("objectID") or "")
row = blank_post()
row.update(
{
"id": object_id,
"author": hit.get("author") or "",
"handle": hit.get("author") or "",
"title": hit.get("title") or "",
"text": "",
"url": hit.get("url") or (
f"https://news.ycombinator.com/item?id={object_id}" if object_id else ""
),
"createdAt": hit.get("created_at") or "",
"replies": hit.get("num_comments"),
"score": hit.get("points"),
}
)
posts.append(row)
return {
"profile": {
"name": "Hacker News",
"handle": label,
"about": about,
"url": "https://news.ycombinator.com/",
},
"posts": posts,
"note": "" if posts else "No public stories matched.",
}
def scrape_rss(query: str, limit: int, log: list[dict]) -> dict:
url = query.strip()
parsed = assert_public(url)
if not parsed.path or parsed.path == "/":
# A bare origin is still a valid feed location sometimes; allow it.
pass
body = fetch_retry(url, "application/rss+xml, application/atom+xml, application/xml, */*", UA_HONEST, log)
title, posts = parse_feed(body, limit)
host = parsed.hostname or url
return {
"profile": {
"name": title or host,
"handle": host,
"about": "Public RSS or Atom feed.",
"url": url,
},
"posts": posts,
"note": "" if posts else "The feed answered, but it had no entries.",
}
def parse_feed(xml_text: str, limit: int) -> tuple[str, list[dict]]:
try:
root = ET.fromstring(xml_text)
except ET.ParseError as exc:
raise AppError("That feed was not valid RSS or Atom.") from exc
title = ""
nodes: list[ET.Element] = []
if local(root.tag) == "rss":
channel = next((child for child in list(root) if local(child.tag) == "channel"), None)
if channel is not None:
title = child_text(channel, "title")
nodes = direct(channel, "item")
else:
title = child_text(root, "title")
nodes = [node for node in root.iter() if local(node.tag) == "entry"]
if not nodes:
nodes = [node for node in root.iter() if local(node.tag) == "item"]
posts = []
for node in nodes[:limit]:
author_el = next(iter(direct(node, "author")), None)
author = ""
if author_el is not None:
author = child_text(author_el, "name") or "".join(author_el.itertext()).strip()
if not author:
author = child_text(node, "creator")
content = (
child_text(node, "content")
or child_text(node, "description")
or child_text(node, "summary")
or child_text(node, "encoded")
)
row = blank_post()
row.update(
{
"id": child_text(node, "id") or child_text(node, "guid") or link_href(node),
"author": html_to_text(author),
"handle": html_to_text(author),
"title": html_to_text(child_text(node, "title")),
"text": html_to_text(content),
"url": link_href(node),
"createdAt": (
child_text(node, "updated")
or child_text(node, "published")
or child_text(node, "pubDate")
or child_text(node, "date")
),
}
)
posts.append(row)
return html_to_text(title), posts
def dispatch(source: str, query: str, limit: int) -> dict:
log: list[dict] = []
if source == "bluesky":
result = scrape_bluesky(query, limit, log)
elif source == "reddit":
result = scrape_reddit(query, limit, log)
elif source == "mastodon":
result = scrape_mastodon(query, limit, log)
elif source == "hackernews":
result = scrape_hn(query, limit, log)
elif source == "rss":
result = scrape_rss(query, limit, log)
else:
raise AppError("Pick a source Rosies knows.")
result["requests"] = log
return result
def emit(payload: dict) -> None:
sys.stdout.write(json.dumps(payload, ensure_ascii=False))
sys.stdout.write("\n")
def main() -> int:
raw = sys.stdin.read()
try:
incoming = json.loads(raw) if raw.strip() else {}
except json.JSONDecodeError:
emit({"ok": False, "error": "The desk sent an unreadable request."})
return 0
if not isinstance(incoming, dict):
emit({"ok": False, "error": "The desk sent an unreadable request."})
return 0
source = incoming.get("source")
query = incoming.get("query")
limit = incoming.get("limit", 12)
if source not in SOURCES:
emit({"ok": False, "error": "Pick a source Rosies knows."})
return 0
if not isinstance(query, str) or not query.strip():
emit({"ok": False, "error": "Enter something to pull."})
return 0
if isinstance(limit, bool) or not isinstance(limit, int) or not 1 <= limit <= 25:
emit({"ok": False, "error": "Ask for between 1 and 25 posts."})
return 0
started = time.perf_counter()
try:
result = dispatch(source, query.strip(), limit)
except AppError as exc:
emit({"ok": False, "error": str(exc)})
return 0
except Exception:
emit({"ok": False, "error": "The source did not answer in a way Rosies could read."})
return 0
elapsed = int((time.perf_counter() - started) * 1000)
emit(
{
"ok": True,
"engine": "python",
"engineDetail": f"Python {sys.version.split()[0]}",
"source": source,
"query": query.strip(),
"fetchedAt": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
"elapsedMs": elapsed,
"profile": result.get("profile"),
"posts": result.get("posts") or [],
"requests": result.get("requests") or [],
"note": result.get("note") or "",
}
)
return 0
if __name__ == "__main__":
raise SystemExit(main())