event-gun/publish.py
2026-10-03 14:12:14 +02:00

464 lines
No EOL
14 KiB
Python

#!/usr/bin/env python3
"""
Fetch ICS + RSS events from do.basspistol.org and publish them as
NIP-52 calendar events (kind 31923) to Nostr relays, requesting
inclusion in a kind 31924 calendar.
Usage:
publish.py # normal run
publish.py --dry-run # log what would be published, no DB writes, no relay calls
"""
import hashlib
import logging
import re
import sqlite3
import sys
import time
from datetime import datetime, timezone
from pathlib import Path
import feedparser
import requests
from icalendar import Calendar
from pynostr.event import Event
from pynostr.key import PrivateKey
from pynostr.relay_manager import RelayManager
try:
import tomllib
except ModuleNotFoundError:
import tomli as tomllib
SCRIPT_DIR = Path(__file__).resolve().parent
CONFIG_PATH = SCRIPT_DIR / "config.toml"
def load_config():
with open(CONFIG_PATH, "rb") as f:
return tomllib.load(f)
def setup_logging(log_path: Path):
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s [%(levelname)s] %(message)s",
handlers=[
logging.FileHandler(log_path, encoding="utf-8"),
logging.StreamHandler(sys.stdout),
],
)
log = logging.getLogger("nostr-calendar")
# ---------------------------------------------------------------------------
# Feed fetching / parsing
# ---------------------------------------------------------------------------
def fetch_text(url: str) -> str:
resp = requests.get(url, timeout=30, headers={"User-Agent": "nostr-calendar/1.0"})
resp.raise_for_status()
return resp.text
def parse_ics(text: str) -> list[dict]:
cal = Calendar.from_ical(text)
return [dict(c) for c in cal.walk("VEVENT")]
def parse_rss(text: str) -> dict:
feed = feedparser.parse(text)
by_url = {}
for entry in feed.entries:
link = entry.get("link", "")
if not link:
continue
image = ""
enclosures = entry.get("enclosures", [])
if enclosures:
image = enclosures[0].get("href", "")
if not image:
for l in entry.get("links", []):
if l.get("rel") == "enclosure":
image = l.get("href", "")
break
if not image:
m = re.search(r'<img[^>]+src="([^"]+)"', entry.get("description", ""))
if m:
image = m.group(1)
categories = [c.get("term", "") for c in entry.get("tags", []) if c.get("term")]
by_url[link] = {
"image": image,
"categories": categories,
"description": entry.get("description", ""),
}
return by_url
# ---------------------------------------------------------------------------
# ICS field helpers
# ---------------------------------------------------------------------------
def ics_str(component: dict, key: str) -> str:
val = component.get(key)
if val is None:
return ""
if hasattr(val, "to_ical"):
try:
raw = val.to_ical().decode("utf-8", errors="replace")
except Exception:
raw = str(val)
else:
raw = str(val)
# Undo ICS text escaping: \, \; \n \\
raw = (
raw.replace("\\,", ",")
.replace("\\;", ";")
.replace("\\n", "\n")
.replace("\\N", "\n")
.replace("\\\\", "\\")
)
return raw.strip()
def ics_dt(component: dict, key: str) -> int | None:
val = component.get(key)
if val is None:
return None
try:
dt = val.dt
if isinstance(dt, datetime):
if dt.tzinfo is None:
dt = dt.replace(tzinfo=timezone.utc)
return int(dt.timestamp())
return int(datetime(dt.year, dt.month, dt.day, tzinfo=timezone.utc).timestamp())
except Exception as e:
log.warning("Could not parse %s: %s", key, e)
return None
# ---------------------------------------------------------------------------
# Content / tags
# ---------------------------------------------------------------------------
def normalize_tag(tag: str) -> str:
tag = tag.strip().lower()
tag = re.sub(r"\s+", "-", tag)
tag = re.sub(r"[^\w\-]", "", tag)
return tag.strip("-")
def build_content(template: str, title: str, location: str, url: str) -> str:
return template.format(title=title, location=location, url=url)
def make_event_id(ics_url: str, uid: str) -> str:
m = re.match(r"^(\d+)@", uid)
if m:
return m.group(1)
return hashlib.sha256(uid.encode()).hexdigest()[:16]
def day_timestamps(start: int, end: int | None) -> list[int]:
days = [start // 86400]
if end is not None:
last = end // 86400
if last > days[0]:
days.extend(range(days[0] + 1, last + 1))
return days
# ---------------------------------------------------------------------------
# Dedupe
# ---------------------------------------------------------------------------
def init_db(path: Path) -> sqlite3.Connection:
conn = sqlite3.connect(path)
conn.execute(
"""
CREATE TABLE IF NOT EXISTS published (
d_tag TEXT PRIMARY KEY,
content_hash TEXT NOT NULL,
event_id TEXT,
published_at INTEGER NOT NULL
)
"""
)
conn.commit()
return conn
def content_hash(*parts) -> str:
h = hashlib.sha256()
for p in parts:
h.update(str(p).encode("utf-8"))
h.update(b"\x00")
return h.hexdigest()
def already_published(conn, d_tag: str, chash: str) -> bool:
row = conn.execute(
"SELECT content_hash FROM published WHERE d_tag = ?", (d_tag,)
).fetchone()
return row is not None and row[0] == chash
def record_published(conn, d_tag: str, chash: str, event_id: str):
conn.execute(
"INSERT OR REPLACE INTO published (d_tag, content_hash, event_id, published_at) VALUES (?, ?, ?, ?)",
(d_tag, chash, event_id, int(time.time())),
)
conn.commit()
# ---------------------------------------------------------------------------
# Event construction
# ---------------------------------------------------------------------------
def make_calendar_event(cfg: dict, pubkey_hex: str, event_d_tags: list[str]) -> Event:
"""Build the kind 31924 calendar, listing every event via an `a` tag."""
cal = cfg["calendar"]
tags = [
["d", cal["d"]],
["title", cal["title"]],
]
for d_tag in event_d_tags:
tags.append(["a", f"31923:{pubkey_hex}:{d_tag}"])
return Event(
kind=31924,
content=cal["description"],
tags=tags,
)
def make_calendar_event_nostr(cfg, pubkey_hex, title, start, end, location, url,
image, tags, d_tag, duration, template) -> Event:
"""Build a kind 31923 event."""
cal_coord = f"31924:{pubkey_hex}:{cfg['calendar']['d']}"
final_end = end if end is not None else start + duration
ev_tags = [
["d", d_tag],
["title", title],
["start", str(start)],
["end", str(final_end)],
["a", cal_coord],
]
if location:
ev_tags.append(["location", location])
if image:
ev_tags.append(["image", image])
for t in tags:
ev_tags.append(["t", t])
for day in day_timestamps(start, final_end):
ev_tags.append(["D", str(day)])
return Event(
kind=31923,
content=build_content(template, title, location, url),
tags=ev_tags,
)
# ---------------------------------------------------------------------------
# Relay diagnostics
# ---------------------------------------------------------------------------
def wait_for_relays(relay_manager, timeout: float = 10.0) -> bool:
"""Wait until all relays report connected=True, or timeout elapses.
Returns True if all connected, False otherwise.
"""
deadline = time.time() + timeout
while time.time() < deadline:
states = {url: r.connected for url, r in relay_manager.relays.items()}
if states and all(states.values()):
log.info("All relays connected: %s", states)
return True
time.sleep(0.25)
states = {url: r.connected for url, r in relay_manager.relays.items()}
log.warning("Timed out waiting for relays. States: %s", states)
return False
def log_relay_states(relay_manager):
"""Log connected state and any messages the relays sent back."""
for url, relay in relay_manager.relays.items():
log.info("Relay %s — connected=%s", url, relay.connected)
pool = getattr(relay, "message_pool", None)
if pool is None:
continue
queue = getattr(pool, "queue", None)
if not queue:
continue
while queue:
msg = queue.pop(0)
try:
log.info(" %s: %s", url, msg)
except Exception:
pass
# ---------------------------------------------------------------------------
# Main
# ---------------------------------------------------------------------------
def main():
dry_run = "--dry-run" in sys.argv
cfg = load_config()
setup_logging(SCRIPT_DIR / cfg.get("log_path", "publish.log"))
if dry_run:
log.info("=== DRY RUN — no DB writes, no relay publishing ===")
secret = cfg["secret_key"]
relays = cfg["relays"]
ics_url = cfg["ics_url"]
rss_url = cfg["rss_url"]
duration = int(cfg.get("default_duration", 3600))
template = cfg.get("content_template", "{title}\n📍 {location}\n🔗 {url}")
rss_only = bool(cfg.get("rss_only", False))
pk = PrivateKey.from_nsec(secret) if secret.startswith("nsec") else PrivateKey(bytes.fromhex(secret))
pubkey_hex = pk.public_key.hex()
log.info("Using pubkey %s", pubkey_hex)
log.info("Relays: %s", ", ".join(relays))
log.info("rss_only=%s", rss_only)
log.info("Fetching ICS …")
ics_text = fetch_text(ics_url)
log.info("Fetching RSS …")
rss_text = fetch_text(rss_url)
log.info("Parsing ICS …")
vevents = parse_ics(ics_text)
log.info("Parsing RSS …")
rss_map = parse_rss(rss_text)
log.info("Found %d ICS events, %d RSS entries", len(vevents), len(rss_map))
conn = init_db(SCRIPT_DIR / cfg.get("db_path", "events.db"))
to_publish = []
skipped = 0
skipped_no_rss = 0
dry_run_count = 0
for vev in vevents:
uid = ics_str(vev, "UID")
if not uid:
continue
title = ics_str(vev, "SUMMARY")
url = ics_str(vev, "URL")
location = ics_str(vev, "LOCATION")
start = ics_dt(vev, "DTSTART")
end = ics_dt(vev, "DTEND")
categories_ics = ics_str(vev, "CATEGORIES")
if start is None:
log.warning("Skipping %s — no DTSTART", uid)
continue
rss = rss_map.get(url, {})
if rss_only and not rss:
skipped_no_rss += 1
continue
image = rss.get("image", "")
rss_cats = rss.get("categories", [])
if categories_ics:
raw_tags = [c.strip() for c in categories_ics.split(",") if c.strip()]
else:
raw_tags = rss_cats
tags = [normalize_tag(t) for t in raw_tags if normalize_tag(t)]
d_tag = make_event_id(ics_url, uid)
chash = content_hash(title, start, end, location, url, image, ",".join(tags))
if dry_run:
log.info("DRY RUN %s: title=%r image=%r tags=%r start=%d end=%s",
d_tag, title, image, tags, start, end)
dry_run_count += 1
continue
if already_published(conn, d_tag, chash):
skipped += 1
continue
ev = make_calendar_event_nostr(
cfg, pubkey_hex, title, start, end, location, url,
image, tags, d_tag, duration, template,
)
to_publish.append((d_tag, chash, ev))
if dry_run:
log.info("DRY RUN complete. Would publish %d, skipped_no_rss=%d",
dry_run_count, skipped_no_rss)
return
log.info("Publishing %d new/changed events (skipped %d unchanged, %d without RSS)",
len(to_publish), skipped, skipped_no_rss)
# Build the full set of d-tags the calendar should reference:
# everything already in the DB, plus everything about to be published.
rows = conn.execute("SELECT d_tag FROM published").fetchall()
all_d_tags = {row[0] for row in rows}
for d_tag, _, _ in to_publish:
all_d_tags.add(d_tag)
all_d_tags = sorted(all_d_tags)
log.info("Calendar will reference %d events", len(all_d_tags))
cal_ev = make_calendar_event(cfg, pubkey_hex, all_d_tags)
cal_ev.sign(pk.hex())
# Open relay connections and wait until they're actually connected
# before pushing messages over them.
relay_manager = RelayManager(timeout=30)
for r in relays:
relay_manager.add_relay(r)
if not wait_for_relays(relay_manager, timeout=10.0):
log.warning("Proceeding despite not all relays being connected.")
# Publish the calendar FIRST so it isn't stuck behind a queue of events.
# Relays don't enforce ordering between an `a` reference and the event
# it points to; clients resolve references when they query.
relay_manager.publish_event(cal_ev)
log.info("Queued calendar kind 31924 (d=%s) referencing %d events",
cfg["calendar"]["d"], len(all_d_tags))
# Then publish the events.
for d_tag, chash, ev in to_publish:
ev.sign(pk.hex())
relay_manager.publish_event(ev)
log.info("Queued event %s — %s", d_tag, ev.content.split("\n", 1)[0])
relay_manager.run_sync()
time.sleep(10)
# Show what the relays said before we tear down the connections.
log_relay_states(relay_manager)
relay_manager.close_all_relay_connections()
for d_tag, chash, ev in to_publish:
record_published(conn, d_tag, chash, ev.id)
log.info("Done. %d events published, calendar references %d.",
len(to_publish), len(all_d_tags))
if __name__ == "__main__":
main()