Files
openclaw/tools/reddit_to_kanban.py
Clawd BotandClaude Opus 4.6 ca9b510922 chore: align with upstream openclaw/openclaw and overlay local additions
- Reset master to upstream/main (16,697 commits)
- Overlay 2,271 local-only files (skills, tools, workspace, configs, apps)
- Restore IDENTITY.md and USER.md templates
- Build verified, gateway running, Discord working

Co-Authored-By: Claude Opus 4.6 <[email protected]>
2026-03-03 07:40:46 +01:00

136 lines
4.0 KiB
Python

#!/usr/bin/env python3
"""Import Reddit deep-dive posts into the Kanban board as idea tasks.
- Reads JSONL produced by reddit_deep_dive.py
- Dedupe by permalink (also skips if permalink already exists in any existing task description)
- Creates tasks in queue, then PATCHes tags
Usage:
reddit_to_kanban.py /home/alex/clawd/notes/reddit-deep-dive-month.jsonl --max-per-sub 150
"""
from __future__ import annotations
import argparse
import json
import re
import sys
import time
import urllib.request
from collections import defaultdict
KANBAN_API = "http://localhost:5003/api/tasks"
KANBAN_KEY = "openclaw-tasks-2026"
def http_json(method: str, url: str, payload: dict | None = None) -> dict:
data = None
headers = {
"User-Agent": "openclaw-reddit-to-kanban/1.0",
}
if payload is not None:
raw = json.dumps(payload).encode("utf-8")
data = raw
headers["Content-Type"] = "application/json"
headers["X-API-Key"] = KANBAN_KEY
req = urllib.request.Request(url, data=data, method=method, headers=headers)
with urllib.request.urlopen(req, timeout=25) as resp:
return json.loads(resp.read().decode("utf-8"))
def get_existing_permalinks() -> set[str]:
tasks = http_json("GET", KANBAN_API)
links: set[str] = set()
for t in tasks:
desc = (t.get("description") or "")
for m in re.finditer(r"https?://www\.reddit\.com/r/[^\s)]+", desc):
links.add(m.group(0).rstrip("/"))
return links
def main(argv: list[str]) -> int:
ap = argparse.ArgumentParser()
ap.add_argument("jsonl", help="path to reddit deep dive jsonl")
ap.add_argument("--max-per-sub", type=int, default=150)
ap.add_argument("--sleep", type=float, default=0.15)
args = ap.parse_args(argv[1:])
existing = get_existing_permalinks()
per_sub_count: dict[str, int] = defaultdict(int)
seen: set[str] = set()
created_ids: list[int] = []
with open(args.jsonl, "r", encoding="utf-8") as f:
for line in f:
d = json.loads(line)
sub = d.get("subreddit") or ""
permalink = (d.get("permalink") or "").rstrip("/")
if not sub or not permalink:
continue
if per_sub_count[sub] >= args.max_per_sub:
continue
if permalink in seen or permalink in existing:
continue
seen.add(permalink)
title = (d.get("title") or "").strip()
score = int(d.get("score") or 0)
comments = int(d.get("num_comments") or 0)
# Make title unique-ish without being ugly.
task_title = f"Idea (reddit/{sub}): {title}"
if len(task_title) > 160:
task_title = task_title[:157] + "…"
desc = (
f"Source: r/{sub} (top/month)\n"
f"Score: {score} | Comments: {comments}\n"
f"Link: {permalink}\n\n"
f"Extract an OpenClaw improvement idea from this thread."
)
task = http_json(
"POST",
KANBAN_API,
{
"title": task_title,
"description": desc,
"status": "queue",
# external_id intentionally omitted (avoid upsert)
},
)
task_id = int(task.get("id"))
tags = "idea,openclaw,reddit," + sub
http_json(
"PATCH",
f"{KANBAN_API}/{task_id}",
{
"tags": tags,
},
)
created_ids.append(task_id)
per_sub_count[sub] += 1
time.sleep(args.sleep)
print(
json.dumps(
{
"created": len(created_ids),
"created_ids": created_ids[:20],
"per_sub": dict(per_sub_count),
},
indent=2,
)
)
return 0
if __name__ == "__main__":
raise SystemExit(main(sys.argv))