- Reset master to upstream/main (16,697 commits) - Overlay 2,271 local-only files (skills, tools, workspace, configs, apps) - Restore IDENTITY.md and USER.md templates - Build verified, gateway running, Discord working Co-Authored-By: Claude Opus 4.6 <[email protected]>
136 lines
4.0 KiB
Python
136 lines
4.0 KiB
Python
#!/usr/bin/env python3
|
|
"""Import Reddit deep-dive posts into the Kanban board as idea tasks.
|
|
|
|
- Reads JSONL produced by reddit_deep_dive.py
|
|
- Dedupe by permalink (also skips if permalink already exists in any existing task description)
|
|
- Creates tasks in queue, then PATCHes tags
|
|
|
|
Usage:
|
|
reddit_to_kanban.py /home/alex/clawd/notes/reddit-deep-dive-month.jsonl --max-per-sub 150
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import json
|
|
import re
|
|
import sys
|
|
import time
|
|
import urllib.request
|
|
from collections import defaultdict
|
|
|
|
|
|
KANBAN_API = "http://localhost:5003/api/tasks"
|
|
KANBAN_KEY = "openclaw-tasks-2026"
|
|
|
|
|
|
def http_json(method: str, url: str, payload: dict | None = None) -> dict:
|
|
data = None
|
|
headers = {
|
|
"User-Agent": "openclaw-reddit-to-kanban/1.0",
|
|
}
|
|
if payload is not None:
|
|
raw = json.dumps(payload).encode("utf-8")
|
|
data = raw
|
|
headers["Content-Type"] = "application/json"
|
|
headers["X-API-Key"] = KANBAN_KEY
|
|
req = urllib.request.Request(url, data=data, method=method, headers=headers)
|
|
with urllib.request.urlopen(req, timeout=25) as resp:
|
|
return json.loads(resp.read().decode("utf-8"))
|
|
|
|
|
|
def get_existing_permalinks() -> set[str]:
|
|
tasks = http_json("GET", KANBAN_API)
|
|
links: set[str] = set()
|
|
for t in tasks:
|
|
desc = (t.get("description") or "")
|
|
for m in re.finditer(r"https?://www\.reddit\.com/r/[^\s)]+", desc):
|
|
links.add(m.group(0).rstrip("/"))
|
|
return links
|
|
|
|
|
|
def main(argv: list[str]) -> int:
|
|
ap = argparse.ArgumentParser()
|
|
ap.add_argument("jsonl", help="path to reddit deep dive jsonl")
|
|
ap.add_argument("--max-per-sub", type=int, default=150)
|
|
ap.add_argument("--sleep", type=float, default=0.15)
|
|
args = ap.parse_args(argv[1:])
|
|
|
|
existing = get_existing_permalinks()
|
|
|
|
per_sub_count: dict[str, int] = defaultdict(int)
|
|
seen: set[str] = set()
|
|
|
|
created_ids: list[int] = []
|
|
|
|
with open(args.jsonl, "r", encoding="utf-8") as f:
|
|
for line in f:
|
|
d = json.loads(line)
|
|
sub = d.get("subreddit") or ""
|
|
permalink = (d.get("permalink") or "").rstrip("/")
|
|
if not sub or not permalink:
|
|
continue
|
|
if per_sub_count[sub] >= args.max_per_sub:
|
|
continue
|
|
if permalink in seen or permalink in existing:
|
|
continue
|
|
seen.add(permalink)
|
|
|
|
title = (d.get("title") or "").strip()
|
|
score = int(d.get("score") or 0)
|
|
comments = int(d.get("num_comments") or 0)
|
|
|
|
# Make title unique-ish without being ugly.
|
|
task_title = f"Idea (reddit/{sub}): {title}"
|
|
if len(task_title) > 160:
|
|
task_title = task_title[:157] + "…"
|
|
|
|
desc = (
|
|
f"Source: r/{sub} (top/month)\n"
|
|
f"Score: {score} | Comments: {comments}\n"
|
|
f"Link: {permalink}\n\n"
|
|
f"Extract an OpenClaw improvement idea from this thread."
|
|
)
|
|
|
|
task = http_json(
|
|
"POST",
|
|
KANBAN_API,
|
|
{
|
|
"title": task_title,
|
|
"description": desc,
|
|
"status": "queue",
|
|
# external_id intentionally omitted (avoid upsert)
|
|
},
|
|
)
|
|
task_id = int(task.get("id"))
|
|
|
|
tags = "idea,openclaw,reddit," + sub
|
|
http_json(
|
|
"PATCH",
|
|
f"{KANBAN_API}/{task_id}",
|
|
{
|
|
"tags": tags,
|
|
},
|
|
)
|
|
|
|
created_ids.append(task_id)
|
|
per_sub_count[sub] += 1
|
|
|
|
time.sleep(args.sleep)
|
|
|
|
print(
|
|
json.dumps(
|
|
{
|
|
"created": len(created_ids),
|
|
"created_ids": created_ids[:20],
|
|
"per_sub": dict(per_sub_count),
|
|
},
|
|
indent=2,
|
|
)
|
|
)
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main(sys.argv))
|