Files
pulse-bridge/bridge.py
T

841 lines
35 KiB
Python
Raw Blame History

This file contains invisible Unicode characters
This file contains invisible Unicode characters that are indistinguishable to humans but may be processed differently by a computer. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""pulse-bridge v1 — mirror the estate's shared work ledger into Pulse posts.
Source: ~/.local/bin/shre-items (JSONL ledger, `feed --since` op stream).
Target: local mib007 (:5520) unified message ENVELOPE API (/api/workspaces/
:ws/messages) — the store the Pulse / Activity feed now reads AND writes
(useActivity.ts stage 3). One ledger item = one post in the "activity"
channel; stage changes / failures / escalations = threaded comments; close =
a final comment.
v0 wrote to the legacy comms channels API. The migration was seamless for
state because mib007's backfill (0103) and mirror trigger (0104) PRESERVE
message ids comms -> envelope, so the existing ledger-id -> post-id map keeps
pointing at valid envelope messages and old threads keep accepting replies.
Two envelope-specific notes:
- Idempotency is server-side now: every write carries an idempotencyKey and a
replay returns the ORIGINAL message (201, same id), so retries after
insert-then-crash can neither duplicate posts nor comments. The v0
find-existing recovery scan (bounded page, can't prove absence) is gone.
- Attribution comes from the AUTHENTICATED actor, never the body. The comms
`userName: "Ledger"` byline override does not exist on the envelope route,
so envelope-native bridge posts render under the service actor (the UI
falls back to "User"). Known cosmetic tradeoff, documented in README.
Stdlib only. State (checkpoint + ledger-id -> post-id map) lives in
~/.shre/pulse-bridge/state.json, written atomically (tmp + rename).
"""
import json
import os
import re
import shutil
import subprocess
import sys
import time
import urllib.error
import urllib.request
from datetime import datetime, timezone
from pathlib import Path
# ── Config (env file, then environment, then defaults) ───────────────────────
HOME = Path.home()
ENV_FILE = Path(os.environ.get("PULSE_BRIDGE_ENV",
Path(__file__).resolve().parent / "bridge.env"))
def _load_env_file(path: Path) -> dict:
conf = {}
if path.exists():
for line in path.read_text(encoding="utf-8").splitlines():
line = line.strip()
if not line or line.startswith("#") or "=" not in line:
continue
k, v = line.split("=", 1)
conf[k.strip()] = v.strip().strip('"').strip("'")
return conf
_FILE = _load_env_file(ENV_FILE)
def cfg(key: str, default: str) -> str:
return os.environ.get(key) or _FILE.get(key) or default
POLL_SEC = int(cfg("PULSE_BRIDGE_POLL_SEC", "30"))
MIB_BASE = cfg("PULSE_BRIDGE_MIB_BASE", "http://127.0.0.1:5520").rstrip("/")
WORKSPACE_ID = cfg("PULSE_BRIDGE_WORKSPACE_ID",
"293d29db-4978-4c54-a92e-b24d4c0a7115") # Nirlab Command Center
CHANNEL_NAME = cfg("PULSE_BRIDGE_CHANNEL", "activity")
MENTION = cfg("PULSE_BRIDGE_MENTION", "nir") # user handle to @mention
# Historical byline. The envelope route has no userName override (attribution
# is the authenticated actor), so this no longer names the posts; it is kept
# because the intake-ignore default below references it and legacy posts
# carry it as actorDisplayName.
POSTER_NAME = cfg("PULSE_BRIDGE_POSTER_NAME", "Ledger")
SHRE_ITEMS = cfg("PULSE_BRIDGE_SHRE_ITEMS", str(HOME / ".local/bin/shre-items"))
STATE_FILE = Path(cfg("PULSE_BRIDGE_STATE",
str(HOME / ".shre/pulse-bridge/state.json")))
TOKEN_FILE = Path(cfg("PULSE_BRIDGE_TOKEN_FILE",
str(HOME / ".shre/service-tokens.json")))
TOKEN_KEY = cfg("PULSE_BRIDGE_TOKEN_KEY", "mib007")
# The envelope attributes writes to the AUTHENTICATED actor. The mib007
# service token authenticates as this fixed board actor (middleware/auth.ts),
# which is how the bridge recognises (and never re-ingests) its own posts.
SELF_ACTOR_ID = cfg("PULSE_BRIDGE_SELF_ACTOR_ID",
"00000000-0000-0000-0000-000000000001")
STAGES = ["queued", "build", "review", "merge", "deploy", "verify"]
KIND_LABELS = {
"needs-you": "human needed",
"failed": "gate failed",
"pipeline": "machine owns it",
"recommended": "idea",
"gap": "machine owns it",
}
STAGE_LABELS = {
"queued": "waiting for agent",
"build": "agent working",
"review": "ready for review",
"merge": "ready to merge",
"deploy": "ready to deploy",
"verify": "prove it live",
}
STAGE_ACTIONS = {
"queued": "wait",
"build": "run",
"review": "review",
"merge": "merge",
"deploy": "deploy",
"verify": "verify",
}
# ── Intake (reverse direction: Pulse post -> ledger item + kanban task) ──────
INTAKE_ENABLED = cfg("PULSE_BRIDGE_INTAKE", "1") == "1"
INTAKE_MAX_CONCURRENT = int(cfg("PULSE_BRIDGE_INTAKE_MAX_CONCURRENT", "3"))
HERMES_DIR = cfg("PULSE_BRIDGE_HERMES_DIR", str(HOME / ".hermes/hermes-agent"))
HERMES_HOME_DIR = cfg("PULSE_BRIDGE_HERMES_HOME", str(HOME / ".hermes"))
KANBAN_MIRROR_ENABLED = cfg("PULSE_BRIDGE_KANBAN_MIRROR", "0") == "1"
KANBAN_MIRROR_KINDS = {k.strip() for k in
cfg("PULSE_BRIDGE_KANBAN_KINDS",
"needs-you,failed,pipeline,gap").split(",")
if k.strip()}
KANBAN_MIRROR_TRIAGE = cfg("PULSE_BRIDGE_KANBAN_TRIAGE", "1") == "1"
KANBAN_MIRROR_INITIAL_STATUS = cfg("PULSE_BRIDGE_KANBAN_INITIAL_STATUS",
"").strip()
# user_names never ingested (service accounts / the bridge itself)
INTAKE_IGNORE_USERS = {u.strip().lower() for u in
cfg("PULSE_BRIDGE_INTAKE_IGNORE",
f"{POSTER_NAME},Ellie,AROS").split(",") if u.strip()}
# chat-only opt-out: posts starting with any of these prefixes (matched
# case-insensitively) are just talk — no ledger item, no task, no comments
INTAKE_OPTOUT_PREFIXES = tuple(
p.strip().lower() for p in
cfg("PULSE_BRIDGE_INTAKE_OPTOUT", "💬,chat:").split(",") if p.strip())
def log(msg: str) -> None:
print(f"{datetime.now(timezone.utc).isoformat(timespec='seconds')} {msg}",
flush=True)
# ── mib007 HTTP ──────────────────────────────────────────────────────────────
def _token() -> str:
return json.loads(TOKEN_FILE.read_text(encoding="utf-8"))[TOKEN_KEY]
def api(method: str, path: str, body=None, timeout=15):
req = urllib.request.Request(
f"{MIB_BASE}{path}",
data=json.dumps(body).encode() if body is not None else None,
method=method,
headers={
"Authorization": f"Bearer {_token()}",
"Content-Type": "application/json",
},
)
with urllib.request.urlopen(req, timeout=timeout) as resp:
raw = resp.read()
return json.loads(raw) if raw else None
# ── State ────────────────────────────────────────────────────────────────────
def load_state() -> dict:
if STATE_FILE.exists():
try:
return json.loads(STATE_FILE.read_text(encoding="utf-8"))
except json.JSONDecodeError:
# Starting fresh here would re-seed every open item -> a duplicate
# post per item. Refuse to run until a human inspects the file.
log(f"FATAL state file corrupt: {STATE_FILE} — fix or remove it "
"deliberately; refusing to reseed automatically")
sys.exit(1)
return {"checkpoint": "", "seen": [], "map": {}, "kanban": {},
"seeded": False, "channel_id": "", "intake_checkpoint": 0,
"intake": {}}
def save_state(state: dict) -> None:
STATE_FILE.parent.mkdir(parents=True, exist_ok=True)
tmp = STATE_FILE.with_suffix(".json.tmp")
with open(tmp, "w", encoding="utf-8") as f:
json.dump(state, f, indent=1)
f.flush()
os.fsync(f.fileno())
os.replace(tmp, STATE_FILE)
# ── Content helpers ──────────────────────────────────────────────────────────
def neutralize(text: str) -> str:
"""Insert a zero-width space after '@' in ledger-derived text.
mib007 treats the FIRST '@handle' in a message as a mention — the envelope
route runs the SAME dispatchMention side-effect the comms route did: a
matching agent fires an AI reply into the thread, and a matching
workspace-member user gets a notification. We neutralize ledger-DERIVED
text (titles, details, notes) because arbitrary item text containing
"@aros"/"@ellie" must not accidentally fire an agent reply into the feed.
The bridge's own needs-you escalation line is the one place that
deliberately carries a live @handle (see update_body).
"""
return (text or "").replace("@", "@")
def lifecycle(item: dict) -> dict:
kind = item.get("kind", "item")
stage = item.get("stage")
out = {
"kind": KIND_LABELS.get(kind, kind),
"owner": "human" if kind == "needs-you" else "machine",
"state": "open",
"action": "none",
}
if kind == "recommended":
out.update(owner="human", state="idea", action="optional")
elif kind == "needs-you":
out.update(state="waiting-on-human", action="answer")
elif kind == "failed":
out.update(state="failed", action="fix-or-retry")
elif stage:
out.update(state=stage, action=STAGE_ACTIONS.get(stage, "run"))
if stage:
out["stage"] = STAGE_LABELS.get(stage, stage)
return out
def lifecycle_line(item: dict) -> str:
facts = lifecycle(item)
bits = [f"{key}:{value}" for key, value in facts.items()]
if item.get("stage"):
bits.append(f"stage_key:{item['stage']}")
return " · ".join(bits)
def tag_line(item: dict) -> str:
bits = [f"#{item.get('kind', 'item')}"]
if item.get("stage"):
bits.append(f"#{item['stage']}")
facts = lifecycle(item)
bits.extend([
f"owner:{facts['owner']}",
f"state:{facts['state']}",
f"action:{facts['action']}",
])
if item.get("stage"):
bits.append(f"stage:{item['stage']}")
for t in item.get("tags", []) or []:
bits.append(f"#{t}")
if item.get("surface"):
bits.append(f"surface:{item['surface']}")
if item.get("project"):
bits.append(f"project:{item['project']}")
bits.append(f"ledger:{item['id']}")
return " · ".join(bits)
def post_body(item: dict) -> str:
head = {"needs-you": "🙋", "failed": "⛔", "pipeline": "🔧",
"recommended": "💡", "gap": "🕳"}.get(item.get("kind", ""), "📌")
parts = [f"{head} {neutralize(item.get('title', '(untitled)'))}"]
if item.get("detail"):
parts.append(neutralize(item["detail"]))
parts.append(lifecycle_line(item))
parts.append(tag_line(item))
return "\n\n".join(parts)
def update_body(rec: dict) -> str:
lines = []
kind = rec.get("kind")
stage = rec.get("stage")
note = neutralize(rec.get("note", ""))
if kind == "needs-you":
# Deliberately a LIVE mention (not neutralized): mib007 resolves
# @<lower(user name)> against active workspace members and creates a
# comment.mention notification, so the escalation actually pings the
# human. Everything else in the line stays neutralized.
lines.append(f"@{MENTION} NEEDS YOU: this item is now"
" waiting on a human." + (f" — {note}" if note else ""))
note = "" # already included
elif kind == "failed":
where = f" at {stage} ({STAGE_LABELS.get(stage, stage)})" if stage else ""
attempt = f", attempt {rec['attempt']}" if rec.get("attempt") else ""
lines.append(f"⛔ failed{where}{attempt}" + (f" — {note}" if note else ""))
note = ""
else:
if kind:
lines.append(f"kind → {kind} ({KIND_LABELS.get(kind, kind)})")
if stage:
lines.append(f"stage → {stage} ({STAGE_LABELS.get(stage, stage)})")
if rec.get("attempt") and not kind:
lines.append(f"attempt {rec['attempt']}")
if rec.get("detail"):
lines.append(f"detail updated: {neutralize(rec['detail'])}")
if note:
lines.append(note)
return "\n".join(lines) or "(updated)"
def close_body(rec: dict) -> str:
why = neutralize(rec.get("why", ""))
if rec.get("status") == "dropped":
return "🗑 dropped" + (f" — {why}" if why else "")
return "✅ done" + (f" — {why}" if why else "")
# ── Ledger access ────────────────────────────────────────────────────────────
def shre_items(*args: str) -> str:
out = subprocess.run([SHRE_ITEMS, *args], capture_output=True, text=True,
timeout=30)
if out.returncode != 0:
raise RuntimeError(f"shre-items {' '.join(args)} rc={out.returncode}: "
f"{out.stderr.strip()[:300]}")
return out.stdout
def feed_since(checkpoint: str) -> list:
args = ["feed"] + (["--since", checkpoint] if checkpoint else [])
records = []
for line in shre_items(*args).splitlines():
line = line.strip()
if not line:
continue
try:
records.append(json.loads(line))
except json.JSONDecodeError:
continue
records.sort(key=lambda r: r.get("at", ""))
return records
def open_items() -> list:
return json.loads(shre_items("list", "--json") or "[]")
# ── Pulse actions (unified message envelope) ─────────────────────────────────
def ensure_channel(state: dict):
"""Resolve the activity channel's envelope id (cached in state).
Comms and envelope channel ids are IDENTICAL for pre-envelope channels
(backfill 0103 preserves ids), so a v0 state file's cached channel_id is
already the envelope id. A missing channel is not created here —
send_message() falls back to channelSlug, which the envelope route
find-or-creates, and the id is adopted from the response.
"""
if state.get("channel_id"):
return state["channel_id"]
channels = api("GET",
f"/api/workspaces/{WORKSPACE_ID}/messages/channels") or []
want = CHANNEL_NAME.lower()
for ch in channels:
if (ch.get("slug") or "").lower() == want \
or (ch.get("name") or "").lower() == want:
state["channel_id"] = ch["id"]
save_state(state)
return ch["id"]
return None
def send_message(state: dict, content: str, thread_id=None, idem=None) -> dict:
"""POST one message into the envelope.
idem is the server-side idempotency key: a replay after a crash or an
insert-then-error returns the ORIGINAL message with its original id, so
every caller can safely retry. Attribution is the authenticated actor;
there is no userName override on this route.
"""
body = {"body": content}
channel_id = ensure_channel(state)
if channel_id:
body["channelId"] = channel_id
else:
body["channelSlug"] = CHANNEL_NAME # find-or-create, id adopted below
if thread_id:
body["threadId"] = thread_id
if idem:
body["idempotencyKey"] = idem
# The envelope route runs side-effects (audit, mention dispatch) before
# responding and can exceed the default 15s under load; a timeout here is
# retried next poll and the idempotency key de-duplicates the replay.
msg = api("POST", f"/api/workspaces/{WORKSPACE_ID}/messages", body,
timeout=30)
if not state.get("channel_id") and msg.get("channelId"):
state["channel_id"] = msg["channelId"]
save_state(state)
return msg
def find_legacy_post(state: dict, ledger_id: str):
"""Adopt an unmapped v0-era post before creating (codex-flagged).
v0 wrote via comms WITHOUT idempotency keys, so a v0 insert-then-error
crash could have left a mirrored post that the map never recorded — the
ledger:<id> idempotency key cannot replay/adopt such a row, and blindly
creating would duplicate it. Bounded to the latest 100 roots: a miss
cannot prove absence (documented trap), but a miss merely creates — and
every v1 row carries the idempotency key, so v1-era retries never reach
this path with a duplicate risk.
"""
needle = f"ledger:{ledger_id}"
try:
for m in channel_roots(state):
if needle in (m.get("content") or ""):
return m["id"]
except Exception:
return None # recovery is best-effort; creation still idempotent
return None
def create_post(state: dict, item: dict, *, mirror_kanban: bool = True) -> None:
if item["id"] in state["map"]:
if mirror_kanban:
ensure_kanban_mirror(state, item)
return # already mirrored (seed/feed overlap)
legacy = find_legacy_post(state, item["id"])
if legacy:
state["map"][item["id"]] = legacy
log(f"post {item['id'][:8]} -> adopted legacy {legacy}")
if mirror_kanban:
ensure_kanban_mirror(state, item)
return
# ledger:<id> as the idempotency key: a retry after insert-then-crash gets
# the original post back (same id), so no duplicate and no recovery scan.
msg = send_message(state, post_body(item), idem=f"ledger:{item['id']}")
state["map"][item["id"]] = msg["id"]
log(f"post {item['id'][:8]} -> {msg['id']} ({item.get('title', '')[:60]})")
if mirror_kanban:
ensure_kanban_mirror(state, item)
def comment(state: dict, ledger_id: str, content: str, what: str,
idem=None) -> None:
post_id = state["map"].get(ledger_id)
if not post_id:
log(f"SKIP {what} {ledger_id[:8]}: no mapped post (predates bridge)")
return
try:
msg = send_message(state, content, thread_id=post_id, idem=idem)
except urllib.error.HTTPError as e:
# 400 here means the thread target is invalid (deleted/foreign post) —
# retrying can never succeed, same as 404/410. 5xx IS retried now:
# the idempotency key makes a replay return the original comment, so
# the v0 drop-on-500 duplicate guard is no longer needed.
if e.code in (400, 404, 410):
log(f"DROP {what} {ledger_id[:8]}: HTTP {e.code} "
"(comment not retryable)")
return
raise
log(f"{what:6} {ledger_id[:8]} -> comment {msg['id']}")
# ── Intake: Pulse posts -> ledger + kanban ───────────────────────────────────
def kanban_cmd() -> list[str]:
"""How to invoke Dash's kanban CLI, resolved fresh on every call.
The shared checkout's venv is not stable: a release cutover can delete or
replace it mid-run (observed 2026-08-23 — every intake tick failed with
ENOENT on `<checkout>/venv/bin/python`). The `dash-agent` / shim entrypoint is
repointed by the release recipe itself, so it is the durable entry point;
the checkout venv stays as a fallback for installs that have no shim.
"""
override = cfg("PULSE_BRIDGE_HERMES_PY", "")
if override:
return [override, "-m", "shiva_cli.main"]
dash_agent = cfg("PULSE_BRIDGE_DASH_AGENT", "dash-agent")
if dash_agent and shutil.which(dash_agent):
return [dash_agent]
dash_shim = HOME / ".local/bin/dash-agent"
if os.access(dash_shim, os.X_OK):
return [str(dash_shim)]
shim = HOME / ".local/bin/hermes"
if os.access(shim, os.X_OK):
return [str(shim)]
return [f"{HERMES_DIR}/venv/bin/python", "-m", "shiva_cli.main"]
def kanban(*args: str, timeout: int = 60) -> str:
cmd = kanban_cmd()
cwd = HERMES_DIR if os.path.isdir(HERMES_DIR) else str(HOME)
out = subprocess.run(
[*cmd, "kanban", *args],
capture_output=True, text=True, timeout=timeout, cwd=cwd,
env={**os.environ, "HERMES_HOME": HERMES_HOME_DIR})
if out.returncode != 0:
raise RuntimeError(f"kanban {args[0]} rc={out.returncode}: "
f"{out.stderr.strip()[:300]}")
return out.stdout
def ensure_kanban_mirror(state: dict, item: dict) -> None:
"""Optionally mirror outbound ledger items into Dash Kanban."""
if not KANBAN_MIRROR_ENABLED:
return
if item.get("kind") not in KANBAN_MIRROR_KINDS:
return
state.setdefault("kanban", {})
if state["kanban"].get(item["id"]):
return
post_id = state.get("map", {}).get(item["id"], "")
title = f"{item.get('kind', 'item')}: {item.get('title', '(untitled)')}"
body = "\n\n".join(x for x in (
neutralize(item.get("detail", "")),
lifecycle_line(item),
f"ledger:{item['id']}",
f"pulse:{post_id}" if post_id else "",
) if x)
args = ["create", title[:120], "--body", body,
"--created-by", "pulse-bridge", "--assignee", "default",
"--idempotency-key", f"shre-items-{item['id']}", "--json"]
if KANBAN_MIRROR_TRIAGE:
args.append("--triage")
elif KANBAN_MIRROR_INITIAL_STATUS:
args.extend(["--initial-status", KANBAN_MIRROR_INITIAL_STATUS])
raw = kanban(*args)
task = json.loads(raw)
task_id = task["id"] if isinstance(task, dict) else task[0]["id"]
state["kanban"][item["id"]] = task_id
save_state(state)
log(f"kanban {item['id'][:8]} -> {task_id}")
def ledger_add_for_post(post: dict) -> str:
"""Create (or adopt) a pipeline ledger item for a Pulse post; return its
FULL id. Idempotent: a crash after `shre-items add` but before the state
save must not create a second item on retry, so existing open items are
searched for this post's marker first."""
marker = f"post:{post['id']}"
for item in open_items():
if marker in (item.get("detail") or ""):
return item["id"]
content = (post.get("content") or "").strip()
title = " ".join(content.split())[:80] or "(untitled pulse post)"
detail = f"{content}\n\n[pulse-intake] {marker} " \
f"author:{post.get('user_name') or post.get('user_id')}"
out = shre_items("add", "--kind", "pipeline", "--stage", "queued",
"--title", title, "--detail", detail,
"--surface", "pulse", "--tag", "pulse-intake")
short = out.split()[0]
for item in open_items():
if item["id"].startswith(short):
return item["id"]
raise RuntimeError(f"cannot resolve full ledger id for {short}")
def intake_active_count(state: dict) -> int:
return sum(1 for e in state["intake"].values()
if e.get("task_id") and not e.get("closed"))
def intake_start_task(state: dict, post_id: str, entry: dict) -> None:
"""Create the kanban task for an intake entry and acknowledge on the post."""
content = entry.get("content", "")
title = " ".join(content.split())[:80] or "(pulse task)"
# Crash recovery leans on the CLI's documented idempotency contract:
# "If a non-archived task with this key exists, its id is returned
# instead of creating a duplicate" — so re-calling after a crash between
# create and save_state adopts the existing task rather than forking one.
raw = kanban("create", title,
"--body", f"{content}\n\n(from Pulse post {post_id}, "
f"ledger {entry['ledger_id'][:8]})",
"--created-by", "pulse-bridge", "--assignee", "default",
"--idempotency-key", f"pulse-{post_id}", "--json")
task = json.loads(raw)
task_id = task["id"] if isinstance(task, dict) else task[0]["id"]
entry["task_id"] = task_id
entry["last_status"] = task.get("status", "todo") if isinstance(task, dict) else "todo"
entry.pop("queued", None)
save_state(state)
log(f"intake {post_id[:8]} -> kanban {task_id}")
try:
send_message(state,
f"🤖 picked up — ledger {entry['ledger_id'][:8]}, "
f"kanban {task_id}", thread_id=post_id,
idem=f"pulse-ack:{post_id}")
except urllib.error.HTTPError as e:
log(f"DROP ack {post_id[:8]}: HTTP {e.code}")
def _created_ms(iso: str) -> int:
"""Envelope timestamps are ISO-8601; the intake checkpoint stays epoch-ms
so a v0 state file's checkpoint keeps meaning the same instant."""
if not iso:
return 0
try:
return int(datetime.fromisoformat(
iso.replace("Z", "+00:00")).timestamp() * 1000)
except ValueError:
return 0
def channel_roots(state: dict) -> list:
"""Read the activity channel's root posts from the envelope, normalized to
the comms row shape the intake logic was written against (id / content /
user_id / user_name / created_at-ms / thread_id=None).
This read HAD to move with the writes: the Pulse UI now posts envelope-
native messages, which never appear in the legacy comms GET — reading
comms would silently stop ingesting exactly the posts users make.
kind=post returns ROOT posts only (replies are nested), which is what the
intake wants anyway.
"""
channel_id = ensure_channel(state)
if not channel_id:
return [] # no activity channel yet -> nothing to ingest
data = api("GET", f"/api/workspaces/{WORKSPACE_ID}/messages"
f"?channelId={channel_id}&kind=post&limit=100") or {}
rows = []
for m in data.get("items") or []:
rows.append({
"id": m.get("sourceId"),
"content": m.get("body") or "",
# comms encoded agents as user_id "agent:<id>"; the envelope keeps
# actorType separately — reconstruct the prefix the filters test.
"user_id": (f"agent:{m.get('actorId')}"
if m.get("actorType") == "agent" else m.get("actorId")),
"user_name": m.get("actorDisplayName"),
"created_at": _created_ms(m.get("createdAt") or ""),
"thread_id": None,
"actor_id": m.get("actorId"),
})
return rows
def intake_ingest(state: dict) -> None:
"""Turn new non-bridge top-level activity posts into ledger items + tasks."""
msgs = channel_roots(state)
cp = int(state.get("intake_checkpoint") or 0)
if cp == 0:
# First intake run: start strictly after the newest existing post;
# never ingest history. (+1 because the window below is INCLUSIVE.)
state["intake_checkpoint"] = 1 + max(
[int(m.get("created_at") or 0) for m in msgs] or [0])
save_state(state)
return
# Inclusive window (ts >= cp) so two posts sharing one millisecond can't
# lose the second to a crash between them; the intake map (keyed by post
# id) is the dedupe, mirroring the outbound seen-set discipline.
candidates = []
for m in msgs:
ts = int(m.get("created_at") or 0)
content = (m.get("content") or "").strip()
if (ts < cp or m.get("thread_id") or m["id"] in state["intake"]
# Envelope-native bridge posts carry NO display name (no
# userName override on the envelope route) — the service
# actor id is how the bridge recognises its own posts now.
or m.get("actor_id") == SELF_ACTOR_ID
or (m.get("user_name") or "").lower() in INTAKE_IGNORE_USERS
or (m.get("user_id") or "").startswith("agent:")
or content.startswith("🤖") or "ledger:" in content
or content.lower().startswith(INTAKE_OPTOUT_PREFIXES)):
continue # opt-out posts are chat-only: no item, no task, no reply
candidates.append(m)
for post in sorted(candidates, key=lambda m: int(m["created_at"])):
ledger_id = ledger_add_for_post(post)
# Dedupe with the outbound mirror: map the ledger item to the USER'S
# post before its op:add reaches the feed, so create_post() skips it.
state["map"][ledger_id] = post["id"]
entry = {"ledger_id": ledger_id, "content": post.get("content") or "",
"author": post.get("user_name") or post.get("user_id")}
state["intake"][post["id"]] = entry
state["intake_checkpoint"] = max(int(state["intake_checkpoint"]),
int(post["created_at"]))
save_state(state)
log(f"intake {post['id'][:8]} -> ledger {ledger_id[:8]}")
backlog = any(e.get("queued") for e in state["intake"].values())
if not backlog and intake_active_count(state) < INTAKE_MAX_CONCURRENT:
intake_start_task(state, post["id"], entry)
else:
entry["queued"] = True
save_state(state)
n = intake_active_count(state)
log(f"intake {post['id'][:8]} queued behind {n} tasks")
try:
send_message(state, f"⏳ queued behind {n} tasks",
thread_id=post["id"],
idem=f"pulse-queued:{post['id']}")
except urllib.error.HTTPError as e:
log(f"DROP queue-note {post['id'][:8]}: HTTP {e.code}")
STATUS_COMMENT = {"running": "▶️ running", "blocked": "⛔ blocked",
"todo": "📋 waiting for a worker"}
def intake_track(state: dict) -> None:
"""Advance ledger + post comments as kanban task statuses change."""
open_entries = {pid: e for pid, e in state["intake"].items()
if e.get("task_id") and not e.get("closed")}
if not open_entries:
return
tasks = {t["id"]: t for t in json.loads(kanban("list", "--json"))}
for post_id, entry in open_entries.items():
task = tasks.get(entry["task_id"])
if task is None:
continue # archived/missing; leave for manual attention
status = task.get("status")
if status == entry.get("last_status"):
continue
entry["last_status"] = status
lid = entry["ledger_id"]
if status == "done":
note = (task.get("result") or "").strip()
body = "✅ completed" + (f" — {neutralize(note[:400])}" if note
else "")
entry["closed"] = True
try:
shre_items("done", lid[:8], "--why",
f"kanban {entry['task_id']} completed")
except RuntimeError as e:
log(f"WARN close {lid[:8]}: {e}")
else:
body = STATUS_COMMENT.get(status, f"status → {status}")
if status == "running":
try:
shre_items("update", lid[:8], "--stage", "build",
"--note", f"kanban {entry['task_id']} running")
except RuntimeError as e:
log(f"WARN stage {lid[:8]}: {e}")
save_state(state)
log(f"track {entry['task_id']} {status} (post {post_id[:8]})")
try:
# Keyed per task+status: re-entering a status (blocked -> running
# -> blocked) dedupes to the first comment, which is the right
# noise level for a status thread.
send_message(state, body, thread_id=post_id,
idem=f"pulse-status:{entry['task_id']}:{status}")
except urllib.error.HTTPError as e:
log(f"DROP track {post_id[:8]}: HTTP {e.code}")
def intake_promote(state: dict) -> None:
"""Start queued intake entries when concurrency slots free up."""
for post_id, entry in state["intake"].items():
if not entry.get("queued"):
continue
if intake_active_count(state) >= INTAKE_MAX_CONCURRENT:
return
intake_start_task(state, post_id, entry)
# ── Main loop ────────────────────────────────────────────────────────────────
def seed(state: dict) -> None:
"""First run: mirror currently-OPEN items only, then start the op stream
from 'now'. Historic adds/closes are deliberately not replayed."""
checkpoint = datetime.now(timezone.utc).isoformat(timespec="seconds")
items = open_items()
log(f"seeding {len(items)} open items")
for item in items:
try:
create_post(state, item, mirror_kanban=False)
except Exception as e: # keep seeding the rest
log(f"ERROR seeding {item.get('id', '?')[:8]}: {e}")
state["seeded"] = True
state["checkpoint"] = checkpoint
save_state(state)
def process(state: dict, rec: dict) -> None:
op, lid = rec.get("op"), rec.get("id")
if not lid or op not in ("add", "update", "close"):
return
if op == "add":
create_post(state, rec)
return
# Intake-managed items already get status comments from intake_track();
# mirroring their ledger ops too would double-comment the user's post.
if any(e.get("ledger_id") == lid for e in state.get("intake", {}).values()):
return
# (id, op, at) is exactly the tuple the seen-set dedupes on, so making it
# the idempotency key gives server-side retries the same identity rule.
idem = f"ledger:{lid}:{op}:{rec.get('at', '')}"
if op == "update":
comment(state, lid, update_body(rec), "update", idem=idem)
else:
comment(state, lid, close_body(rec), "close", idem=idem)
def poll(state: dict) -> None:
records = feed_since(state["checkpoint"])
seen = {tuple(s) for s in state.get("seen", [])}
new = [r for r in records
if (r.get("id"), r.get("op"), r.get("at")) not in seen]
for rec in new:
try:
process(state, rec)
except Exception as e:
log(f"ERROR {rec.get('op')} {str(rec.get('id'))[:8]}: {e}")
# don't advance past a failed record's timestamp; retry next poll
break
seen.add((rec.get("id"), rec.get("op"), rec.get("at")))
state["checkpoint"] = max(state["checkpoint"], rec.get("at", ""))
# keep only boundary-second entries for the inclusive-window dedupe
state["seen"] = [list(t) for t in seen if t[2] >= state["checkpoint"]]
save_state(state)
def main() -> int:
log(f"pulse-bridge v1 (envelope) starting: mib={MIB_BASE} ws={WORKSPACE_ID} "
f"channel={CHANNEL_NAME} poll={POLL_SEC}s state={STATE_FILE}")
state = load_state()
if not state.get("seeded"):
try:
seed(state)
except Exception as e:
log(f"FATAL seed failed: {e}")
time.sleep(POLL_SEC)
return 1 # launchd KeepAlive restarts us
# Older state files predate intake; give them the new keys.
state.setdefault("intake_checkpoint", 0)
state.setdefault("intake", {})
while True:
try:
poll(state)
except Exception as e:
log(f"ERROR poll: {e}")
if INTAKE_ENABLED:
# track first (frees slots), then promote the queued backlog
# (FIFO), then ingest new posts — so newcomers can't jump ahead.
for step in (intake_track, intake_promote, intake_ingest):
try:
step(state)
except Exception as e:
log(f"ERROR {step.__name__}: {e}")
time.sleep(POLL_SEC)
if __name__ == "__main__":
sys.exit(main())