Files
claude-code-gnome-extension/hooks/claude-status-hook.py
T
av 75bfe77950 Survive a crash, and a reboot after one
kill, a closed window, a reboot: no SessionEnd arrives and the state file
stays. Each case was tried rather than reasoned about, and one of the
three was broken.

A killed session was already handled -- the process is gone, so the file
and its lock are removed within the 20 s liveness tick. An interrupted
hook write left its temporary file behind forever; those are now swept
once they are five minutes old, which is late enough that a hook part-way
through writing one does not lose the update.

The reboot case was the broken one. State files outlive a reboot and pids
are handed out afresh, so "does /proc/<pid> exist" only answers "is some
process wearing that number". Verified by giving an unrelated live process
the pid of a dead session: the ghost sat in the panel as a session waiting
for input, and would have stayed there forever, asking for an answer
nobody could give. The pid is now pinned to the process start time from
/proc/<pid>/stat, recorded when the state is written and compared when it
is read.

Files written before that field existed compare only on existence, as
before, so a session open across the upgrade is not evicted.

An abandoned flock needed nothing: the kernel drops it when the holder
dies, so there is no deadlock to recover from.
2026-08-09 19:43:31 +03:00

380 lines
14 KiB
Python
Executable File

#!/usr/bin/env python3
"""Claude Code hook -> per-session state file for the GNOME panel indicator.
Reads one hook event as JSON on stdin and maps it to a session state, written
to $XDG_STATE_HOME/claude-code-status/<session_id>.json. The GNOME extension
watches that directory; no polling, no daemon, no socket.
Design notes that are easy to get wrong:
* Writes are skipped when the state does not change. PostToolUse fires on every
tool call, and its only job here is to clear "blocked" once a permission has
been granted -- letting it rewrite the file each time would make the directory
monitor fire hundreds of times per turn for no new information.
* Hooks are registered async, so two events can race (the last PostToolUse of a
turn against that turn's Stop). The whole read-decide-write runs under a file
lock and an event older than the stored one is refused; without both, a late
"busy" buries "waiting" and the panel claims a session is working while it
actually waits for input.
* No stdlib import beyond what is needed: this runs once per tool call.
"""
import fcntl
import json
import os
import sys
import time
STATE_DIR = os.path.join(
os.environ.get("XDG_STATE_HOME") or os.path.expanduser("~/.local/state"),
"claude-code-status",
)
# Notification covers several unrelated things; only some mean "the session
# stopped and is asking me something". auth_success / elicitation_complete /
# elicitation_response are progress chatter and must not touch the state.
# Observed: a question put to the user (AskUserQuestion) arrives as
# "permission_prompt" too, with the same generic message as a tool asking to
# run. The two are therefore not separable here, which is why the panel has one
# "blocked" state rather than telling a permission from a question.
NOTIFICATION_STATES = {
"permission_prompt": "blocked",
"agent_needs_input": "blocked",
"elicitation_dialog": "blocked",
"idle_prompt": "waiting",
"agent_completed": "waiting",
}
# Tools that spawn a subagent. Matched again here, not just in the hook
# registration: the matcher is a regex and a mistake there would silently
# inflate the count with TaskCreate, TaskUpdate and the like.
AGENT_TOOLS = {"Agent", "Task"}
EVENT_STATES = {
# A session that has just opened is waiting for your first prompt, which is
# the same thing as one that has finished a turn: the input line is free
# and the next move is yours. It had a state of its own once; it was only
# ever reachable before the first prompt, so it bought a fourth glyph in
# the panel that nobody saw.
"SessionStart": "waiting",
"PreToolUse": "busy",
"UserPromptSubmit": "busy",
"PreCompact": "busy",
"PostToolUse": "busy",
"Stop": "waiting",
}
def debug_log(event):
"""Append raw events when a 'debug' marker file exists in the state dir.
Gated on a file rather than an env var because the hook inherits claude's
environment, which cannot be changed without restarting the session.
"""
marker = os.path.join(STATE_DIR, "debug")
if not os.path.exists(marker):
return
try:
with open(marker, "a") as fh:
stamped = dict(event, _at=time.strftime("%H:%M:%S"))
fh.write(json.dumps(stamped, sort_keys=True)[:2000] + "\n")
except OSError:
pass
def derive_state(event):
"""Return the new state, 'end' to drop the session, or None to ignore."""
name = event.get("hook_event_name")
if name == "SessionEnd":
return "end"
# A notification is honoured whoever raised it: it means a human is needed,
# and that is just as true when the agent that got stuck is a subagent.
if name == "Notification":
return NOTIFICATION_STATES.get(event.get("notification_type"))
# SubagentStop is the counter's decrement and carries agent_id itself, so
# it has to pass the filter below. Its state is decided in apply_event,
# which is where the count is known.
if name == "SubagentStop":
return "busy"
# A subagent's own tool calls also reach the parent session's hooks
# (measured: PostToolUse carrying agent_id and agent_type). They are
# ignored, because the count already says a subagent is running and these
# would only add write traffic.
if event.get("agent_id"):
return None
if name == "PreToolUse" and event.get("tool_name") not in AGENT_TOOLS:
return None
return EVENT_STATES.get(name)
def read_environ(pid):
"""Environment of a process as a dict, empty if it is gone or not ours."""
try:
with open("/proc/%d/environ" % pid, "rb") as fh:
raw = fh.read()
except OSError:
return {}
env = {}
for entry in raw.split(b"\0"):
if not entry:
continue
key, sep, value = entry.partition(b"=")
if sep:
env[key.decode("utf-8", "replace")] = value.decode("utf-8", "replace")
return env
def read_cmdline(pid):
try:
with open("/proc/%d/cmdline" % pid, "rb") as fh:
return fh.read().replace(b"\0", b" ").decode("utf-8", "replace")
except OSError:
return ""
def parent_of(pid):
try:
with open("/proc/%d/status" % pid, "r") as fh:
for line in fh:
if line.startswith("PPid:"):
return int(line.split()[1])
except (OSError, ValueError):
pass
return 0
def find_claude_pid():
"""Nearest ancestor that is the claude process itself, or 0 if unknown.
The hook is spawned through a shell, so the immediate parent is usually not
claude. Walking beyond a handful of levels risks latching onto an outer
claude when one session drives another, so the search stops early.
Returning 0 rather than guessing matters: the reader deletes state files
whose pid is gone, and the obvious fallback -- the shell that spawned this
hook -- exits milliseconds later, which would make the session flicker in
and out of the panel forever.
"""
pid = os.getppid()
for _ in range(6):
if pid <= 1:
break
# Matched anywhere in the command line, not just argv[0]: installs that
# run it as `node .../claude/cli.js` are just as valid as a direct one.
if "claude" in read_cmdline(pid):
return pid
pid = parent_of(pid)
return 0
def pid_start_time(pid):
"""Field 22 of /proc/<pid>/stat: when the process started, in clock ticks.
Pins a pid to one particular process. Pids are reused, and state files
outlive reboots -- without this, a file left by a crashed session whose pid
is later handed to something unrelated reads as a live session forever.
"""
try:
with open("/proc/%d/stat" % pid) as fh:
data = fh.read()
except OSError:
return 0
# Field 2 is the command name, parenthesised, and may itself contain spaces
# and a ')'. Everything after the last ')' is field 3 onwards.
tail = data[data.rfind(")") + 2:].split()
try:
return int(tail[19])
except (IndexError, ValueError):
return 0
def alive(pid, start=0):
if pid <= 0 or not os.path.exists("/proc/%d" % pid):
return False
# A file written before start times were recorded has nothing to compare.
return not start or pid_start_time(pid) == start
def sweep_dead(keep):
"""Drop state files whose claude process is gone.
A killed terminal never sends SessionEnd, so files leak. Cleaning up on
SessionStart keeps the sweep off the hot path -- the extension only has to
hide stale entries, not own their lifetime.
"""
try:
names = os.listdir(STATE_DIR)
except OSError:
return
for name in names:
if not name.endswith(".json") or name == keep:
continue
path = os.path.join(STATE_DIR, name)
try:
with open(path, "r") as fh:
stale = json.load(fh)
pid = stale.get("pid", 0)
except (OSError, ValueError, AttributeError):
continue
# pid 0 means the hook could not identify the process; there is nothing
# to test for liveness, so leave it to the reader's age cutoff.
if pid and not alive(pid, stale.get("pid_start", 0)):
for victim in (path, path + ".lock"):
try:
os.unlink(victim)
except OSError:
pass
def write_atomic(path, payload):
tmp = "%s.%d.tmp" % (path, os.getpid())
with open(tmp, "w") as fh:
json.dump(payload, fh)
os.replace(tmp, path)
def apply_event(event, state, path, now):
"""Read the current state, decide, and write. Must run under the lock."""
if event.get("hook_event_name") == "SessionStart":
# Swept before any early return: a resumed session keeps its id, so its
# SessionStart finds an unchanged state and would otherwise bail out
# before ever reaching the sweep.
sweep_dead(keep=os.path.basename(path))
if state == "end":
for victim in (path, path + ".lock"):
try:
os.unlink(victim)
except OSError:
pass
return
previous = None
try:
with open(path, "r") as fh:
previous = json.load(fh)
except (OSError, ValueError):
previous = None
if not isinstance(previous, dict):
previous = None
# Normalised here so the comparison below is against what would actually be
# stored: comparing a stored "" to a raw notification message rewrites the
# file on every idle_prompt for no change at all.
message = event.get("message", "") if state == "blocked" else ""
name = event.get("hook_event_name")
agents = int(previous.get("agents", 0)) if previous else 0
# Whether the main agent has finished its turn. Tracked separately from the
# state because with background subagents both are true at once: the turn is
# over and work is still running.
stopped = bool(previous.get("stopped")) if previous else False
if name in ("SessionStart", "UserPromptSubmit"):
# A new turn from you starts a new batch. This also bounds the damage
# when a subagent dies without its SubagentStop ever arriving: the count
# cannot leak past the next thing you type.
agents, stopped = 0, False
elif name == "PreToolUse":
agents += 1
elif name == "SubagentStop":
agents = max(0, agents - 1)
elif name == "Stop":
stopped = True
elif name == "PostToolUse":
stopped = False
if name == "SubagentStop":
# The last subagent finishing is what finally frees a session whose main
# agent stopped long ago.
state = "waiting" if (stopped and agents == 0) else "busy"
elif state == "waiting" and agents > 0:
# The turn ended but the batch is still running, and the session will
# pick the results up itself. Calling it "waiting" would send you to a
# terminal that does not need you.
state = "busy"
if previous:
# Auto-compaction raises SessionStart again, in the middle of a turn the
# session is still working on. Taking it at face value would flip a busy
# session to idle until the next tool call corrected it.
if event.get("hook_event_name") == "SessionStart" and event.get("source") == "compact":
return
# Refuse events that lost a race with a newer one.
if previous.get("event_ts", 0) > now:
return
# Nothing new to publish: stay quiet so the directory monitor stays quiet.
if (previous.get("state") == state
and previous.get("message", "") == message
and previous.get("agents", 0) == agents
and bool(previous.get("stopped")) == stopped):
return
claude_pid = find_claude_pid()
env = read_environ(claude_pid) if claude_pid else {}
write_atomic(path, {
"session_id": event.get("session_id"),
"state": state,
"cwd": event.get("cwd") or "",
# Age is measured from the moment the state was entered, not from the
# last event, so "waiting 40 min" survives unrelated later writes.
"since": previous["since"] if previous and previous.get("state") == state else now,
"event_ts": now,
# 0 means "could not tell"; the reader must not take that for "dead".
"pid": claude_pid,
"pid_start": pid_start_time(claude_pid) if claude_pid else 0,
"event": event.get("hook_event_name", ""),
"notification_type": event.get("notification_type", ""),
"message": message,
"agents": agents,
"stopped": stopped,
"zellij_session": env.get("ZELLIJ_SESSION_NAME", ""),
"zellij_pane": env.get("ZELLIJ_PANE_ID", ""),
"transcript": event.get("transcript_path", ""),
})
def main():
now = time.time()
try:
event = json.load(sys.stdin)
except (ValueError, OSError):
return 0
if not isinstance(event, dict):
return 0
session_id = event.get("session_id")
if not session_id or "/" in session_id:
return 0
debug_log(event)
state = derive_state(event)
if state is None:
return 0
os.makedirs(STATE_DIR, exist_ok=True)
path = os.path.join(STATE_DIR, "%s.json" % session_id)
# Hooks for one session run concurrently -- the last PostToolUse of a turn
# races that turn's Stop. Comparing timestamps is not enough on its own:
# without a lock both processes read the same "previous" and the loser's
# write still lands last, pinning a finished session at "busy". The lock is
# a separate file because write_atomic replaces the inode of the real one.
with open(path + ".lock", "w") as lock:
fcntl.flock(lock, fcntl.LOCK_EX)
apply_event(event, state, path, now)
return 0
if __name__ == "__main__":
try:
sys.exit(main())
except Exception:
# A hook that fails loudly would spam every session with error output;
# a missing panel update is the cheaper failure.
sys.exit(0)