Files
claude-code-gnome-extension/hooks/claude-status-hook.py
T
av f4a06cdc44 Identify the claude process by argument, not by substring
Testing the reboot case exposed a live regression, and the identity check
added for reboots is what made it visible.

Matching "claude" anywhere in an ancestor's command line was too loose. The
hook is spawned as `/bin/sh -c /.../claude-status-hook.py`, so its parent's
command line contains "claude" -- in the path to this very script -- while
being a shell that exits milliseconds later. That shell's pid was being
recorded as the session's. Existence checks alone hid it: the pid was dead,
the file was deleted, the next event recreated it, and the session flickered.
Once the pid was pinned to a process start time the session vanished
outright.

The rule was broadened in the first place to cover npm-style installs that
run `node .../claude-code/cli.js`, which was a real gap. It is now matched
per argument instead: argv[0] named claude, or a cli.js under a claude path.
Both installs pass, and the spawning shell does not.

The pid is also resolved before the unchanged-check rather than after, so a
pid that has changed forces a write. A session resumed under a new pid --
which is exactly what --resume does, and what this session had done -- kept
its old pid for as long as its state happened not to change, and the reader
would drop it as dead.

The debug log now records the process ancestry, which is what made this
diagnosable at all rather than guessable.
2026-08-09 19:47:19 +03:00

411 lines
15 KiB
Python
Executable File

#!/usr/bin/env python3
"""Claude Code hook -> per-session state file for the GNOME panel indicator.
Reads one hook event as JSON on stdin and maps it to a session state, written
to $XDG_STATE_HOME/claude-code-status/<session_id>.json. The GNOME extension
watches that directory; no polling, no daemon, no socket.
Design notes that are easy to get wrong:
* Writes are skipped when the state does not change. PostToolUse fires on every
tool call, and its only job here is to clear "blocked" once a permission has
been granted -- letting it rewrite the file each time would make the directory
monitor fire hundreds of times per turn for no new information.
* Hooks are registered async, so two events can race (the last PostToolUse of a
turn against that turn's Stop). The whole read-decide-write runs under a file
lock and an event older than the stored one is refused; without both, a late
"busy" buries "waiting" and the panel claims a session is working while it
actually waits for input.
* No stdlib import beyond what is needed: this runs once per tool call.
"""
import fcntl
import json
import os
import sys
import time
STATE_DIR = os.path.join(
os.environ.get("XDG_STATE_HOME") or os.path.expanduser("~/.local/state"),
"claude-code-status",
)
# Notification covers several unrelated things; only some mean "the session
# stopped and is asking me something". auth_success / elicitation_complete /
# elicitation_response are progress chatter and must not touch the state.
# Observed: a question put to the user (AskUserQuestion) arrives as
# "permission_prompt" too, with the same generic message as a tool asking to
# run. The two are therefore not separable here, which is why the panel has one
# "blocked" state rather than telling a permission from a question.
NOTIFICATION_STATES = {
"permission_prompt": "blocked",
"agent_needs_input": "blocked",
"elicitation_dialog": "blocked",
"idle_prompt": "waiting",
"agent_completed": "waiting",
}
# Tools that spawn a subagent. Matched again here, not just in the hook
# registration: the matcher is a regex and a mistake there would silently
# inflate the count with TaskCreate, TaskUpdate and the like.
AGENT_TOOLS = {"Agent", "Task"}
EVENT_STATES = {
# A session that has just opened is waiting for your first prompt, which is
# the same thing as one that has finished a turn: the input line is free
# and the next move is yours. It had a state of its own once; it was only
# ever reachable before the first prompt, so it bought a fourth glyph in
# the panel that nobody saw.
"SessionStart": "waiting",
"PreToolUse": "busy",
"UserPromptSubmit": "busy",
"PreCompact": "busy",
"PostToolUse": "busy",
"Stop": "waiting",
}
def debug_log(event):
"""Append raw events when a 'debug' marker file exists in the state dir.
Gated on a file rather than an env var because the hook inherits claude's
environment, which cannot be changed without restarting the session.
"""
marker = os.path.join(STATE_DIR, "debug")
if not os.path.exists(marker):
return
try:
with open(marker, "a") as fh:
chain, pid = [], os.getppid()
for _ in range(6):
if pid <= 1:
break
chain.append("%d %s" % (pid, read_cmdline(pid)[:90]))
pid = parent_of(pid)
stamped = dict(event, _at=time.strftime("%H:%M:%S"), _ancestry=chain)
fh.write(json.dumps(stamped, sort_keys=True)[:2000] + "\n")
except OSError:
pass
def derive_state(event):
"""Return the new state, 'end' to drop the session, or None to ignore."""
name = event.get("hook_event_name")
if name == "SessionEnd":
return "end"
# A notification is honoured whoever raised it: it means a human is needed,
# and that is just as true when the agent that got stuck is a subagent.
if name == "Notification":
return NOTIFICATION_STATES.get(event.get("notification_type"))
# SubagentStop is the counter's decrement and carries agent_id itself, so
# it has to pass the filter below. Its state is decided in apply_event,
# which is where the count is known.
if name == "SubagentStop":
return "busy"
# A subagent's own tool calls also reach the parent session's hooks
# (measured: PostToolUse carrying agent_id and agent_type). They are
# ignored, because the count already says a subagent is running and these
# would only add write traffic.
if event.get("agent_id"):
return None
if name == "PreToolUse" and event.get("tool_name") not in AGENT_TOOLS:
return None
return EVENT_STATES.get(name)
def read_environ(pid):
"""Environment of a process as a dict, empty if it is gone or not ours."""
try:
with open("/proc/%d/environ" % pid, "rb") as fh:
raw = fh.read()
except OSError:
return {}
env = {}
for entry in raw.split(b"\0"):
if not entry:
continue
key, sep, value = entry.partition(b"=")
if sep:
env[key.decode("utf-8", "replace")] = value.decode("utf-8", "replace")
return env
def read_cmdline(pid):
try:
with open("/proc/%d/cmdline" % pid, "rb") as fh:
return fh.read().replace(b"\0", b" ").decode("utf-8", "replace")
except OSError:
return ""
def parent_of(pid):
try:
with open("/proc/%d/status" % pid, "r") as fh:
for line in fh:
if line.startswith("PPid:"):
return int(line.split()[1])
except (OSError, ValueError):
pass
return 0
def looks_like_claude(cmdline):
"""Is this command line the claude binary itself?
Matched per argument, never against the raw string. The hook is spawned as
`/bin/sh -c /.../claude-status-hook.py`, so its parent's command line
contains the word "claude" -- in a path -- without being claude at all.
Latching onto that shell records a pid that exits milliseconds later, and
the session then flickers in and out of the panel.
"""
for token in cmdline.split(" "):
if not token:
continue
base = os.path.basename(token)
if base == "claude":
return True
# npm-style install: node /path/to/claude-code/cli.js
if base == "cli.js" and "claude" in token:
return True
return False
def find_claude_pid():
"""Nearest ancestor that is the claude process itself, or 0 if unknown.
The hook is spawned through a shell, so the immediate parent is not claude.
Walking beyond a handful of levels risks latching onto an outer claude when
one session drives another, so the search stops early.
Returning 0 rather than guessing matters: the reader deletes state files
whose process is gone, and the obvious fallback -- the shell that spawned
this hook -- exits immediately.
"""
pid = os.getppid()
for _ in range(6):
if pid <= 1:
break
if looks_like_claude(read_cmdline(pid)):
return pid
pid = parent_of(pid)
return 0
def pid_start_time(pid):
"""Field 22 of /proc/<pid>/stat: when the process started, in clock ticks.
Pins a pid to one particular process. Pids are reused, and state files
outlive reboots -- without this, a file left by a crashed session whose pid
is later handed to something unrelated reads as a live session forever.
"""
try:
with open("/proc/%d/stat" % pid) as fh:
data = fh.read()
except OSError:
return 0
# Field 2 is the command name, parenthesised, and may itself contain spaces
# and a ')'. Everything after the last ')' is field 3 onwards.
tail = data[data.rfind(")") + 2:].split()
try:
return int(tail[19])
except (IndexError, ValueError):
return 0
def alive(pid, start=0):
if pid <= 0 or not os.path.exists("/proc/%d" % pid):
return False
# A file written before start times were recorded has nothing to compare.
return not start or pid_start_time(pid) == start
def sweep_dead(keep):
"""Drop state files whose claude process is gone.
A killed terminal never sends SessionEnd, so files leak. Cleaning up on
SessionStart keeps the sweep off the hot path -- the extension only has to
hide stale entries, not own their lifetime.
"""
try:
names = os.listdir(STATE_DIR)
except OSError:
return
for name in names:
if not name.endswith(".json") or name == keep:
continue
path = os.path.join(STATE_DIR, name)
try:
with open(path, "r") as fh:
stale = json.load(fh)
pid = stale.get("pid", 0)
except (OSError, ValueError, AttributeError):
continue
# pid 0 means the hook could not identify the process; there is nothing
# to test for liveness, so leave it to the reader's age cutoff.
if pid and not alive(pid, stale.get("pid_start", 0)):
for victim in (path, path + ".lock"):
try:
os.unlink(victim)
except OSError:
pass
def write_atomic(path, payload):
tmp = "%s.%d.tmp" % (path, os.getpid())
with open(tmp, "w") as fh:
json.dump(payload, fh)
os.replace(tmp, path)
def apply_event(event, state, path, now):
"""Read the current state, decide, and write. Must run under the lock."""
if event.get("hook_event_name") == "SessionStart":
# Swept before any early return: a resumed session keeps its id, so its
# SessionStart finds an unchanged state and would otherwise bail out
# before ever reaching the sweep.
sweep_dead(keep=os.path.basename(path))
if state == "end":
for victim in (path, path + ".lock"):
try:
os.unlink(victim)
except OSError:
pass
return
previous = None
try:
with open(path, "r") as fh:
previous = json.load(fh)
except (OSError, ValueError):
previous = None
if not isinstance(previous, dict):
previous = None
# Normalised here so the comparison below is against what would actually be
# stored: comparing a stored "" to a raw notification message rewrites the
# file on every idle_prompt for no change at all.
message = event.get("message", "") if state == "blocked" else ""
name = event.get("hook_event_name")
agents = int(previous.get("agents", 0)) if previous else 0
# Whether the main agent has finished its turn. Tracked separately from the
# state because with background subagents both are true at once: the turn is
# over and work is still running.
stopped = bool(previous.get("stopped")) if previous else False
if name in ("SessionStart", "UserPromptSubmit"):
# A new turn from you starts a new batch. This also bounds the damage
# when a subagent dies without its SubagentStop ever arriving: the count
# cannot leak past the next thing you type.
agents, stopped = 0, False
elif name == "PreToolUse":
agents += 1
elif name == "SubagentStop":
agents = max(0, agents - 1)
elif name == "Stop":
stopped = True
elif name == "PostToolUse":
stopped = False
if name == "SubagentStop":
# The last subagent finishing is what finally frees a session whose main
# agent stopped long ago.
state = "waiting" if (stopped and agents == 0) else "busy"
elif state == "waiting" and agents > 0:
# The turn ended but the batch is still running, and the session will
# pick the results up itself. Calling it "waiting" would send you to a
# terminal that does not need you.
state = "busy"
# Resolved before the unchanged-check, not after, so that a pid which has
# changed forces a write. A session resumed under a new pid, or one whose
# pid was recorded wrongly, would otherwise keep the stale value for as
# long as its state happens not to change -- and the reader, finding that
# process gone, would drop a perfectly live session from the panel.
claude_pid = find_claude_pid()
if previous:
# Auto-compaction raises SessionStart again, in the middle of a turn the
# session is still working on. Taking it at face value would flip a busy
# session to idle until the next tool call corrected it.
if event.get("hook_event_name") == "SessionStart" and event.get("source") == "compact":
return
# Refuse events that lost a race with a newer one.
if previous.get("event_ts", 0) > now:
return
# Nothing new to publish: stay quiet so the directory monitor stays quiet.
if (previous.get("state") == state
and previous.get("message", "") == message
and previous.get("agents", 0) == agents
and bool(previous.get("stopped")) == stopped
and previous.get("pid", 0) == claude_pid):
return
env = read_environ(claude_pid) if claude_pid else {}
write_atomic(path, {
"session_id": event.get("session_id"),
"state": state,
"cwd": event.get("cwd") or "",
# Age is measured from the moment the state was entered, not from the
# last event, so "waiting 40 min" survives unrelated later writes.
"since": previous["since"] if previous and previous.get("state") == state else now,
"event_ts": now,
# 0 means "could not tell"; the reader must not take that for "dead".
"pid": claude_pid,
"pid_start": pid_start_time(claude_pid) if claude_pid else 0,
"event": event.get("hook_event_name", ""),
"notification_type": event.get("notification_type", ""),
"message": message,
"agents": agents,
"stopped": stopped,
"zellij_session": env.get("ZELLIJ_SESSION_NAME", ""),
"zellij_pane": env.get("ZELLIJ_PANE_ID", ""),
"transcript": event.get("transcript_path", ""),
})
def main():
now = time.time()
try:
event = json.load(sys.stdin)
except (ValueError, OSError):
return 0
if not isinstance(event, dict):
return 0
session_id = event.get("session_id")
if not session_id or "/" in session_id:
return 0
debug_log(event)
state = derive_state(event)
if state is None:
return 0
os.makedirs(STATE_DIR, exist_ok=True)
path = os.path.join(STATE_DIR, "%s.json" % session_id)
# Hooks for one session run concurrently -- the last PostToolUse of a turn
# races that turn's Stop. Comparing timestamps is not enough on its own:
# without a lock both processes read the same "previous" and the loser's
# write still lands last, pinning a finished session at "busy". The lock is
# a separate file because write_atomic replaces the inode of the real one.
with open(path + ".lock", "w") as lock:
fcntl.flock(lock, fcntl.LOCK_EX)
apply_event(event, state, path, now)
return 0
if __name__ == "__main__":
try:
sys.exit(main())
except Exception:
# A hook that fails loudly would spam every session with error output;
# a missing panel update is the cheaper failure.
sys.exit(0)