feat: Claude Code Monitor — lanes, pipelines and a merged workspace

Internal SmartGift build of a Claude Code monitoring dashboard.

Lanes: a durable unit of parallel agent work, one per working directory,
tracked across session restarts. Managed lanes are git worktrees the
dashboard provisions and can reset or remove behind a three-check destroy
guard and a counted preflight; adopted lanes are directories you already
own and are never destroyable.

Pipelines: a lane moves through pipeline stages. A stage the agent declares
with evidence renders green; a stage inferred from the tool-event stream
renders dashed amber and never counts as done. Detection is forward-only
within a 30-minute window, and never writes the declared stage.

Workspace: one page at /run with a lane grid, the selected lane's pipeline,
and a full Claude console behind a disclosure.
This commit is contained in:
2026-07-29 17:07:45 +07:00
commit 57dc91585d
783 changed files with 221743 additions and 0 deletions
@@ -0,0 +1,136 @@
apiVersion: apps/v1
kind: Deployment
metadata:
name: agent-monitor-blue
namespace: agent-monitor
labels:
app.kubernetes.io/name: agent-monitor
app.kubernetes.io/instance: agent-monitor
app.kubernetes.io/version: "1.0.0"
app.kubernetes.io/component: server
app.kubernetes.io/managed-by: kustomize
slot: blue
spec:
replicas: 3
revisionHistoryLimit: 5
strategy:
type: RollingUpdate
rollingUpdate:
maxSurge: 1
maxUnavailable: 0
selector:
matchLabels:
app.kubernetes.io/name: agent-monitor
app.kubernetes.io/instance: agent-monitor
slot: blue
template:
metadata:
labels:
app.kubernetes.io/name: agent-monitor
app.kubernetes.io/instance: agent-monitor
app.kubernetes.io/version: "1.0.0"
app.kubernetes.io/component: server
app.kubernetes.io/managed-by: kustomize
slot: blue
spec:
serviceAccountName: agent-monitor
automountServiceAccountToken: false
terminationGracePeriodSeconds: 30
securityContext:
runAsNonRoot: true
runAsUser: 1000
runAsGroup: 1000
fsGroup: 1000
fsGroupChangePolicy: OnRootMismatch
seccompProfile:
type: RuntimeDefault
affinity:
podAntiAffinity:
requiredDuringSchedulingIgnoredDuringExecution:
- labelSelector:
matchExpressions:
- key: app.kubernetes.io/name
operator: In
values:
- agent-monitor
topologyKey: kubernetes.io/hostname
containers:
- name: agent-monitor
image: ${IMAGE_REGISTRY}/agent-monitor:blue
imagePullPolicy: IfNotPresent
ports:
- name: http
containerPort: 4820
protocol: TCP
envFrom:
- configMapRef:
name: agent-monitor-config
resources:
requests:
memory: "256Mi"
cpu: "200m"
limits:
memory: "1Gi"
cpu: "1000m"
startupProbe:
httpGet:
path: /api/health
port: http
failureThreshold: 30
periodSeconds: 2
readinessProbe:
httpGet:
path: /api/health
port: http
initialDelaySeconds: 5
periodSeconds: 5
timeoutSeconds: 3
successThreshold: 1
failureThreshold: 3
livenessProbe:
httpGet:
path: /api/health
port: http
initialDelaySeconds: 15
periodSeconds: 15
timeoutSeconds: 5
successThreshold: 1
failureThreshold: 3
securityContext:
runAsNonRoot: true
runAsUser: 1000
runAsGroup: 1000
readOnlyRootFilesystem: true
allowPrivilegeEscalation: false
capabilities:
drop:
- ALL
volumeMounts:
- name: data
mountPath: /app/data
- name: tmp
mountPath: /tmp
lifecycle:
preStop:
exec:
command: ["sh", "-c", "sleep 5"]
volumes:
- name: data
persistentVolumeClaim:
claimName: agent-monitor-data
- name: tmp
emptyDir:
sizeLimit: 100Mi
@@ -0,0 +1,136 @@
apiVersion: apps/v1
kind: Deployment
metadata:
name: agent-monitor-green
namespace: agent-monitor
labels:
app.kubernetes.io/name: agent-monitor
app.kubernetes.io/instance: agent-monitor
app.kubernetes.io/version: "1.0.0"
app.kubernetes.io/component: server
app.kubernetes.io/managed-by: kustomize
slot: green
spec:
replicas: 3
revisionHistoryLimit: 5
strategy:
type: RollingUpdate
rollingUpdate:
maxSurge: 1
maxUnavailable: 0
selector:
matchLabels:
app.kubernetes.io/name: agent-monitor
app.kubernetes.io/instance: agent-monitor
slot: green
template:
metadata:
labels:
app.kubernetes.io/name: agent-monitor
app.kubernetes.io/instance: agent-monitor
app.kubernetes.io/version: "1.0.0"
app.kubernetes.io/component: server
app.kubernetes.io/managed-by: kustomize
slot: green
spec:
serviceAccountName: agent-monitor
automountServiceAccountToken: false
terminationGracePeriodSeconds: 30
securityContext:
runAsNonRoot: true
runAsUser: 1000
runAsGroup: 1000
fsGroup: 1000
fsGroupChangePolicy: OnRootMismatch
seccompProfile:
type: RuntimeDefault
affinity:
podAntiAffinity:
requiredDuringSchedulingIgnoredDuringExecution:
- labelSelector:
matchExpressions:
- key: app.kubernetes.io/name
operator: In
values:
- agent-monitor
topologyKey: kubernetes.io/hostname
containers:
- name: agent-monitor
image: ${IMAGE_REGISTRY}/agent-monitor:green
imagePullPolicy: IfNotPresent
ports:
- name: http
containerPort: 4820
protocol: TCP
envFrom:
- configMapRef:
name: agent-monitor-config
resources:
requests:
memory: "256Mi"
cpu: "200m"
limits:
memory: "1Gi"
cpu: "1000m"
startupProbe:
httpGet:
path: /api/health
port: http
failureThreshold: 30
periodSeconds: 2
readinessProbe:
httpGet:
path: /api/health
port: http
initialDelaySeconds: 5
periodSeconds: 5
timeoutSeconds: 3
successThreshold: 1
failureThreshold: 3
livenessProbe:
httpGet:
path: /api/health
port: http
initialDelaySeconds: 15
periodSeconds: 15
timeoutSeconds: 5
successThreshold: 1
failureThreshold: 3
securityContext:
runAsNonRoot: true
runAsUser: 1000
runAsGroup: 1000
readOnlyRootFilesystem: true
allowPrivilegeEscalation: false
capabilities:
drop:
- ALL
volumeMounts:
- name: data
mountPath: /app/data
- name: tmp
mountPath: /tmp
lifecycle:
preStop:
exec:
command: ["sh", "-c", "sleep 5"]
volumes:
- name: data
persistentVolumeClaim:
claimName: agent-monitor-data
- name: tmp
emptyDir:
sizeLimit: 100Mi
@@ -0,0 +1,32 @@
apiVersion: v1
kind: Service
metadata:
name: agent-monitor
namespace: agent-monitor
labels:
app.kubernetes.io/name: agent-monitor
app.kubernetes.io/instance: agent-monitor
app.kubernetes.io/version: "1.0.0"
app.kubernetes.io/component: server
app.kubernetes.io/managed-by: kustomize
annotations:
# Document which slot is currently active
# To switch traffic: kubectl patch svc agent-monitor -n agent-monitor \
# -p '{"spec":{"selector":{"slot":"green"}}}'
agent-monitor.io/active-slot: blue
spec:
type: ClusterIP
sessionAffinity: ClientIP
sessionAffinityConfig:
clientIP:
timeoutSeconds: 10800
selector:
app.kubernetes.io/name: agent-monitor
app.kubernetes.io/instance: agent-monitor
# Toggle this value between "blue" and "green" to switch traffic
slot: blue
ports:
- name: http
port: 80
targetPort: http
protocol: TCP
@@ -0,0 +1,97 @@
apiVersion: argoproj.io/v1alpha1
kind: AnalysisTemplate
metadata:
name: agent-monitor-canary-analysis
namespace: agent-monitor
labels:
app.kubernetes.io/name: agent-monitor
app.kubernetes.io/instance: agent-monitor
app.kubernetes.io/version: "1.0.0"
app.kubernetes.io/component: canary-analysis
app.kubernetes.io/managed-by: kustomize
spec:
args:
- name: service-name
value: agent-monitor-canary
- name: namespace
value: agent-monitor
metrics:
# Success rate must be above 99%
- name: success-rate
interval: 60s
count: 5
successCondition: result[0] >= 0.99
failureLimit: 2
provider:
prometheus:
address: http://prometheus.monitoring.svc.cluster.local:9090
query: |
sum(
rate(
http_requests_total{
namespace="{{args.namespace}}",
service="{{args.service-name}}",
code!~"5.."
}[2m]
)
)
/
sum(
rate(
http_requests_total{
namespace="{{args.namespace}}",
service="{{args.service-name}}"
}[2m]
)
)
# P99 latency must be under 500ms
- name: p99-latency
interval: 60s
count: 5
successCondition: result[0] < 500
failureLimit: 2
provider:
prometheus:
address: http://prometheus.monitoring.svc.cluster.local:9090
query: |
histogram_quantile(
0.99,
sum(
rate(
http_request_duration_milliseconds_bucket{
namespace="{{args.namespace}}",
service="{{args.service-name}}"
}[2m]
)
) by (le)
)
# Error rate must stay below 1%
- name: error-rate
interval: 60s
count: 5
successCondition: result[0] <= 0.01
failureLimit: 2
provider:
prometheus:
address: http://prometheus.monitoring.svc.cluster.local:9090
query: |
sum(
rate(
http_requests_total{
namespace="{{args.namespace}}",
service="{{args.service-name}}",
code=~"5.."
}[2m]
)
)
/
sum(
rate(
http_requests_total{
namespace="{{args.namespace}}",
service="{{args.service-name}}"
}[2m]
)
)
@@ -0,0 +1,124 @@
apiVersion: apps/v1
kind: Deployment
metadata:
name: agent-monitor-canary
namespace: agent-monitor
labels:
app.kubernetes.io/name: agent-monitor
app.kubernetes.io/instance: agent-monitor-canary
app.kubernetes.io/version: "1.0.0"
app.kubernetes.io/component: server
app.kubernetes.io/managed-by: kustomize
track: canary
spec:
replicas: 1
revisionHistoryLimit: 5
selector:
matchLabels:
app.kubernetes.io/name: agent-monitor
app.kubernetes.io/instance: agent-monitor-canary
track: canary
template:
metadata:
labels:
app.kubernetes.io/name: agent-monitor
app.kubernetes.io/instance: agent-monitor-canary
app.kubernetes.io/version: "1.0.0"
app.kubernetes.io/component: server
app.kubernetes.io/managed-by: kustomize
track: canary
annotations:
prometheus.io/scrape: "true"
prometheus.io/port: "4820"
prometheus.io/path: "/api/health"
spec:
serviceAccountName: agent-monitor
automountServiceAccountToken: false
terminationGracePeriodSeconds: 30
securityContext:
runAsNonRoot: true
runAsUser: 1000
runAsGroup: 1000
fsGroup: 1000
fsGroupChangePolicy: OnRootMismatch
seccompProfile:
type: RuntimeDefault
containers:
- name: agent-monitor
image: ${IMAGE_REGISTRY}/agent-monitor:canary
imagePullPolicy: Always
ports:
- name: http
containerPort: 4820
protocol: TCP
envFrom:
- configMapRef:
name: agent-monitor-config
resources:
requests:
memory: "256Mi"
cpu: "200m"
limits:
memory: "1Gi"
cpu: "1000m"
startupProbe:
httpGet:
path: /api/health
port: http
failureThreshold: 30
periodSeconds: 2
readinessProbe:
httpGet:
path: /api/health
port: http
initialDelaySeconds: 5
periodSeconds: 5
timeoutSeconds: 3
successThreshold: 1
failureThreshold: 3
livenessProbe:
httpGet:
path: /api/health
port: http
initialDelaySeconds: 15
periodSeconds: 15
timeoutSeconds: 5
successThreshold: 1
failureThreshold: 3
securityContext:
runAsNonRoot: true
runAsUser: 1000
runAsGroup: 1000
readOnlyRootFilesystem: true
allowPrivilegeEscalation: false
capabilities:
drop:
- ALL
volumeMounts:
- name: data
mountPath: /app/data
- name: tmp
mountPath: /tmp
lifecycle:
preStop:
exec:
command: ["sh", "-c", "sleep 5"]
volumes:
- name: data
persistentVolumeClaim:
claimName: agent-monitor-data
- name: tmp
emptyDir:
sizeLimit: 100Mi