feat: Claude Code Monitor — lanes, pipelines and a merged workspace
Internal SmartGift build of a Claude Code monitoring dashboard. Lanes: a durable unit of parallel agent work, one per working directory, tracked across session restarts. Managed lanes are git worktrees the dashboard provisions and can reset or remove behind a three-check destroy guard and a counted preflight; adopted lanes are directories you already own and are never destroyable. Pipelines: a lane moves through pipeline stages. A stage the agent declares with evidence renders green; a stage inferred from the tool-event stream renders dashed amber and never counts as done. Detection is forward-only within a 30-minute window, and never writes the declared stage. Workspace: one page at /run with a lane grid, the selected lane's pipeline, and a full Claude console behind a disclosure.
This commit is contained in:
@@ -0,0 +1,97 @@
|
||||
apiVersion: argoproj.io/v1alpha1
|
||||
kind: AnalysisTemplate
|
||||
metadata:
|
||||
name: agent-monitor-canary-analysis
|
||||
namespace: agent-monitor
|
||||
labels:
|
||||
app.kubernetes.io/name: agent-monitor
|
||||
app.kubernetes.io/instance: agent-monitor
|
||||
app.kubernetes.io/version: "1.0.0"
|
||||
app.kubernetes.io/component: canary-analysis
|
||||
app.kubernetes.io/managed-by: kustomize
|
||||
spec:
|
||||
args:
|
||||
- name: service-name
|
||||
value: agent-monitor-canary
|
||||
- name: namespace
|
||||
value: agent-monitor
|
||||
metrics:
|
||||
# Success rate must be above 99%
|
||||
- name: success-rate
|
||||
interval: 60s
|
||||
count: 5
|
||||
successCondition: result[0] >= 0.99
|
||||
failureLimit: 2
|
||||
provider:
|
||||
prometheus:
|
||||
address: http://prometheus.monitoring.svc.cluster.local:9090
|
||||
query: |
|
||||
sum(
|
||||
rate(
|
||||
http_requests_total{
|
||||
namespace="{{args.namespace}}",
|
||||
service="{{args.service-name}}",
|
||||
code!~"5.."
|
||||
}[2m]
|
||||
)
|
||||
)
|
||||
/
|
||||
sum(
|
||||
rate(
|
||||
http_requests_total{
|
||||
namespace="{{args.namespace}}",
|
||||
service="{{args.service-name}}"
|
||||
}[2m]
|
||||
)
|
||||
)
|
||||
|
||||
# P99 latency must be under 500ms
|
||||
- name: p99-latency
|
||||
interval: 60s
|
||||
count: 5
|
||||
successCondition: result[0] < 500
|
||||
failureLimit: 2
|
||||
provider:
|
||||
prometheus:
|
||||
address: http://prometheus.monitoring.svc.cluster.local:9090
|
||||
query: |
|
||||
histogram_quantile(
|
||||
0.99,
|
||||
sum(
|
||||
rate(
|
||||
http_request_duration_milliseconds_bucket{
|
||||
namespace="{{args.namespace}}",
|
||||
service="{{args.service-name}}"
|
||||
}[2m]
|
||||
)
|
||||
) by (le)
|
||||
)
|
||||
|
||||
# Error rate must stay below 1%
|
||||
- name: error-rate
|
||||
interval: 60s
|
||||
count: 5
|
||||
successCondition: result[0] <= 0.01
|
||||
failureLimit: 2
|
||||
provider:
|
||||
prometheus:
|
||||
address: http://prometheus.monitoring.svc.cluster.local:9090
|
||||
query: |
|
||||
sum(
|
||||
rate(
|
||||
http_requests_total{
|
||||
namespace="{{args.namespace}}",
|
||||
service="{{args.service-name}}",
|
||||
code=~"5.."
|
||||
}[2m]
|
||||
)
|
||||
)
|
||||
/
|
||||
sum(
|
||||
rate(
|
||||
http_requests_total{
|
||||
namespace="{{args.namespace}}",
|
||||
service="{{args.service-name}}"
|
||||
}[2m]
|
||||
)
|
||||
)
|
||||
@@ -0,0 +1,124 @@
|
||||
apiVersion: apps/v1
|
||||
kind: Deployment
|
||||
metadata:
|
||||
name: agent-monitor-canary
|
||||
namespace: agent-monitor
|
||||
labels:
|
||||
app.kubernetes.io/name: agent-monitor
|
||||
app.kubernetes.io/instance: agent-monitor-canary
|
||||
app.kubernetes.io/version: "1.0.0"
|
||||
app.kubernetes.io/component: server
|
||||
app.kubernetes.io/managed-by: kustomize
|
||||
track: canary
|
||||
spec:
|
||||
replicas: 1
|
||||
revisionHistoryLimit: 5
|
||||
selector:
|
||||
matchLabels:
|
||||
app.kubernetes.io/name: agent-monitor
|
||||
app.kubernetes.io/instance: agent-monitor-canary
|
||||
track: canary
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
app.kubernetes.io/name: agent-monitor
|
||||
app.kubernetes.io/instance: agent-monitor-canary
|
||||
app.kubernetes.io/version: "1.0.0"
|
||||
app.kubernetes.io/component: server
|
||||
app.kubernetes.io/managed-by: kustomize
|
||||
track: canary
|
||||
annotations:
|
||||
prometheus.io/scrape: "true"
|
||||
prometheus.io/port: "4820"
|
||||
prometheus.io/path: "/api/health"
|
||||
spec:
|
||||
serviceAccountName: agent-monitor
|
||||
automountServiceAccountToken: false
|
||||
terminationGracePeriodSeconds: 30
|
||||
|
||||
securityContext:
|
||||
runAsNonRoot: true
|
||||
runAsUser: 1000
|
||||
runAsGroup: 1000
|
||||
fsGroup: 1000
|
||||
fsGroupChangePolicy: OnRootMismatch
|
||||
seccompProfile:
|
||||
type: RuntimeDefault
|
||||
|
||||
containers:
|
||||
- name: agent-monitor
|
||||
image: ${IMAGE_REGISTRY}/agent-monitor:canary
|
||||
imagePullPolicy: Always
|
||||
|
||||
ports:
|
||||
- name: http
|
||||
containerPort: 4820
|
||||
protocol: TCP
|
||||
|
||||
envFrom:
|
||||
- configMapRef:
|
||||
name: agent-monitor-config
|
||||
|
||||
resources:
|
||||
requests:
|
||||
memory: "256Mi"
|
||||
cpu: "200m"
|
||||
limits:
|
||||
memory: "1Gi"
|
||||
cpu: "1000m"
|
||||
|
||||
startupProbe:
|
||||
httpGet:
|
||||
path: /api/health
|
||||
port: http
|
||||
failureThreshold: 30
|
||||
periodSeconds: 2
|
||||
|
||||
readinessProbe:
|
||||
httpGet:
|
||||
path: /api/health
|
||||
port: http
|
||||
initialDelaySeconds: 5
|
||||
periodSeconds: 5
|
||||
timeoutSeconds: 3
|
||||
successThreshold: 1
|
||||
failureThreshold: 3
|
||||
|
||||
livenessProbe:
|
||||
httpGet:
|
||||
path: /api/health
|
||||
port: http
|
||||
initialDelaySeconds: 15
|
||||
periodSeconds: 15
|
||||
timeoutSeconds: 5
|
||||
successThreshold: 1
|
||||
failureThreshold: 3
|
||||
|
||||
securityContext:
|
||||
runAsNonRoot: true
|
||||
runAsUser: 1000
|
||||
runAsGroup: 1000
|
||||
readOnlyRootFilesystem: true
|
||||
allowPrivilegeEscalation: false
|
||||
capabilities:
|
||||
drop:
|
||||
- ALL
|
||||
|
||||
volumeMounts:
|
||||
- name: data
|
||||
mountPath: /app/data
|
||||
- name: tmp
|
||||
mountPath: /tmp
|
||||
|
||||
lifecycle:
|
||||
preStop:
|
||||
exec:
|
||||
command: ["sh", "-c", "sleep 5"]
|
||||
|
||||
volumes:
|
||||
- name: data
|
||||
persistentVolumeClaim:
|
||||
claimName: agent-monitor-data
|
||||
- name: tmp
|
||||
emptyDir:
|
||||
sizeLimit: 100Mi
|
||||
Reference in New Issue
Block a user