{
  "id": "databricks-platform-reliability-agent",
  "name": "Databricks Platform Reliability Agent",
  "version": "0.1.0",
  "type": "agent",
  "provider": "databricks",
  "harnesses": [
    "codex",
    "copilot",
    "claude-code",
    "cursor",
    "gemini",
    "kiro"
  ],
  "summary": "Diagnose and design compute, job, and pipeline reliability: operational evidence from system tables (`system.compute.*`, `system.lakeflow.*`, `system.billing.*`, `system.access.audit`), job and pipeline run reliability (timeouts, retries, dependencies, cascade failure), cluster policies as reliability and cost controls, instance pools and idle-termination behavior, quota and rate-limit headroom, managed disaster recovery posture and RPO/RTO discipline, and incident-evidence gathering from logs and table scans.",
  "source_type": "original",
  "official_docs": [
    "https://docs.databricks.com/aws/en/admin/system-tables/",
    "https://docs.databricks.com/aws/en/admin/system-tables/audit-logs",
    "https://docs.databricks.com/aws/en/resources/limits",
    "https://docs.databricks.com/aws/en/jobs/configure-task",
    "https://docs.databricks.com/aws/en/jobs/monitor",
    "https://docs.databricks.com/aws/en/admin/clusters/policy-definition",
    "https://docs.databricks.com/aws/en/compute/pools",
    "https://docs.databricks.com/aws/en/compute/choose-compute",
    "https://docs.databricks.com/aws/en/admin/disaster-recovery",
    "https://docs.databricks.com/aws/en/lakehouse-architecture/deployment-guide/ha-dr"
  ],
  "security_notes": "Diagnostic and design review of reliability controls, system-table evidence, and disaster-recovery posture. Never executes SQL or DDL, never queries a live workspace, never modifies a cluster policy or job configuration, and never triggers a failover. Reads job and pipeline definitions, cluster policies, system-table schemas (not customer data), and disaster-recovery documentation; a claim about incident root cause or recovery time that cannot be verified against log tables or job configurations is labeled assumption, never confirmed. A request for live workspace querying or failover testing enters the live-guard gate.",
  "last_verified": "2026-08-17",
  "path": "agents/databricks/databricks-platform-reliability-agent/",
  "harness_variants": {
    "codex": "agents/databricks/databricks-platform-reliability-agent/harnesses/codex.toml",
    "copilot": "agents/databricks/databricks-platform-reliability-agent/harnesses/copilot.agent.md",
    "claude-code": "agents/databricks/databricks-platform-reliability-agent/harnesses/claude-code.agent.md",
    "cursor": "agents/databricks/databricks-platform-reliability-agent/harnesses/cursor.agent.md",
    "gemini": "agents/databricks/databricks-platform-reliability-agent/harnesses/gemini.agent.md",
    "kiro-ide": "agents/databricks/databricks-platform-reliability-agent/harnesses/kiro-ide.agent.md",
    "kiro-cli": "agents/databricks/databricks-platform-reliability-agent/harnesses/kiro-cli.agent.json"
  },
  "companion_skills": [
    "databricks-platform-reliability"
  ],
  "execution_tier": "static-review",
  "lifecycle": "experimental",
  "author": "github: VincentChuWaiChow",
  "routing_keywords": [
    "system tables",
    "job failure",
    "retry",
    "timeout",
    "task dependency",
    "cluster policy",
    "instance pool",
    "auto-termination",
    "quota",
    "rate limit",
    "disaster recovery",
    "rpo",
    "rto",
    "failover",
    "incident"
  ]
}
