← Files TemporalARCHIVED FILE

skills/temporal-cloud-setup/scripts/provision.sh

104 KB · Oct 2, 2026 · 00:08 UTC

↓ Download file

#!/usr/bin/env bash
#
# provision.sh — deterministic executor for the temporal-cloud-setup skill.
#
# ===========================================================================
#  DO NOT EDIT THIS FILE WHILE RUNNING THE SKILL. Invoke it as shipped.
#  It is written to work UNCHANGED on every platform (macOS bash 3.2 / BSD
#  awk included). Do not reformat its punctuation, quoting, regexes, or flags.
#  If a prerelease flag has genuinely drifted, that is a deliberate human
#  maintenance edit -- not something an agent does mid-setup.
# ===========================================================================
#
# WHY THIS EXISTS
#   The skill's variance-prone work (create the namespace, mint + capture the
#   API key, write the client-config TOML) is the part where an agent tends to
#   improvise — polling, switching output formats, hunting the filesystem for an
#   account-id, or leaking the secret into a rendered diff. This script makes
#   that work identical on every run: fixed flags, baked-in error handling, and
#   the secret never leaves the script (it is written straight into the locked
#   TOML; only the non-secret KeyId is ever printed).
#
#   The agent INVOKES this script and PARSES its structured RESULT block — it
#   does not reassemble these commands itself. The wizard UX (phases, tracker,
#   checkpoints, narration rules) stays in SKILL.md; the determinism lives here.
#
# PORTABILITY
#   Pure bash, self-locating, no Claude/Codex-specific assumptions. Runs the
#   same under Claude Code and Codex (both just shell out to it).
#
# OUTPUT CONTRACT
#   Human-readable progress goes to STDERR (shown in the agent's tool block).
#   The machine-readable result goes to STDOUT as a single delimited block:
#
#       === RESULT ===
#       status=ok                # ok | error | skipped
#       <key>=<value>            # operation-specific (see each subcommand)
#       === END ===
#
#   On failure: status=error plus error_code=<slug> and message=<one line>.
#   The token (eyJ...) is NEVER emitted on stdout, stderr, or in any result key.
#
# CLI FLAGS ARE PINNED HERE (single source of truth)
#   These match the flags verified end-to-end against the prerelease CLI and
#   documented in references/unified-cli.md. The unified CLI is in PRERELEASE;
#   if a flag is rejected, fix it HERE (and in the reference) — one place, not
#   per-run in the agent's head.
#
set -uo pipefail

SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
SKILL_DIR="$(cd "$SCRIPT_DIR/.." && pwd)"

# ---- output helpers ---------------------------------------------------------

log()  { printf '%s\n' "$*" >&2; }                 # human progress -> stderr
die()  {                                            # structured error -> stdout
  local code="$1"; shift
  printf '=== RESULT ===\nstatus=error\nerror_code=%s\nmessage=%s\n=== END ===\n' \
    "$code" "$*"
  exit 1
}
result_open()  { printf '=== RESULT ===\n'; printf 'status=%s\n' "${1:-ok}"; }
result_kv()    { printf '%s=%s\n' "$1" "$2"; }
result_close() { printf '=== END ===\n'; }

require_cmd() { command -v "$1" >/dev/null 2>&1; }

# ---- secret redaction (single source of truth) ------------------------------
# redact  -> filter: read stdin, write a SCRUBBED copy to stdout. Removes the two
# secret shapes this skill can surface: an `api_key = <value>` / `apikey: <value>`
# assignment (value replaced with "(redacted)") and any JWT-shaped token (eyJ... ->
# eyJ...(redacted); API keys are JWTs). Factored out of cmd_create_key /
# cmd_verify_config so EVERY place that echoes captured CLI stderr uses one scrub —
# no secret ever reaches stdout, stderr, or a log unredacted. `sed -E` + `{6,}` are
# already used elsewhere here, so this stays within the macOS bash 3.2 / BSD sed target.
redact() {
  sed -E 's/([Aa][Pp][Ii][_-]?[Kk][Ee][Yy][[:space:]]*[=:]).*/\1 (redacted)/g; s/eyJ[A-Za-z0-9._-]{6,}/eyJ...(redacted)/g'
}

# ---- per-call timeout (single source of truth) ------------------------------
# run_bounded <secs> <cmd...>  -> run cmd with a per-call wall-clock timeout so a
# single wedged invocation can't block a polling loop. Prefers `timeout` / `gtimeout`
# when present (they also bound the whole process group); otherwise a pure-shell
# watchdog, because macOS ships no `timeout`: background the call, poll up to <secs>
# in 1s ticks, then TERM (and, as a backstop, KILL) it. `wait` reaps the child in
# every path (no zombies) and yields the command's REAL exit status when it finished
# in time; on expiry we return RUN_BOUNDED_TIMEOUT (124, matching GNU `timeout`) so a
# caller can tell "timed out" apart from "ran and returned non-zero". The caller owns
# stdout/stderr redirection (it is inherited by the backgrounded command as-is).
RUN_BOUNDED_TIMEOUT=124
run_bounded() {
  local secs="$1"; shift
  if require_cmd timeout;  then timeout  "$secs" "$@"; return $?; fi
  if require_cmd gtimeout; then gtimeout "$secs" "$@"; return $?; fi
  "$@" &
  local cpid=$! waited=0 rc
  while kill -0 "$cpid" 2>/dev/null; do
    if [ "$waited" -ge "$secs" ]; then
      kill -TERM "$cpid" 2>/dev/null
      sleep 1
      kill -KILL "$cpid" 2>/dev/null
      wait "$cpid" 2>/dev/null
      return "$RUN_BOUNDED_TIMEOUT"
    fi
    sleep 1
    waited=$(( waited + 1 ))
  done
  wait "$cpid"; rc=$?
  return "$rc"
}

# ---- config-file location (TEMPORAL_CONFIG_FILE wins, else per-OS default) ---

config_path() {
  if [ -n "${TEMPORAL_CONFIG_FILE:-}" ]; then
    printf '%s' "$TEMPORAL_CONFIG_FILE"; return
  fi
  case "$(uname -s)" in
    Darwin) printf '%s' "$HOME/Library/Application Support/temporalio/temporal.toml" ;;
    Linux)  printf '%s' "${XDG_CONFIG_HOME:-$HOME/.config}/temporalio/temporal.toml" ;;
    *)      printf '%s' "$HOME/.config/temporalio/temporal.toml" ;;   # best effort
  esac
}

# ---- per-SDK repo table (single source of truth; mirrors references/sdk-cloud.md)
# Branch is always money-transfer-project-cloud-setup (pre-wired for Cloud).

repo_url_for_sdk() {
  case "$1" in
    python)     printf 'https://github.com/temporalio/money-transfer-project-template-python' ;;
    go)         printf 'https://github.com/temporalio/money-transfer-project-template-go' ;;
    ts|typescript) printf 'https://github.com/temporalio/money-transfer-project-template-ts' ;;
    java)       printf 'https://github.com/temporalio/money-transfer-project-java' ;;
    dotnet|.net|net) printf 'https://github.com/temporalio/money-transfer-project-template-dotnet' ;;
    ruby)       printf 'https://github.com/temporalio/money-transfer-project-template-ruby' ;;
    *)          printf '' ;;
  esac
}

# ---- per-SDK run matrix (verified against the money-transfer-project-cloud-setup
# branches; mirrors references/sdk-cloud.md). Worker command is long-running; the
# starter blocks until the Workflow returns its result. These are the SINGLE source
# of truth for the run step so the agent never guesses entrypoints/targets.
worker_cmd_for() {
  case "$1" in
    python)          printf 'source env/bin/activate && python run_worker.py' ;;
    go)              printf 'go run worker/main.go' ;;
    ts|typescript)   printf 'npm run worker' ;;
    java)            printf 'mvn -q compile exec:java -Dexec.mainClass=moneytransferapp.MoneyTransferWorker -Dorg.slf4j.simpleLogger.defaultLogLevel=warn' ;;
    dotnet|.net|net) printf 'dotnet run --project MoneyTransferWorker' ;;
    ruby)            printf 'ruby worker.rb' ;;
    *)               printf '' ;;
  esac
}
starter_cmd_for() {
  case "$1" in
    python)          printf 'source env/bin/activate && python run_workflow.py' ;;
    go)              printf 'go run start/main.go' ;;
    ts|typescript)   printf 'npm run client' ;;
    java)            printf 'mvn -q compile exec:java -Dexec.mainClass=moneytransferapp.TransferApp -Dorg.slf4j.simpleLogger.defaultLogLevel=warn' ;;
    dotnet|.net|net) printf 'dotnet run --project MoneyTransferClient' ;;
    ruby)            printf 'ruby starter.rb' ;;
    *)               printf '' ;;
  esac
}
# task-queue name the worker polls — the readiness probe checks for a poller here.
task_queue_for() {
  case "$1" in
    python|go)          printf 'TRANSFER_MONEY_TASK_QUEUE' ;;
    java|dotnet|.net|net) printf 'MONEY_TRANSFER_TASK_QUEUE' ;;
    ts|typescript|ruby) printf 'money-transfer' ;;
    *)                  printf '' ;;
  esac
}

# ---- per-(SDK, package-manager) adaptation matrices --------------------------
# Single source of truth for which managers each SDK sample ACTUALLY supports,
# derived from the money-transfer-project-cloud-setup branches
# (see references/sdk-cloud.md). Only python and ts ship a choice; the rest are
# single-manager. We never offer a manager the sample can't honor (e.g. poetry on
# the python sample, which ships no pyproject.toml).

# managers_for <sdk> -> space-separated candidate managers in PREFERENCE order
# (default = first one whose binary is on PATH). For SDKs that ship a lockfile the
# lockfile's manager leads, so the default honors the committed lockfile (ts->npm).
managers_for() {
  case "$1" in
    python)          printf 'pip uv' ;;        # sample has no manifest; pip default, uv faster override
    ts|typescript)   printf 'npm pnpm yarn' ;; # package-lock.json committed -> npm leads
    go)              printf 'go' ;;
    java)            printf 'maven' ;;         # pom.xml only; no mvnw/gradlew/build.gradle
    dotnet|.net|net) printf 'dotnet' ;;
    ruby)            printf 'bundler' ;;
    *)               printf '' ;;
  esac
}

# runtime_for <sdk> -> the language-runtime binary whose version we check/warn on.
runtime_for() {
  case "$1" in
    python)          printf 'python3' ;;
    ts|typescript)   printf 'node' ;;
    go)              printf 'go' ;;
    java)            printf 'java' ;;
    dotnet|.net|net) printf 'dotnet' ;;
    ruby)            printf 'ruby' ;;
    *)               printf '' ;;
  esac
}

# manager_bin <manager> -> the executable to look for on PATH for that manager.
manager_bin() {
  case "$1" in
    pip)     printf 'python3' ;;   # pip is invoked as `python3 -m pip`
    uv)      printf 'uv' ;;
    npm)     printf 'npm' ;;
    pnpm)    printf 'pnpm' ;;
    yarn)    printf 'yarn' ;;
    go)      printf 'go' ;;
    maven)   printf 'mvn' ;;
    dotnet)  printf 'dotnet' ;;
    bundler) printf 'bundle' ;;
    *)       printf '' ;;
  esac
}

# min_version_for <tool> -> minimum tested version (advisory; a shortfall is a
# WARNING with remediation, never an auto-fix or a hard block). Empty = no minimum.
min_version_for() {
  case "$1" in
    python3) printf '3.8' ;;
    node)    printf '18' ;;        # Node 16 is EOL; @temporalio/* targets 18+
    go)      printf '1.21' ;;
    java)    printf '8' ;;         # Temporal Java SDK supports JDK 8+ (deliberately below the 11 line)
    dotnet)  printf '6.0' ;;
    ruby)    printf '3.0' ;;
    uv)      printf '0.1' ;;
    npm)     printf '8' ;;
    pnpm)    printf '8' ;;
    yarn)    printf '1.22' ;;
    maven)   printf '3.6' ;;       # mvn
    bundler) printf '2.0' ;;       # bundle
    *)       printf '' ;;
  esac
}

# install_cmd_for <sdk> <manager> -> the dependency-install command string.
# SINGLE SOURCE OF TRUTH for installs, shared by cmd_install_deps, the internal
# provision-and-scaffold path, and `preview`. Empty = unsupported (sdk,manager) pair.
install_cmd_for() {
  local sdk="$1" mgr="$2"
  case "$sdk" in
    python)
      case "$mgr" in
        pip) printf 'python3 -m venv env && . env/bin/activate && python -m pip install -q temporalio' ;;
        uv)  printf 'uv venv env && . env/bin/activate && uv pip install -q temporalio' ;;
        *)   printf '' ;;
      esac ;;
    ts|typescript)
      case "$mgr" in
        npm)  printf 'npm install' ;;
        pnpm) printf 'pnpm install' ;;
        yarn) printf 'yarn install' ;;
        *)    printf '' ;;
      esac ;;
    go)              [ "$mgr" = go ]      && printf 'go mod download' ;;
    java)            [ "$mgr" = maven ]   && printf 'mvn -q -DskipTests dependency:resolve' ;;
    dotnet|.net|net) [ "$mgr" = dotnet ]  && printf 'dotnet restore' ;;
    ruby)            [ "$mgr" = bundler ] && printf 'bundle install' ;;
    *)               printf '' ;;
  esac
}

# install_location_note <sdk> -> short phrase stating WHERE the install writes, so the
# disclosure is explicit when deps land in a shared GLOBAL cache outside the repo
# (Go/Java/.NET/Ruby) vs. a location local to the clone (Python venv, TS node_modules).
install_location_note() {
  case "$1" in
    python)          printf 'into a local venv (env/) inside the repo' ;;
    ts|typescript)   printf 'into node_modules inside the repo' ;;
    go)              printf 'into the shared Go module cache - GLOBAL, outside the repo (~/go/pkg/mod)' ;;
    java)            printf 'into the shared Maven cache - GLOBAL, outside the repo (~/.m2)' ;;
    dotnet|.net|net) printf 'into the global NuGet cache - GLOBAL, outside the repo (~/.nuget/packages)' ;;
    ruby)            printf 'into globally-installed gems - GLOBAL, outside the repo' ;;
    *)               printf 'into the repo' ;;
  esac
}

# resolve_pkg_inputs <sdk> <manager-or-empty>  — resolve the package manager and its
# install command for an SDK. Sets globals PKG_MGR / PKG_ICMD / PKG_CANDIDATES and
# returns 0; returns 2 if the SDK has no manager mapping, 3 if the chosen manager isn't
# supported. Uses globals + a return code (not stdout + `die`) on purpose: a `die` inside
# a `$(...)` call would exit only the subshell, so the CALLER maps the code to the error.
resolve_pkg_inputs() {
  local sdk="$1" mgr_in="$2"
  PKG_CANDIDATES="$(managers_for "$sdk")"
  [ -n "$PKG_CANDIDATES" ] || return 2
  PKG_MGR="${mgr_in:-${PKG_CANDIDATES%% *}}"
  PKG_ICMD="$(install_cmd_for "$sdk" "$PKG_MGR")"
  [ -n "$PKG_ICMD" ] || return 3
  return 0
}

# die_pkg <rc> <sdk>  — map resolve_pkg_inputs' non-zero return to the right error.
die_pkg() {
  case "$1" in
    2) die unknown-sdk "no package-manager mapping for SDK '$2'" ;;
    3) die unsupported-manager "manager '$PKG_MGR' not supported for SDK '$2' (supported: $(printf '%s' "$PKG_CANDIDATES" | tr ' ' ','))" ;;
  esac
}

# ver_norm <string> -> leading dotted-numeric version (e.g. "go1.22.0"->"1.22.0").
ver_norm() { printf '%s' "$1" | grep -oE '[0-9]+(\.[0-9]+){0,3}' | head -n1; }

# ver_ge <have> <want> -> exit 0 if HAVE >= WANT (component-wise, bash 3.2-safe).
# Splits on '.' via a saved/restored IFS — NOT a `... | read` pipe, which would
# run read in a subshell and lose the captured components.
ver_ge() {
  local have want h1 h2 h3 w1 w2 w3 oldifs
  have="$(ver_norm "$1")"; want="$(ver_norm "$2")"
  [ -n "$have" ] || return 1
  [ -n "$want" ] || return 0
  oldifs="$IFS"; IFS=.
  set -- $have; h1="${1:-0}"; h2="${2:-0}"; h3="${3:-0}"
  set -- $want; w1="${1:-0}"; w2="${2:-0}"; w3="${3:-0}"
  IFS="$oldifs"
  if [ "$h1" -ne "$w1" ]; then [ "$h1" -gt "$w1" ]; return; fi
  if [ "$h2" -ne "$w2" ]; then [ "$h2" -gt "$w2" ]; return; fi
  [ "$h3" -ge "$w3" ]
}

# tool_version <runtime-or-manager-binary> -> a bare version string, or empty if
# absent/unparseable. Handles the binaries whose --version output is non-standard
# (go, java's legacy 1.8 scheme on stderr).
tool_version() {
  local bin="$1" raw v
  require_cmd "$bin" || return 0
  case "$bin" in
    go)   raw="$(go version 2>/dev/null)" ;;
    java) raw="$(java -version 2>&1 | head -n1)" ;;       # java prints version to stderr
    mvn)  raw="$(mvn -version 2>/dev/null | head -n1)" ;;
    *)    raw="$("$bin" --version 2>&1 | head -n1)" ;;
  esac
  v="$(ver_norm "$raw")"
  # Java's legacy "1.8.0_292" scheme: map 1.N -> N so numeric compare treats 8 as 8.
  if [ "$bin" = java ]; then
    case "$v" in 1.*) v="${v#1.}" ;; esac
  fi
  printf '%s' "$v"
}

# ---- JSON field extraction (jq preferred, python3 fallback) -----------------

json_field() {  # json_field <file> <field>
  local file="$1" field="$2"
  if require_cmd jq; then
    jq -r --arg f "$field" '.[$f] // empty' "$file" 2>/dev/null
  elif require_cmd python3; then
    python3 -c 'import json,sys
try:
    d=json.load(open(sys.argv[1]))
    v=d.get(sys.argv[2])
    print(v if v is not None else "")
except Exception:
    print("")' "$file" "$field"
  else
    printf ''
  fi
}

# =============================================================================
# Subcommands
# =============================================================================

cmd_preflight() {
  # SDK is optional and only used to echo the repo_url. Accept --sdk (the
  # convention every other subcommand uses) and a bare positional for back-compat.
  local sdk=""
  while [ $# -gt 0 ]; do
    case "$1" in
      --sdk) sdk="$2"; shift 2 ;;
      -*) die bad-args "unknown arg: $1" ;;
      *) sdk="$1"; shift ;;
    esac
  done
  log "Preflight checks..."
  local os; os="$(uname -s)"
  local cfg; cfg="$(config_path)"

  local warnings=()
  require_cmd git || warnings+=("git-missing")
  if ! require_cmd jq && ! require_cmd python3; then
    warnings+=("no-json-parser")   # needed to capture the API key safely
  fi
  if [ "$os" = "Darwin" ] && ! require_cmd brew; then
    warnings+=("brew-missing")     # only a problem if the CLI also isn't installed
  fi
  # Stray TEMPORAL_* env vars OVERRIDE the TOML profile (env-config precedence).
  local stray=()
  [ -n "${TEMPORAL_ADDRESS:-}" ]   && stray+=("TEMPORAL_ADDRESS")
  [ -n "${TEMPORAL_NAMESPACE:-}" ] && stray+=("TEMPORAL_NAMESPACE")
  [ -n "${TEMPORAL_API_KEY:-}" ]   && stray+=("TEMPORAL_API_KEY")

  # NB: no up-front internet probe. A coarse HTTPS reachability check can't see the
  # case that actually bites -- a PARTIAL block where login/whoami work (offline /
  # loopback) but the gRPC Cloud API is blocked -- and it risks false-positives behind
  # a captive portal / proxy. The authoritative connectivity gate is the post-login
  # region pulse in cmd_regions (validates output, not exit code); a fully-offline user
  # also fails loudly at install-cli / login. So we do NOT gate preflight on the network.

  # Config-dir writable: writing the client-config TOML is OUR op -- no external
  # tool emits an error for us -- and today it fails LATE, after a billable API key
  # is minted. A tiny create/write/remove probe catches an unwritable config dir up
  # front. (The one fs check worth keeping; work-dir/disk checks are left to git/npm/
  # pip, which already fail loudly on them.)
  local cfg_dir; cfg_dir="$(dirname "$cfg")"
  local probe="$cfg_dir/.tcloud-write-probe.$$"
  if ! ( mkdir -p "$cfg_dir" 2>/dev/null && : > "$probe" 2>/dev/null ); then
    die config-dir-unwritable "Can't write to the Temporal config directory ($cfg_dir). Fix its permissions, or set TEMPORAL_CONFIG_FILE to a writable path, then re-run -- otherwise the client-config TOML can't be saved after the API key is minted."
  fi
  rm -f "$probe" 2>/dev/null

  result_open ok
  result_kv os "$os"
  result_kv config_path "$cfg"
  result_kv cli_installed "$(cloud_cli_present && echo true || echo false)"
  [ -n "$sdk" ] && result_kv repo_url "$(repo_url_for_sdk "$sdk")"
  result_kv warnings "$(IFS=,; echo "${warnings[*]:-}")"
  result_kv stray_env "$(IFS=,; echo "${stray[*]:-}")"
  result_close
}

# cmd_detect_tools: local-setup adaptation. For the chosen SDK,
# detect which package managers the sample supports AND are installed, pick a
# deterministic default (first available in preference order), report tool
# versions, and surface discrepancies (missing runtime/manager, version-too-old)
# so the agent can present them EARLY (Phase 1) — not deep in a later phase.
# Side-effect free: only reads versions, never installs or clones. Deterministic:
# same environment -> same default every run (no state file).
cmd_detect_tools() {
  local sdk=""
  while [ $# -gt 0 ]; do
    case "$1" in
      --sdk) sdk="$2"; shift 2 ;;
      *) die bad-args "unknown arg: $1" ;;
    esac
  done
  [ -n "$sdk" ] || die bad-args "--sdk is required"
  local candidates; candidates="$(managers_for "$sdk")"
  [ -n "$candidates" ] || die unknown-sdk "no package-manager mapping for SDK '$sdk'"

  log "Detecting tools for '$sdk'..."
  local runtime rt_ver rt_min available="" default="" versions="" disc=""
  runtime="$(runtime_for "$sdk")"

  # Runtime: presence + version vs. minimum.
  rt_ver="$(tool_version "$runtime")"
  rt_min="$(min_version_for "$runtime")"
  if [ -z "$rt_ver" ]; then
    disc="${disc:+$disc,}tool-missing:$runtime"
  else
    versions="${versions:+$versions,}$runtime@$rt_ver"
    if [ -n "$rt_min" ] && ! ver_ge "$rt_ver" "$rt_min"; then
      disc="${disc:+$disc,}version-too-old:$runtime@$rt_ver(min$rt_min)"
    fi
  fi

  # Managers: build the available list (preference order) + record versions/discrepancies.
  local m bin ver vmin
  for m in $candidates; do
    bin="$(manager_bin "$m")"
    if require_cmd "$bin"; then
      available="${available:+$available,}$m"
      [ -z "$default" ] && default="$m"
      ver="$(tool_version "$bin")"
      [ -n "$ver" ] && versions="${versions:+$versions,}$m@$ver"
      vmin="$(min_version_for "$m")"
      if [ -n "$ver" ] && [ -n "$vmin" ] && ! ver_ge "$ver" "$vmin"; then
        disc="${disc:+$disc,}version-too-old:$m@$ver(min$vmin)"
      fi
    fi
  done
  # No supported manager installed at all -> name the preferred (first candidate) as missing.
  if [ -z "$available" ]; then
    disc="${disc:+$disc,}manager-not-found:${candidates%% *}"
  fi

  result_open ok
  result_kv sdk "$sdk"
  result_kv runtime "${rt_ver:+$runtime@$rt_ver}"
  result_kv managers "$available"
  result_kv default "$default"
  result_kv candidates "$(printf '%s' "$candidates" | tr ' ' ',')"
  result_kv versions "$versions"
  result_kv discrepancies "$disc"
  result_close
}

# cloud_cli_present -> 0 iff the unified CLI's `cloud` command GROUP works. `temporal`
# alone can be the OSS CLI without the cloud group, so test the group, not just the
# binary. Shared by preflight (cli_installed) and install-cli so the two presence
# checks never disagree. Probe with `temporal cloud --help`: it exits 0 whenever the
# `cloud` group exists and needs no auth. (Do NOT use `temporal cloud version` — on the
# prerelease CLI `version` is a top-level flag, not a `cloud` subcommand, so it exits 1
# even when the CLI is installed; that false negative sent install-cli into a reinstall
# loop that died with install-verify-failed.)
cloud_cli_present() { temporal cloud --help >/dev/null 2>&1; }

# cli_version -> the real CLI version line (e.g. "temporal version 1.7.2 (...)").
# NB: from `temporal --version`, NOT `temporal cloud version` (the latter prints help,
# which is the "version=The Temporal Cloud CLI provides commands…" bug).
cli_version() { temporal --version 2>/dev/null | head -n1; }

cmd_install_cli() {
  if cloud_cli_present; then
    # BETA stopgap (PE-79): the prerelease temporal-cloud CLI has no meaningful
    # version numbers yet, so we can't do a real "is it out of date?" check. For the
    # beta period we ALWAYS try to pull the latest from the prerelease tap instead of
    # skipping. Never yank a working install: if we can't update (Homebrew missing,
    # non-macOS host, or brew errors) we warn and proceed with the CLI that's there.
    # Replace this with a real version comparison once the CLI ships versions.
    local up_out up_rc
    case "$(uname -s)" in
      Darwin)
        if require_cmd brew; then
          log "Temporal CLI present — updating to the latest prerelease..."
          up_out="$(brew upgrade temporalio/prerelease/temporal-cloud 2>&1)"; up_rc=$?
          result_open ok
          # The updated/up-to-date label is best-effort (matched against brew's
          # unstructured output) and NON-load-bearing: every branch here emits
          # status=ok and proceeds, so a mislabel never breaks a working install.
          if [ "$up_rc" -ne 0 ]; then
            log "brew upgrade failed; keeping the working CLI. Output:"
            log "$up_out"
            result_kv update failed
          elif printf '%s\n' "$up_out" | grep -qi 'upgrading'; then
            result_kv update updated
          else
            result_kv update up-to-date
          fi
          result_kv version "$(cli_version)"
          result_close; return
        fi
        log "Temporal CLI present but Homebrew not found — can't auto-update; proceeding with the installed CLI. Install Homebrew (https://brew.sh) or download temporal-cloud from https://github.com/temporalio/cloud-cli/releases/latest to update manually."
        result_open ok
        result_kv update skipped
        result_kv reason brew-missing
        result_kv version "$(cli_version)"
        result_close; return ;;
      *)
        log "Temporal CLI present on a non-macOS host — no prerelease tap to update from; proceeding with the installed CLI. Update temporal-cloud manually from https://github.com/temporalio/cloud-cli/releases/latest."
        result_open ok
        result_kv update skipped
        result_kv reason unsupported-os
        result_kv version "$(cli_version)"
        result_close; return ;;
    esac
  fi
  case "$(uname -s)" in
    Darwin)
      if require_cmd brew; then
        log "Installing temporal-cloud via Homebrew..."
        if brew install temporalio/prerelease/temporal-cloud >&2; then
          # (b) Verify the install actually took, rather than trusting brew's exit.
          if ! cloud_cli_present; then
            die install-verify-failed "brew reported success but 'temporal cloud --help' still fails — check PATH / shell rehash, then re-run."
          fi
          result_open ok
          result_kv installed_via brew
          result_kv version "$(cli_version)"
          result_close; return
        fi
        die install-failed "brew install temporalio/prerelease/temporal-cloud failed; see output above"
      fi
      die brew-missing "Homebrew not found. Install from https://brew.sh, or download temporal-cloud from https://github.com/temporalio/cloud-cli/releases/latest and put it on PATH."
      ;;
    *)
      die manual-install "Download temporal-cloud from https://github.com/temporalio/cloud-cli/releases/latest, extract, and put it on PATH; then re-run."
      ;;
  esac
}

cmd_login() {
  # `temporal cloud login` opens a browser and BLOCKS until the user finishes.
  log "Opening browser for Temporal Cloud sign-in... complete it in the browser."
  if ! temporal cloud login >&2; then
    die login-failed "temporal cloud login did not complete; finish the browser sign-in and retry."
  fi
  local who; who="$(temporal cloud whoami 2>/dev/null | head -n1)"
  if [ -z "$who" ]; then
    die not-authenticated "whoami returned empty after login — sign-in may not have completed."
  fi
  result_open ok; result_kv identity "$who"; result_close
}

cmd_regions() {
  # `whoami` is the cheap "are we even signed in" check -- it proves a CREDENTIAL is
  # PRESENT, nothing more. It runs OFFLINE (cached token, no live API call), so it is
  # NOT a connectivity signal: never treat a passing whoami as proof the Cloud API is
  # reachable. The region fetch below is the authoritative connectivity + authz pulse.
  if ! temporal cloud whoami >/dev/null 2>&1; then
    die not-authenticated "Not signed in. Run the login step first."
  fi

  # ---- POST-LOGIN CONNECTIVITY PULSE (the one authoritative "can we actually reach
  # the Cloud API with these creds" gate) --------------------------------------------
  # The prerelease CLI returns exit 0 even when the network call fails, so we validate
  # the OUTPUT, never the exit code (exit-0 hardening -- deliberately narrowed to this
  # pulse + the preflight probe; downstream steps trust the connection this establishes).
  # `region list` is non-empty for ANY valid account, so an empty result is an
  # unambiguous "couldn't reach the API" -- almost always a blocked/partial network
  # (login + whoami succeed offline; the gRPC API is blocked). We STOP with
  # cloud-unreachable and do NOT loop on re-auth: re-auth cannot fix a blocked network,
  # and that misdiagnosis (empty list -> "confirm auth, re-run") is the exact bug this
  # pulse closes. We reuse this single fetch for the SELECTION step below -- connectivity
  # is validated HERE, in one place; selection carries no network logic and no extra call.
  log "Listing available Cloud regions..."
  local err_file; err_file="$(mktemp "${TMPDIR:-/tmp}/tcloud-regions.XXXXXX")"
  local list; list="$(temporal cloud region list 2>"$err_file")"
  local err; err="$(cat "$err_file" 2>/dev/null)"; rm -f "$err_file"
  if [ -z "$list" ]; then
    # Empty output = failed pulse. Surface any stderr as a hint, but never DEPEND on it:
    # the exit code lied, and the text may be empty or drift between prerelease builds.
    die cloud-unreachable "Temporal Cloud returned no regions -- the Cloud API is unreachable with your current session. This is a network/sandbox block, NOT missing auth (login and whoami work offline). Do NOT re-run login. Check outbound network and your sandbox's egress to the Temporal Cloud gRPC API (*.tmprl.cloud), then re-run.${err:+ (CLI stderr: $err)}"
  fi
  # Region guard: flag regions whose CloudProvider renders as UNKNOWN. On some accounts
  # those (e.g. azure-centralus) ACCEPT a namespace create but never provision it (a
  # phantom that later stalls await-namespace). Emit them so the agent can warn the user
  # and steer to a known AWS/GCP region. ($1 = region id, $2 = CloudProvider column.)
  local unsupported
  unsupported="$(printf '%s\n' "$list" | awk '$2=="UNKNOWN"{print $1}' | tr '\n' ',' | sed 's/,$//')"
  # Echo the raw list to stderr so the agent can present it; emit nothing
  # secret. Recommendation is left to the agent (timezone heuristic) so the
  # script never invents a region that isn't in the list.
  printf '%s\n' "$list" >&2
  result_open ok
  [ -n "$unsupported" ] && result_kv unsupported_regions "$unsupported"
  result_close
}

# user_tag  -> up to 8 [a-z0-9] from the Temporal user's email LOCAL-PART (everything
# before '@'; the '@' and domain are never included, so the name is never email-shaped).
# Falls back to 'user' if the email is unavailable or sanitizes to empty.
user_tag() {
  local email=""
  if require_cmd jq; then
    email="$(temporal cloud whoami --output json 2>/dev/null | jq -r '.user.spec.email // empty' 2>/dev/null)"
  elif require_cmd python3; then
    email="$(temporal cloud whoami --output json 2>/dev/null | python3 -c 'import json,sys
try:
    print(json.load(sys.stdin).get("user",{}).get("spec",{}).get("email","") or "")
except Exception:
    print("")' 2>/dev/null)"
  fi
  local tag
  tag="$(printf '%s' "$email" | tr '[:upper:]' '[:lower:]' | cut -d@ -f1 | tr -cd 'a-z0-9' | cut -c1-8)"
  [ -n "$tag" ] || tag="user"
  printf '%s' "$tag"
}

# rand_tag  -> 8 random [a-z0-9] for per-namespace uniqueness (portable: /dev/urandom).
rand_tag() {
  local r
  r="$(LC_ALL=C tr -dc 'a-z0-9' < /dev/urandom 2>/dev/null | head -c 8)"
  [ "${#r}" -eq 8 ] || r="$(date +%s | tail -c 9)"   # fallback if /dev/urandom is unavailable
  printf '%s' "$r"
}

# namespace_name_for  -> unique name:  quickstartai-<user>-<random>
#   <user>   = up to 8 chars of the Temporal user's email local-part (see user_tag)
#   <random> = 8 random [a-z0-9] (see rand_tag) for per-namespace uniqueness
# No SDK / timestamp / telemetry marker — just an anonymized per-user tag + uniqueness.
# Cloud appends .<account-id> itself, which does NOT count toward the <=39 char limit.
# (The SDK arg is ignored now; kept for call-site compatibility.)
namespace_name_for() {
  local name
  name="quickstartai-$(user_tag)-$(rand_tag)"
  [ "${#name}" -le 39 ] || die name-too-long "namespace name '$name' exceeds 39 chars"
  printf '%s' "$name"
}

# namespace_active_info <exact-name>  -> prints "<handle>|<grpc-address>" + returns 0 when
# the namespace EXISTS and is ACTIVE. Return codes let the caller tell a phantom (a create
# that was accepted but never provisions) from a slow-but-real provision:
#   0 = ACTIVE (handle|address printed)
#   3 = EXISTS but not yet ACTIVE (ACTIVATING) — genuinely provisioning
#   1 = ABSENT (not in the list) or list unreadable
#
# Why the server-side exact `--name` filter (not a plain `namespace list`): an account
# can hold hundreds of namespaces and `namespace list` PAGES at 100, so scanning a single
# unfiltered page silently misses ours when its name sorts past the first page. The old
# resolver did exactly that, then fell back to CONSTRUCTING `<name>.<acct>` from another
# entry and returning immediately — so it reported "provisioned" on the first poll
# without ever confirming the namespace existed or was ready, leaving create-key/await-auth
# to run against an endpoint that wasn't serving yet ("no children to pick from"). The
# filtered result also carries the authoritative endpoint, so we COPY the address from
# the API response — never assemble it. Primary = `endpoints.mtls_grpc_address` (the
# tmprl.cloud form API-key auth uses), fallback = `endpoints.grpc_address` (regional);
# both straight from the response. (Constructing it broke for providers whose endpoint
# isn't the `<handle>.tmprl.cloud` shape.) Needs jq or python3.
#
# State: this CLI build reports ACTIVE as state==3 (the `namespace list` table column
# renders 3 as "ACTIVE"); a just-created namespace is state==1 (ACTIVATING) for a few
# minutes. API-key auth uses mtls_grpc_address (`<handle>.tmprl.cloud:7233`), which only
# starts serving once the namespace reaches ACTIVE.
namespace_active_info() {
  local name="$1" raw fields handle state mtls grpc addr
  raw="$(temporal cloud namespace list --name "$name" -o json 2>/dev/null)" || return 1
  [ -n "$raw" ] || return 1
  # Pull all four fields in ONE pass (jq, or python3 fallback), pipe-delimited. Use '|'
  # (not tab) as the delimiter so IFS-splitting preserves EMPTY fields — none of the
  # values (handle, numeric state, host:port endpoints) ever contains a '|'.
  if require_cmd jq; then
    fields="$(printf '%s' "$raw" | jq -r 'if (.Namespaces|length)>0 then (.Namespaces[0] | [(.namespace//""),((.state // "")|tostring),(.endpoints.mtls_grpc_address//""),(.endpoints.grpc_address//"")] | join("|")) else "" end' 2>/dev/null)"
  elif require_cmd python3; then
    fields="$(printf '%s' "$raw" | python3 -c 'import json,sys
try:
    a=json.load(sys.stdin).get("Namespaces",[]); x=a[0] if a else {}
    e=x.get("endpoints",{}) or {}
    print("|".join([str(x.get("namespace","")), str(x.get("state","")), str(e.get("mtls_grpc_address","")), str(e.get("grpc_address",""))]) if a else "")
except Exception:
    print("")')"
  else
    return 1
  fi
  IFS='|' read -r handle state mtls grpc <<EOF
$fields
EOF
  [ -n "$handle" ] || return 1                          # absent — not in the list
  [ "$state" = "3" ] || return 3                        # exists but ACTIVATING (state 1), not ACTIVE yet
  # Copy the endpoint from the API: mtls primary, regional grpc fallback. Only as an
  # absolute last resort (both fields absent — not seen on an ACTIVE namespace) do we
  # construct, so a missing field can't strand the run.
  addr="$mtls"; [ -n "$addr" ] || addr="$grpc"
  [ -n "$addr" ] || addr="${handle}.tmprl.cloud:7233"
  printf '%s|%s' "$handle" "$addr"
}

# await_active_namespace <exact-name> <max-secs>  -> poll until the namespace is ACTIVE.
# Echoes "<handle>|<address>" + returns 0 on success. Return codes distinguish the two
# failure modes so the caller can message them differently:
#   2 = PHANTOM — the namespace never appeared in the list within NS_PHANTOM_GRACE_SECS.
#       A real create shows up as ACTIVATING within seconds; persistent absence means the
#       async create was accepted but isn't provisioning (commonly an unavailable region,
#       e.g. azure-centralus whose provider reads UNKNOWN). Fail FAST instead of burning
#       the full --max-secs.
#   1 = TIMEOUT — it appeared (ACTIVATING) but didn't reach ACTIVE within --max-secs.
# This is where the provisioning wait BELONGS (it used to leak into await-auth, which then
# mislabeled a still-provisioning namespace as "API key rejected").
await_active_namespace() {
  local name="$1" max="${2:-600}" waited=0 interval=2 info rc ever_seen=0
  local grace="${NS_PHANTOM_GRACE_SECS:-75}"
  while :; do
    info="$(namespace_active_info "$name")"; rc=$?
    [ "$rc" -eq 0 ] && { printf '%s' "$info"; return 0; }
    [ "$rc" -eq 3 ] && ever_seen=1                       # appeared (ACTIVATING) -> genuinely provisioning
    [ "$ever_seen" -eq 0 ] && [ "$waited" -ge "$grace" ] && return 2   # never showed up -> phantom
    [ "$waited" -ge "$max" ] && return 1
    sleep "$interval"; waited=$(( waited + interval ))
  done
}

# cmd_start_namespace: TRUE fire-and-forget. Submit the create with --async
# and return immediately — provisioning runs SERVER-SIDE (no local background job to
# survive across tool calls, so this works identically on Claude Code/Codex/Cursor).
# We GENERATE the name (namespace_name_for), so we don't need the create's output;
# join later with `await-namespace --name <name>`, which reads the handle from
# `namespace list -o jsonl`.
cmd_start_namespace() {
  local sdk="" region=""
  while [ $# -gt 0 ]; do
    case "$1" in
      --sdk) sdk="$2"; shift 2 ;;
      --region) region="$2"; shift 2 ;;
      *) die bad-args "unknown arg: $1" ;;
    esac
  done
  [ -n "$sdk" ]    || die bad-args "--sdk is required"
  [ -n "$region" ] || die bad-args "--region is required"
  temporal cloud whoami >/dev/null 2>&1 || die not-authenticated "Not signed in; run login again."

  local name out rc
  name="$(namespace_name_for "$sdk")"
  log "Submitting namespace '$name' in '$region' (--async; provisions on Temporal's servers)..."
  # --async returns as soon as the create is accepted; the namespace then provisions
  # server-side over a few minutes. A non-zero exit here is a SUBMIT failure (region/
  # name), surfaced immediately so a bad value never leaves a half-created namespace.
  # Capture rc on the SAME line — a later `printf | tail` (or the `if !` negation) would
  # reset $? to the pipeline's status and mask the real create exit code.
  out="$(temporal cloud namespace create \
        --name "$name" --region "$region" \
        --api-key-auth-enabled --retention-days 30 --auto-confirm --async 2>&1)"; rc=$?
  if [ "$rc" -ne 0 ]; then
    printf '%s\n' "$out" | tail -n 20 >&2
    die create-rejected "namespace create --async was rejected (exit $rc) — usually region or name format; see output above."
  fi
  printf '%s\n' "$out" >&2

  result_open ok
  result_kv namespace_name "$name"
  result_kv region "$region"
  result_close
}

# cmd_await_namespace: the JOIN. Poll `namespace list --name <name> -o json` until the
# async namespace is ACTIVE (state==3), bounded by NS_AWAIT_MAX_SECS. Reads the handle and
# endpoint from that authoritative, pagination-immune result. Gating on ACTIVE (not merely
# "visible") is the fix for the false-ready bug that left downstream steps connecting to a
# still-provisioning endpoint.
cmd_await_namespace() {
  local name="" max="${NS_AWAIT_MAX_SECS:-600}"
  while [ $# -gt 0 ]; do
    case "$1" in
      --name) name="$2"; shift 2 ;;
      --max-secs) max="$2"; shift 2 ;;
      *) die bad-args "unknown arg: $1" ;;
    esac
  done
  [ -n "$name" ] || die bad-args "--name is required (namespace_name from start-namespace)"
  temporal cloud whoami >/dev/null 2>&1 || die not-authenticated "Not signed in; run login again."

  log "Waiting for namespace '$name' to become ACTIVE (polling 'namespace list --name')..."
  local info
  local rc
  info="$(await_active_namespace "$name" "$max")"; rc=$?
  case "$rc" in
    0) ;;
    2) die namespace-not-provisioning "namespace '$name' never appeared after ~${NS_PHANTOM_GRACE_SECS:-75}s — the create was accepted but isn't provisioning (commonly an unavailable region, e.g. azure-centralus). Re-run start-namespace with an AWS/GCP region." ;;
    *) die namespace-timeout "namespace '$name' not ACTIVE after ~${max}s; a brand-new namespace can take a few minutes to provision. Re-run await-namespace to resume the wait. Do NOT switch endpoints or edit the profile." ;;
  esac

  result_open ok
  result_kv namespace_handle "${info%%|*}"
  result_kv address "${info#*|}"
  result_close
}

# cmd_create_namespace: blocking convenience wrapper = start + await in one call
# (preserves the simple synchronous op for tests / a non-parallel fallback). To
# overlap provisioning with clone+deps, call start-namespace / await-namespace
# directly instead.
cmd_create_namespace() {
  local sdk="" region=""
  while [ $# -gt 0 ]; do
    case "$1" in
      --sdk) sdk="$2"; shift 2 ;;
      --region) region="$2"; shift 2 ;;
      *) die bad-args "unknown arg: $1" ;;
    esac
  done
  [ -n "$sdk" ]    || die bad-args "--sdk is required"
  [ -n "$region" ] || die bad-args "--region is required"

  # Start (RESULT captured to extract the generated name; its logs flow to stderr),
  # then await prints the single final RESULT.
  local start_out name
  start_out="$(cmd_start_namespace --sdk "$sdk" --region "$region")"
  name="$(printf '%s\n' "$start_out" | sed -n 's/^namespace_name=//p')"
  if [ -z "$name" ]; then
    printf '%s\n' "$start_out"   # re-emit the start error RESULT
    exit 1
  fi
  cmd_await_namespace --name "$name"
}

# key_quota_stderr <file>  -> 0 (true) if the captured stderr reads like an API-key
# CAP/quota rejection. Broad on the quota words (limit/maximum/quota/exceeded/too
# many) but ANCHORED to key/apikey in proximity, so an unrelated 'limit' elsewhere
# in CLI chatter doesn't misfire. Case-insensitive; either word order.
key_quota_stderr() {
  grep -qiE '(api[ _-]?key|key)[^\n]*(limit|maximum|quota|exceeded|too many)|(limit|maximum|quota|exceeded|too many)[^\n]*(api[ _-]?key|key)' "$1" 2>/dev/null
}

# die_key_limit_reached  -> the single, actionable message for the API-key cap.
# Used by both create-key failure paths (non-zero exit and exit-0-with-error drift).
die_key_limit_reached() {
  die key-limit-reached "API-key limit reached for this account — the mint was rejected at the cap, not an auth or output problem. Delete stale keys, then re-run create-key: list them with 'temporal cloud apikey list' and remove old money-transfer-cloud-setup-* keys with 'temporal cloud apikey delete --key-id <id>'."
}

cmd_create_key() {
  # Unique display name per run: `apikey create-for-me` ERRORS when a key matching the spec
  # (same display name) already exists and --idempotent isn't set — and --idempotent would
  # "succeed" without returning a fresh token (it's write-once), which we need. A random
  # suffix guarantees a brand-new key every time, so a re-run (or re-mint) never conflicts.
  local handle="" address="" keyname="money-transfer-cloud-setup-$(rand_tag)"
  while [ $# -gt 0 ]; do
    case "$1" in
      --handle) handle="$2"; shift 2 ;;
      --address) address="$2"; shift 2 ;;
      --display-name) keyname="$2"; shift 2 ;;
      *) die bad-args "unknown arg: $1" ;;
    esac
  done
  [ -n "$handle" ]  || die bad-args "--handle is required (from create-namespace)"
  [ -n "$address" ] || die bad-args "--address is required (from create-namespace)"
  if ! require_cmd jq && ! require_cmd python3; then
    die no-json-parser "need jq or python3 to capture the API key safely; install one and retry."
  fi

  # Re-verify auth immediately before minting the key.
  temporal cloud whoami >/dev/null 2>&1 || die not-authenticated "Not signed in; run login again before the key step."

  # Capture the key WITHOUT printing it. The token goes only into 0600 temp files and
  # is read straight into a shell var. We capture stdout AND stderr to SEPARATE locked
  # files: this prerelease CLI has emitted the freshly-minted key to stderr / human text
  # rather than as JSON on stdout, so the old `2>/dev/null` discarded the token and the
  # result looked empty (`key-empty`). We parse JSON when it's there and fall back to a
  # JWT-pattern match over EITHER stream, so capture works regardless of output shape.
  umask 077
  local tmp tmperr
  tmp="$(mktemp "${TMPDIR:-/tmp}/tcloud-key.XXXXXX")"
  tmperr="$(mktemp "${TMPDIR:-/tmp}/tcloud-key.XXXXXX")"
  chmod 600 "$tmp" "$tmperr"
  # shellcheck disable=SC2064
  trap "rm -f '$tmp' '$tmperr'" EXIT

  log "Creating API key (token captured to a locked file — never printed)..."
  if ! temporal cloud apikey create-for-me \
        --display-name "$keyname" \
        --description "money-transfer Cloud setup" \
        --expiry-duration 25h \
        --auto-confirm \
        -o json > "$tmp" 2>"$tmperr"; then
    # Show a redacted tail for context — never the token.
    redact < "$tmperr" 2>/dev/null | tail -n 5 >&2
    # Account at the API-key cap? The mint is rejected at create time with quota
    # language; surface that as a distinct, actionable code rather than a generic fail.
    if key_quota_stderr "$tmperr"; then die_key_limit_reached; fi
    die key-create-failed "apikey create-for-me exited non-zero; re-check auth (whoami/login). See the redacted output above."
  fi
  if [ ! -s "$tmp" ] && [ ! -s "$tmperr" ]; then
    die key-empty "apikey create returned nothing on either stream — almost always an expired login; run whoami/login and retry once."
  fi

  local token keyid
  # 1. Preferred: structured JSON on stdout.
  token="$(json_field "$tmp" token)"
  [ -n "$token" ] || token="$(json_field "$tmp" secretKey)"
  keyid="$(json_field "$tmp" keyId)"
  [ -n "$keyid" ] || keyid="$(json_field "$tmp" id)"
  # 2. Fallback for output drift: the key was emitted to stderr or as human text. Pull
  #    the JWT by pattern from EITHER stream (API keys are JWTs: 'eyJ' + two more
  #    dot-separated base64url segments). Stays in a var — never printed, never argv.
  if [ -z "$token" ]; then
    token="$(grep -hoE 'eyJ[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+' "$tmp" "$tmperr" 2>/dev/null | head -n1)"
  fi
  if [ -z "$keyid" ]; then  # key_id is NOT secret; best-effort scrape if JSON wasn't present
    keyid="$(grep -hoiE 'key[ _-]?id["'"'"':= ]+[A-Za-z0-9_-]{16,}' "$tmp" "$tmperr" 2>/dev/null | head -n1 | grep -oE '[A-Za-z0-9_-]{16,}$')"
  fi
  if [ -z "$token" ]; then
    # (a) Automatic capture failed (no JSON, no JWT pattern). Fall back to a HIDDEN
    # paste straight from the controlling terminal — the token goes into a local var
    # and then into the locked file; it is never echoed, printed, or passed as argv.
    # If there is no interactive tty, bail with a code the agent maps to "ask the user
    # to paste, then re-run".
    if [ -r /dev/tty ]; then
      log "Automatic key capture failed. Paste the one-time API key (input hidden), then press Enter:"
      IFS= read -rs -t 120 token < /dev/tty || token=""   # -t guards against a hang when no input arrives
      printf '\n' >&2
    fi
  fi
  if [ -z "$token" ]; then
    # Exit-0-with-error drift: some prerelease builds print a cap/quota rejection to
    # stderr yet still exit 0, so there's no token to capture. Classify that here (we
    # only reach this with no token) as key-limit-reached rather than manual-key-needed.
    if key_quota_stderr "$tmperr"; then
      redact < "$tmperr" 2>/dev/null | tail -n 5 >&2
      die_key_limit_reached
    fi
    die manual-key-needed "could not capture the key from the CLI output (checked stdout+stderr, JSON + token pattern) and no terminal to paste into — have the user paste it, then re-run create-key from a context with a terminal."
  fi

  # Write the cloud-setup profile straight into the TOML (default untouched).
  write_profile "$handle" "$address" "$token"
  # token goes out of scope; wipe the temp files now (trap is the backstop).
  rm -f "$tmp" "$tmperr"; trap - EXIT

  local cfg; cfg="$(config_path)"
  result_open ok
  result_kv key_id "${keyid:-unknown}"          # KeyId is NOT secret — safe to show
  result_kv config_path "$cfg"
  result_kv profile cloud-setup
  result_close
}

# strip_cloud_setup <file>  -> stdout
# Emits <file> with EVERY [profile.cloud-setup] and [profile.cloud-setup.tls]
# section removed, preserving all other profiles (notably [profile.default] and
# its oauth/login session). Purely textual line-skipping, so it repairs a file
# that duplicate sections have already made TOML-invalid. Tolerates leading
# whitespace on headers; the (\.|\]) boundary keeps it from matching a different
# profile that merely shares the prefix (e.g. [profile.cloud-setup-staging]).
strip_cloud_setup() {
  awk '
    /^[[:space:]]*\[profile\.cloud-setup(\.|\])/ { skip=1; next }
    /^[[:space:]]*\[/ { if (skip) skip=0 }
    skip { next }
    { print }
  ' "$1"
}

# write_profile <handle> <address> <token>
# Replaces any existing [profile.cloud-setup(.tls)] block(s), preserves
# everything else (notably [profile.default] with its oauth/login session),
# chmod 600.
write_profile() {
  local handle="$1" address="$2" token="$3"
  local cfg dir tmp oldumask
  # Self-contained perms: don't depend on the caller's umask. The temp file holds the
  # live token, so it must be 0600 from the moment of creation; tighten the config dir too.
  oldumask="$(umask)"; umask 077
  cfg="$(config_path)"
  dir="$(dirname "$cfg")"
  mkdir -p "$dir"; chmod 700 "$dir" 2>/dev/null || true
  tmp="$(mktemp "${dir}/temporal.toml.XXXXXX")"
  chmod 600 "$tmp"
  # A failed mv (read-only target, full disk) must NOT strand a 0600 temp file holding the
  # token. RETURN trap is the backstop for any early return; the mv path rm's explicitly.
  # shellcheck disable=SC2064
  trap "rm -f '$tmp'" RETURN

  if [ -f "$cfg" ]; then
    strip_cloud_setup "$cfg" > "$tmp"   # drop existing/duplicate cloud-setup section(s)
  fi

  {
    printf '\n[profile.cloud-setup]\n'
    printf 'address = "%s"\n' "$address"
    printf 'namespace = "%s"\n' "$handle"
    printf 'api_key = "%s"\n' "$token"
    printf '\n[profile.cloud-setup.tls]\n'
    printf 'disabled = false\n'   # TLS is required for Temporal Cloud; set it explicitly (empty section left it ambiguous, causing connect failures)
  } >> "$tmp"

  if ! mv "$tmp" "$cfg"; then
    rm -f "$tmp"; umask "$oldumask"
    die config-write-failed "could not write the profile to $cfg — check directory permissions / free disk, then re-run create-key."
  fi
  chmod 600 "$cfg"
  umask "$oldumask"
  log "Wrote [profile.cloud-setup] to $cfg (chmod 600; api_key not shown)."
}

cmd_verify_config() {
  # `config list` lists profile names, not the key value — but scrub defensively
  # so that even a CLI version that DID echo the key can't leak it through here.
  local out
  if ! out="$(temporal --profile cloud-setup config list 2>&1)"; then
    die profile-missing "could not read the cloud-setup profile; was create-key run?"
  fi
  printf '%s\n' "$out" | redact >&2
  result_open ok; result_kv profile cloud-setup; result_close
}

# cmd_await_auth: deterministic readiness gate to run BEFORE the Worker. A
# freshly-minted API key (and the namespace's just-enabled API-key auth) is not
# accepted by the data plane immediately, so a Worker that connects too soon hits
# `Request unauthorized` and crashes. Poll the cheapest authorized data-plane call
# (`workflow list`, exit 0 == accepted) until it succeeds, bounded. Foreground,
# synchronous — no background, no parallelism.
# await_auth_permanent <file>  -> 0 (true) if the poll's stderr is a HIGH-CONFIDENCE
# PERMANENT key failure that will NEVER clear by waiting. Deliberately NARROW: it
# requires a permanent qualifier (expired / invalid / revoked / not found / disabled /
# malformed) ANCHORED to jwt/key/token/credential context — mirroring key_quota_stderr's
# proximity anchoring.
#
# Anchor set includes `jwt` because the Temporal Cloud API gateway (Envoy) returns the
# auth failure as a gRPC status whose desc is a JWT-filter message, NOT "api key" text.
# Real captures against the prerelease CLI (`workflow list`):
#   * no/empty key   -> `Unauthenticated desc = Jwt is missing`
#   * bad/wrong key  -> `Unauthenticated desc = Jwt issuer is not configured`
#   * expired key    -> `Unauthenticated desc = Jwt is expired`   (Envoy default;
#                       e.g. a ~25h key polled next-day)
# See references/unified-cli.md. Only `expired` is treated as unambiguously permanent
# here: a valid key that is merely PROPAGATING has a future exp, so it can never emit
# "Jwt is expired" — the safety bias holds. A bare transient `Jwt is missing` /
# `issuer is not configured` / `Request unauthorized` / `permission denied` (which can
# appear during propagation) must NOT match, so the poll keeps waiting instead of
# wrong-fast-failing. `expired` stays anchored so a TLS/certificate "expired" (no
# jwt/key/token context) never masquerades as a key failure.
await_auth_permanent() {
  grep -qiE '(jwt|api[ _-]?key|token|credential)[^\n]*(expired|invalid|revoked|not found|disabled|malformed)|(expired|invalid|revoked|malformed)[^\n]*(jwt|api[ _-]?key|token|credential)' "$1" 2>/dev/null
}

cmd_await_auth() {
  local max="${AUTH_READY_MAX_SECS:-90}" interval="${AUTH_POLL_INTERVAL_SECS:-5}" waited=0
  # Per-call timeout so a single wedged `workflow list` can't block the whole loop.
  # `waited` accrues real wall-clock (poll time + sleep), so a slow/hung endpoint
  # still honors ~max seconds of total budget rather than max*call_to.
  local call_to="${AUTH_POLL_CALL_TIMEOUT:-15}"
  log "Waiting for the API key to be accepted before starting the Worker (auth readiness)..."
  # Capture the poll's stderr (never stdout) to a locked temp file so a permanent auth
  # failure is diagnosable instead of collapsing into a generic timeout. EXIT trap so
  # the file is removed on success (return) AND on any die (exit).
  umask 077
  local tmperr; tmperr="$(mktemp "${TMPDIR:-/tmp}/tcloud-auth.XXXXXX")"; chmod 600 "$tmperr"
  # shellcheck disable=SC2064
  trap "rm -f '$tmperr'" EXIT
  local last_err="" rc t0 t1
  while :; do
    : > "$tmperr"
    # exit 0 == the key is accepted. The poll's own exit code still gates success;
    # stderr is only ever used to CLASSIFY a failure (and, redacted, to report it).
    t0="$(date +%s)"
    run_bounded "$call_to" temporal --profile cloud-setup workflow list --limit 1 >/dev/null 2>"$tmperr"
    rc=$?
    if [ "$rc" -eq 0 ]; then
      rm -f "$tmperr"; trap - EXIT
      result_open ok; result_kv auth_ready true; result_kv waited_secs "$waited"; result_close
      return 0
    fi
    # Keep the most recent NON-timeout CLI stderr as the diagnostic tail (a per-call
    # timeout produces no CLI message; don't let it erase the last real one). Redacted
    # at capture, so nothing secret is ever held or surfaced.
    if [ "$rc" -ne "$RUN_BOUNDED_TIMEOUT" ] && [ -s "$tmperr" ]; then
      last_err="$(redact < "$tmperr" | grep -v '^[[:space:]]*$' | tail -n 1)"
    fi
    # High-confidence PERMANENT key failure -> fast-fail; a dead/expired/revoked key
    # will NEVER clear by waiting, so don't spin the full bound. The classifier is
    # deliberately narrow (see await_auth_permanent): a bare transient `unauthorized`
    # /`permission denied` during propagation stays transient and keeps polling — a
    # wrong fast-fail is worse than a bounded wait. NOTE: if a prerelease exited 0 on
    # an auth *failure* the poll would false-pass above before we get here; that's a
    # connectivity-pulse concern, handled separately.
    if [ "$rc" -ne "$RUN_BOUNDED_TIMEOUT" ] && await_auth_permanent "$tmperr"; then
      die key-expired "API key rejected as expired/invalid — it will not clear by waiting (keys auto-expire in ~25h, so a next-day re-test hits a dead key). Re-run create-key to mint a fresh key (it overwrites the [profile.cloud-setup] block). If you are at the API-key cap, delete stale keys first (temporal cloud apikey list; temporal cloud apikey delete --key-id <id>), then re-run create-key.${last_err:+ (CLI stderr: $last_err)}"
    fi
    if [ "$waited" -ge "$max" ]; then
      die auth-timeout "API key still not accepted after ${max}s. A just-minted key/namespace can lag — wait and re-run await-auth; if it never clears, re-run create-key. Do NOT switch endpoints or edit the profile.${last_err:+ (last CLI stderr: $last_err)}"
    fi
    sleep "$interval"
    # Accrue real elapsed (poll + sleep) so the ~max budget is wall-clock, not
    # per-interval: a slow/hung endpoint would otherwise take max*call_to seconds.
    t1="$(date +%s)"
    waited=$(( waited + (t1 - t0) ))
  done
}

# worker_polling <task-queue>  -> exit 0 if >=1 Worker is polling the queue.
# API-based readiness: `task-queue describe -o json` reports a top-level `pollers`
# array (null/empty when no Worker has polled). This is a DATA-plane call (works on
# the prerelease, unlike the control-plane `cloud namespace get -o json`); it also
# returns non-zero while auth is not yet accepted, which the caller treats as "not
# polling yet" (the Worker would be failing for the same reason).
worker_polling() {
  local tq="$1" raw n
  raw="$(temporal --profile cloud-setup task-queue describe --task-queue "$tq" -o json 2>/dev/null)" || return 1
  [ -n "$raw" ] || return 1
  if require_cmd jq; then
    n="$(printf '%s' "$raw" | jq -r '(.pollers // []) | length' 2>/dev/null)"
  elif require_cmd python3; then
    n="$(printf '%s' "$raw" | python3 -c 'import json,sys
try:
    d=json.load(sys.stdin); p=d.get("pollers") or []
    print(len(p))
except Exception:
    print(0)')"
  else
    n=0
  fi
  [ "${n:-0}" -ge 1 ]
}

# latest_workflow_ids  -> echoes "<workflowId> <runId>" for the most recent execution
# in the cloud-setup namespace. `workflow list -o json` is an array whose [0] is the
# newest run; element shape is { "execution": { "workflowId", "runId" }, ... }. The
# namespace is dedicated to this setup, so --limit 1 is the run we just completed.
latest_workflow_ids() {
  local raw wid="" rid=""
  raw="$(temporal --profile cloud-setup workflow list --limit 1 -o json 2>/dev/null)" || return 1
  [ -n "$raw" ] || return 1
  if require_cmd jq; then
    wid="$(printf '%s' "$raw" | jq -r '.[0].execution.workflowId // empty' 2>/dev/null)"
    rid="$(printf '%s' "$raw" | jq -r '.[0].execution.runId // empty' 2>/dev/null)"
  elif require_cmd python3; then
    wid="$(printf '%s' "$raw" | python3 -c 'import json,sys
try:
    a=json.load(sys.stdin); e=(a[0] if a else {}).get("execution",{}) or {}
    print(e.get("workflowId",""))
except Exception:
    print("")')"
    rid="$(printf '%s' "$raw" | python3 -c 'import json,sys
try:
    a=json.load(sys.stdin); e=(a[0] if a else {}).get("execution",{}) or {}
    print(e.get("runId",""))
except Exception:
    print("")')"
  fi
  [ -n "$wid" ] && [ -n "$rid" ] || return 1
  printf '%s %s' "$wid" "$rid"
}

# workflow_status_for <workflowId> [runId]  -> echoes the execution status string
# (e.g. WORKFLOW_EXECUTION_STATUS_COMPLETED) for that Workflow's run, or EMPTY if
# it isn't found yet / the call is momentarily unreadable (treat empty as "not
# yet"). Pin the run with [runId] when known so a reused WorkflowId can never
# read a prior run's status. Always returns 0 so a flaky call doesn't abort.
workflow_status_for() {
  local wid="$1" rid="${2:-}" raw
  if [ -n "$rid" ]; then
    raw="$(temporal --profile cloud-setup workflow describe --workflow-id "$wid" --run-id "$rid" -o json 2>/dev/null)" || return 0
  else
    raw="$(temporal --profile cloud-setup workflow describe --workflow-id "$wid" -o json 2>/dev/null)" || return 0
  fi
  [ -n "$raw" ] || return 0
  if require_cmd jq; then
    printf '%s' "$raw" | jq -r '.workflowExecutionInfo.status // empty' 2>/dev/null
  elif require_cmd python3; then
    printf '%s' "$raw" | python3 -c 'import json,sys
try:
    print(json.load(sys.stdin).get("workflowExecutionInfo",{}).get("status","") or "")
except Exception:
    print("")'
  fi
}

# stop_group <pid>  -> terminate a backgrounded job AND its children. The job is
# launched under `set -m` so it leads its own process group (pgid == pid); a kill
# on the negative pid reaches the whole tree (go/mvn/npm spawn child processes that
# a bare `kill <pid>` would orphan, leaving Workers polling Cloud). Best-effort.
stop_group() {
  local p="${1:-}"
  [ -n "$p" ] || return 0
  kill -TERM "-$p" 2>/dev/null || kill -TERM "$p" 2>/dev/null || true
  sleep 1
  kill -KILL "-$p" 2>/dev/null || kill -KILL "$p" 2>/dev/null || true
  wait "$p" 2>/dev/null || true
  return 0
}

# cmd_run_workflow: THE single-call worker+starter path. The Worker is a
# long-running process that must stay alive WHILE the starter triggers a Workflow —
# but background jobs don't reliably survive across tool calls on Codex/Cursor. So,
# like provision-and-scaffold, this owns the whole dance inside ONE synchronous call:
# launch the Worker detached (own process group) -> wait until it is polling the task
# queue (API-based readiness, not ps/pgrep) -> run the starter in the foreground until
# the Workflow returns -> tear the Worker down -> emit one RESULT block. The Worker only
# needs to live within this call, so no cross-tool-call survival is required.
# Run await-auth BEFORE this so the just-minted key is already accepted.
cmd_run_workflow() {
  local sdk="" dir="" demo="" max="${RUN_WF_MAX_SECS:-180}" ready_max="${WORKER_READY_MAX_SECS:-120}"
  while [ $# -gt 0 ]; do
    case "$1" in
      --sdk) sdk="$2"; shift 2 ;;
      --dir) dir="$2"; shift 2 ;;
      --demo-failure) demo="$2"; shift 2 ;;
      --max-secs) max="$2"; shift 2 ;;
      *) die bad-args "unknown arg: $1" ;;
    esac
  done
  [ -n "$sdk" ] || die bad-args "--sdk is required"
  [ -n "$dir" ] || die bad-args "--dir is required (repo_path from provision-and-scaffold)"
  [ -d "$dir" ] || die bad-args "repo dir not found: $dir"
  local wcmd scmd tq
  wcmd="$(worker_cmd_for "$sdk")"; scmd="$(starter_cmd_for "$sdk")"; tq="$(task_queue_for "$sdk")"
  { [ -n "$wcmd" ] && [ -n "$scmd" ] && [ -n "$tq" ]; } || die unknown-sdk "no run-command mapping for SDK '$sdk'"
  # SAFETY INVARIANT: these are SDK-keyed CONSTANT command strings (eval'd below). User
  # input (dir/region/handle/name) is NEVER interpolated into them — it's applied via
  # `cd "$dir"` and `export WORKFLOW_ID=...`. Assert no shell expansion ever sneaks in.
  case "$wcmd$scmd" in *'$'*|*'`'*) die internal-bad-cmd "run-command table must not contain shell expansions" ;; esac
  [ "$demo" = "" ] || [ "$demo" = "transient" ] || die bad-args "--demo-failure only supports 'transient'"

  # Workflow ID: distinct, self-describing names so the clean run and the
  # failure-and-recovery run show up as two different Workflows in Cloud. The
  # sample starters read WORKFLOW_ID from the env (defaulting to the clean name),
  # so this never edits the sample source.
  local wf_name="money-transfer-demo"
  [ "$demo" = "transient" ] && wf_name="money-transfer-demo-recovery"

  # Pre-compile compiled SDKs so Worker and starter skip build time entirely,
  # keeping the timing windows (ready_max, max) bounded to Temporal operations only.
  case "$sdk" in
    java)
      log "  -> pre-compiling Java sample app (mvn compile)..."
      ( cd "$dir" || exit 127; mvn -q compile ) || die precompile-failed "Maven compile failed; check output above. Ensure Java + Maven are installed and the machine has network access for first-time dependency download."
      ;;
    dotnet|.net|net)
      log "  -> pre-compiling .NET sample app (dotnet build)..."
      ( cd "$dir" || exit 127; dotnet build -v q ) || die precompile-failed "dotnet build failed; check output above. Ensure the .NET SDK is installed."
      ;;
  esac

  local job wlog slog wpid="" sp="" src waited swaited
  job="$(mktemp -d "${TMPDIR:-/tmp}/tcloud-run.XXXXXX")"
  wlog="$job/worker.log"; slog="$job/starter.log"
  # Safety net: tear down the Worker + starter and the temp dir on ANY exit —
  # including an external SIGTERM (e.g. the agent's Bash-tool timeout firing before
  # our own --max-secs). Without this, a kill mid-poll orphans the Worker, leaving
  # it polling Cloud (a reported leak). SIGKILL can't be trapped, but a tool-timeout
  # sends TERM, and a normal/`die` exit runs the EXIT trap. stop_group tolerates an
  # empty pid, so this is safe before wpid/sp are assigned.
  # NOTE the `${var:-}` guards: at a NORMAL exit this function has already returned,
  # so its `local` sp/wpid/job are out of scope when the EXIT trap fires — a bare
  # `$sp` under `set -u` prints "sp: unbound variable" after the result block. The
  # `:-` makes those no-ops at normal exit, while a SIGTERM mid-run fires the trap
  # with the function still on the stack (locals in scope), so the Worker is killed.
  trap 'stop_group "${sp:-}" 2>/dev/null; stop_group "${wpid:-}" 2>/dev/null; [ -n "${job:-}" ] && rm -rf "$job" 2>/dev/null' EXIT INT TERM

  # 1. Launch the Worker detached, in its own process group (set -m), output -> wlog.
  if [ "$demo" = "transient" ]; then
    log "Starting the Worker with DEMO_FAILURE=transient (background) for '$sdk'..."
  else
    log "Starting the Worker (background) for '$sdk'..."
  fi
  set -m
  ( cd "$dir" || exit 127
    [ "$demo" = "transient" ] && export DEMO_FAILURE=transient
    eval "$wcmd" ) >"$wlog" 2>&1 &
  wpid=$!
  set +m

  # 2. Readiness: wait until the Worker registers as a poller (API), or it dies first.
  log "  -> waiting for the Worker to start polling task queue '$tq'..."
  waited=0
  while :; do
    if worker_polling "$tq"; then break; fi
    if ! kill -0 "$wpid" 2>/dev/null; then
      tail -n 20 "$wlog" >&2
      if grep -qiE 'unauthorized|permission denied' "$wlog" 2>/dev/null; then
        rm -rf "$job" 2>/dev/null
        die worker-unauthorized "Worker exited with an auth error before polling — the key may not be accepted yet. Run await-auth, then retry run-workflow."
      fi
      rm -rf "$job" 2>/dev/null
      die worker-start-failed "Worker exited before polling; see log tail above (e.g. missing deps/venv or wrong dir). Repo: $dir"
    fi
    if [ "$waited" -ge "$ready_max" ]; then
      # API didn't show a poller in time; accept a log marker as a fallback signal.
      if grep -qiE 'poll|worker started|started worker' "$wlog" 2>/dev/null; then
        log "  [warn] no poller via task-queue API within ${ready_max}s, but the Worker log shows polling — proceeding."
        break
      fi
      stop_group "$wpid"
      tail -n 20 "$wlog" >&2
      rm -rf "$job" 2>/dev/null
      die worker-not-polling "Worker did not register as a poller on '$tq' within ${ready_max}s; see log tail above. Worker log: $wlog"
    fi
    sleep 3; waited=$(( waited + 3 ))
  done
  log "  [ok] Worker is polling '$tq'."

  # 3. Run the starter. It submits the Workflow; some samples then block on the
  #    result, but others (e.g. Java) start it async and exit immediately. Launch
  #    it in its own group so a timeout kill is clean.
  log "  -> running the starter (Workflow runs against Cloud)..."
  set -m
  ( cd "$dir" || exit 127
    export WORKFLOW_ID="$wf_name"
    eval "$scmd" ) >"$slog" 2>&1 &
  sp=$!
  set +m

  # 3b. Wait for the WORKFLOW itself to reach COMPLETED on the server — NOT merely
  #     for the starter to exit. An async starter returns before the Workflow
  #     finishes, so gating on its exit code falsely reports success and tears the
  #     Worker down mid-run, orphaning the execution. Keep the Worker alive while
  #     we poll, bounded by --max-secs. Pin the RunId (parsed from the starter
  #     log) so a reused WorkflowId can never read a prior run's status.
  local wf_id="$wf_name" run_id="" wf_status=""
  swaited=0; src=""; local nowf_waited=0
  while :; do
    # Pin THIS run's RunId from the starter log as soon as it's printed (Java/Go/TS).
    if [ -z "$run_id" ]; then
      run_id="$(grep -ioE 'run[ _-]?id["'"'"':= ]+[A-Za-z0-9-]{8,}' "$slog" 2>/dev/null | head -n1 | grep -oE '[A-Za-z0-9-]{8,}$')"
    fi
    # Reap the starter when it exits. Once it's fully reaped its stdout is flushed,
    # so make a definitive RunId parse here before we'd fall back to a by-id query.
    if [ -z "$src" ] && ! kill -0 "$sp" 2>/dev/null; then
      wait "$sp"; src=$?
      [ -z "$run_id" ] && run_id="$(grep -ioE 'run[ _-]?id["'"'"':= ]+[A-Za-z0-9-]{8,}' "$slog" 2>/dev/null | head -n1 | grep -oE '[A-Za-z0-9-]{8,}$')"
    fi

    # Only read status once it can be attributed to THIS run — either the RunId is
    # pinned, or the starter has exited 0 (a sync SDK that waited for completion, so
    # the new run is now the latest). Querying by WorkflowId before then could read a
    # prior reused-WorkflowId run's COMPLETED status and bail before our run even ran.
    if [ -n "$run_id" ]; then
      wf_status="$(workflow_status_for "$wf_id" "$run_id")"
    elif [ -n "$src" ] && [ "$src" -eq 0 ]; then
      wf_status="$(workflow_status_for "$wf_id")"
    else
      wf_status=""
    fi
    case "$wf_status" in
      *COMPLETED) break ;;
      *FAILED|*TERMINATED|*CANCELED|*TIMED_OUT)
        stop_group "$sp" 2>/dev/null; stop_group "$wpid"
        tail -n 30 "$slog" >&2; rm -rf "$job" 2>/dev/null
        die workflow-failed "Workflow ended ${wf_status##*STATUS_} (not COMPLETED). See log tail above." ;;
    esac
    # Starter died non-zero and never produced a trackable Workflow -> a real failure.
    if [ -n "$src" ] && [ "$src" -ne 0 ] && [ -z "$run_id" ] && [ -z "$wf_status" ]; then
      stop_group "$wpid"
      tail -n 30 "$slog" >&2; rm -rf "$job" 2>/dev/null
      die workflow-failed "starter exited ${src} and no Workflow was submitted. See log tail above."
    fi
    # Fast-fail: a starter that exited 0 but produced NO trackable Workflow (no RunId
    # pinned, and a by-id lookup stays empty) never actually submitted one — e.g. a client
    # that catches its own connect/start error and still exits 0 (the .NET sample does
    # exactly this). Give the server a brief settle window for a flaky read, then fail
    # clearly instead of burning the full --max-secs as a misleading timeout.
    if [ -n "$src" ] && [ "$src" -eq 0 ] && [ -z "$run_id" ] && [ -z "$wf_status" ]; then
      nowf_waited=$(( nowf_waited + 2 ))
      if [ "$nowf_waited" -ge "${NOWF_SETTLE_SECS:-15}" ]; then
        stop_group "$wpid"
        tail -n 30 "$slog" >&2; rm -rf "$job" 2>/dev/null
        die workflow-not-submitted "starter exited 0 but no Workflow was submitted within ${NOWF_SETTLE_SECS:-15}s — it likely swallowed a connect/start error (some samples catch-and-exit-0). See log tail above."
      fi
    else
      nowf_waited=0
    fi
    if [ "$swaited" -ge "$max" ]; then
      stop_group "$sp" 2>/dev/null; stop_group "$wpid"
      # Only terminate if the execution is genuinely still RUNNING — never kill a
      # Workflow that actually completed but whose status was momentarily unreadable.
      if [ -n "$wf_status" ] && [ -z "${wf_status##*RUNNING}" ]; then
        log "  [cleanup] terminating orphaned workflow '$wf_id' (local timeout)..."
        if [ -n "$run_id" ]; then
          temporal --profile cloud-setup workflow terminate --workflow-id "$wf_id" --run-id "$run_id" --reason "run-workflow timed out after ${max}s" >/dev/null 2>&1 || true
        else
          temporal --profile cloud-setup workflow terminate --workflow-id "$wf_id" --reason "run-workflow timed out after ${max}s" >/dev/null 2>&1 || true
        fi
      fi
      tail -n 20 "$slog" >&2; rm -rf "$job" 2>/dev/null
      die workflow-timeout "Workflow did not reach COMPLETED within ${max}s; see log tail above. Worker log: $wlog"
    fi
    sleep 2; swaited=$(( swaited + 2 ))
  done

  # 4. Workflow reached COMPLETED on the server — now tear the Worker down.
  stop_group "$sp" 2>/dev/null
  stop_group "$wpid"
  log "  [ok] Worker stopped."

  # IDs for the run link. The Workflow ID is authoritative ($wf_name — set via
  # WORKFLOW_ID and honored by every sample), so report it directly. run_id was
  # pinned from the starter log during the completion wait above; if the sample
  # never printed it (a sync SDK that already completed), resolve it from the
  # data plane (retried briefly, only accepting the row that matches OUR
  # Workflow ID). May still be empty, in which case the caller links to the
  # Workflow detail page (never the bare namespace list).
  if [ -z "$run_id" ]; then
    local ids="" tries=0
    while [ "$tries" -lt 6 ]; do
      ids="$(latest_workflow_ids)" || ids=""
      if [ -n "$ids" ] && [ "${ids%% *}" = "$wf_id" ]; then run_id="${ids##* }"; break; fi
      tries=$(( tries + 1 )); sleep 2
    done
  fi

  result_open ok
  result_kv workflow_status COMPLETED
  result_kv workflow_id "$wf_id"
  result_kv run_id "${run_id:-unknown}"
  result_kv task_queue "$tq"
  result_kv worker_log "$wlog"
  [ "$demo" = "transient" ] && result_kv demo_failure transient
  result_close
}

cmd_clone() {
  local sdk="" dir=""
  while [ $# -gt 0 ]; do
    case "$1" in
      --sdk) sdk="$2"; shift 2 ;;
      --dir) dir="$2"; shift 2 ;;
      *) die bad-args "unknown arg: $1" ;;
    esac
  done
  [ -n "$sdk" ] || die bad-args "--sdk is required"
  local url; url="$(repo_url_for_sdk "$sdk")"
  [ -n "$url" ] || die unknown-sdk "no repo mapping for SDK '$sdk'"
  local target="${dir:-$(basename "$url")}"

  log "Cloning ${url} (branch money-transfer-project-cloud-setup) into ${target}..."
  if ! git clone --branch money-transfer-project-cloud-setup --single-branch "$url" "$target" >&2; then
    die clone-failed "git clone failed for $url"
  fi
  result_open ok
  result_kv repo_url "$url"
  result_kv path "$target"
  result_close
}

# deps_already_present <sdk> <dir>  -> exit 0 if deps look installed (skip them).
# Manager-independent: an env dir / node_modules / satisfied bundle means a re-run
# need not reinstall. Best-effort; called from inside the target dir.
deps_already_present() {
  case "$1" in
    python)        [ -d env ] ;;
    ts|typescript) [ -d node_modules ] ;;
    ruby)          bundle check >/dev/null 2>&1 ;;
    *)             return 1 ;;   # go/java/dotnet: install is cheap / handled at compile
  esac
}

# install_deps_for <sdk> <manager> <dir>  — manager-parameterized dependency
# install, detect-then-skip, backed by the (sdk,manager) matrix in install_cmd_for.
# All output goes to stderr (caller redirects) so it never pollutes the RESULT
# block. Best-effort: a deps hiccup warns but does not abort provisioning.
install_deps_for() {
  local sdk="$1" mgr="$2" dir="$3" cmd
  cmd="$(install_cmd_for "$sdk" "$mgr")"
  if [ -z "$cmd" ]; then
    log "  [warn] no install command for sdk '$sdk' + manager '$mgr' — skipping deps"
    return 0
  fi
  # SAFETY INVARIANT: install_cmd_for returns SDK/manager-keyed CONSTANT strings (eval'd
  # below). No user input is interpolated; assert no shell expansion can sneak in.
  case "$cmd" in *'$'*|*'`'*) die internal-bad-cmd "install command table must not contain shell expansions" ;; esac
  ( cd "$dir" 2>/dev/null || { log "  [warn] could not enter $dir for deps"; exit 0; }
    if deps_already_present "$sdk"; then
      log "  [skip] $sdk dependencies already present"
    else
      log "  -> installing deps via '$mgr': $cmd"
      eval "$cmd"
    fi
  ) || log "  [warn] dependency install reported a problem (continuing; can be retried)"
}

# cmd_install_deps: standalone manager-parameterized install. Lets the
# agent install (or re-install) deps with an explicitly chosen/overridden manager.
# Defaults --manager to the SDK's deterministic default when omitted. Dies with
# manager-not-found if the chosen manager's binary isn't on PATH (a clear,
# fail-fast error rather than a confusing downstream Worker crash).
cmd_install_deps() {
  local sdk="" mgr="" dir=""
  while [ $# -gt 0 ]; do
    case "$1" in
      --sdk) sdk="$2"; shift 2 ;;
      --manager) mgr="$2"; shift 2 ;;
      --dir) dir="$2"; shift 2 ;;
      *) die bad-args "unknown arg: $1" ;;
    esac
  done
  [ -n "$sdk" ] || die bad-args "--sdk is required"
  [ -n "$dir" ] || die bad-args "--dir is required (repo_path)"
  [ -d "$dir" ] || die bad-args "repo dir not found: $dir"
  local candidates bin
  resolve_pkg_inputs "$sdk" "$mgr" || die_pkg $? "$sdk"
  mgr="$PKG_MGR"; candidates="$PKG_CANDIDATES"   # mgr defaulted to the preferred candidate
  bin="$(manager_bin "$mgr")"
  require_cmd "$bin" || die manager-not-found "package manager '$mgr' (needs '$bin') is not installed; install it or pick another (supported: $(printf '%s' "$candidates" | tr ' ' ','))."

  install_deps_for "$sdk" "$mgr" "$dir"
  result_open ok
  result_kv sdk "$sdk"
  result_kv manager "$mgr"
  result_kv repo_path "$dir"
  result_close
}

# cmd_scaffold: set up the app only — clone the cloud-ready sample + install deps,
# NO namespace work. Pairs with start-namespace/await-namespace so the app
# setup is its own gated step that overlaps the server-side namespace provisioning.
# Validates the manager fail-fast (manager-not-found / unsupported-manager) BEFORE
# cloning, same as provision-and-scaffold.
cmd_scaffold() {
  local sdk="" dir="" mgr=""
  while [ $# -gt 0 ]; do
    case "$1" in
      --sdk) sdk="$2"; shift 2 ;;
      --dir) dir="$2"; shift 2 ;;
      --manager) mgr="$2"; shift 2 ;;
      *) die bad-args "unknown arg: $1" ;;
    esac
  done
  [ -n "$sdk" ] || die bad-args "--sdk is required"
  local url candidates target bin
  url="$(repo_url_for_sdk "$sdk")"
  [ -n "$url" ] || die unknown-sdk "no repo mapping for SDK '$sdk'"
  resolve_pkg_inputs "$sdk" "$mgr" || die_pkg $? "$sdk"
  mgr="$PKG_MGR"; candidates="$PKG_CANDIDATES"
  bin="$(manager_bin "$mgr")"
  require_cmd "$bin" || die manager-not-found "package manager '$mgr' (needs '$bin') is not installed; install it or pick another (supported: $(printf '%s' "$candidates" | tr ' ' ','))."
  target="${dir:-$(basename "$url")}"

  log "Cloning the sample for '$sdk' into '$target'..."
  if ! git clone --branch money-transfer-project-cloud-setup --single-branch "$url" "$target" >&2; then
    die clone-failed "git clone failed for $url"
  fi
  log "  [ok] cloned"
  log "  -> installing dependencies ($sdk via $mgr)..."
  install_deps_for "$sdk" "$mgr" "$target" >&2
  log "  [ok] dependencies ready"

  result_open ok
  result_kv repo_path "$target"
  result_kv manager "$mgr"
  result_close
}

# cmd_provision_and_scaffold: THE single-call parallel path. Runs as one
# ordinary SYNCHRONOUS foreground command; internally backgrounds the (synchronous,
# never --async) namespace create and overlaps it with clone + deps, then `wait`s
# and joins. Because the background job lives and dies inside this one invocation,
# no cross-tool-call survival is needed -> works on Claude Code, Codex, and Cursor.
# Emits append-only progress lines to stderr; one RESULT block on stdout.
cmd_provision_and_scaffold() {
  local sdk="" region="" dir="" mgr=""
  while [ $# -gt 0 ]; do
    case "$1" in
      --sdk) sdk="$2"; shift 2 ;;
      --region) region="$2"; shift 2 ;;
      --dir) dir="$2"; shift 2 ;;
      --manager) mgr="$2"; shift 2 ;;
      *) die bad-args "unknown arg: $1" ;;
    esac
  done
  [ -n "$sdk" ]    || die bad-args "--sdk is required"
  [ -n "$region" ] || die bad-args "--region is required"
  temporal cloud whoami >/dev/null 2>&1 || die not-authenticated "Not signed in; run login again."

  local url name target nsout nspid rc out handle candidates bin
  url="$(repo_url_for_sdk "$sdk")"
  [ -n "$url" ] || die unknown-sdk "no repo mapping for SDK '$sdk'"
  # Resolve the manager up front (default = preferred candidate) and fail fast if
  # an explicitly chosen one is unsupported/uninstalled, BEFORE we create the
  # namespace + clone (so a bad --manager never leaves half-provisioned state).
  resolve_pkg_inputs "$sdk" "$mgr" || die_pkg $? "$sdk"
  mgr="$PKG_MGR"; candidates="$PKG_CANDIDATES"
  bin="$(manager_bin "$mgr")"
  require_cmd "$bin" || die manager-not-found "package manager '$mgr' (needs '$bin') is not installed; install it or pick another (supported: $(printf '%s' "$candidates" | tr ' ' ','))."
  name="$(namespace_name_for "$sdk")"
  target="${dir:-$(basename "$url")}"

  log "Provisioning in parallel: namespace '$name' (background) + sample app for '$sdk'..."

  # Background the SYNCHRONOUS create (never --async); capture its output to a file.
  nsout="$(mktemp "${TMPDIR:-/tmp}/tcloud-ns.XXXXXX")"
  ( temporal cloud namespace create \
      --name "$name" --region "$region" \
      --api-key-auth-enabled --retention-days 30 --auto-confirm ) >"$nsout" 2>&1 &
  nspid=$!
  log "  -> namespace create started in the background (provisioning takes a few minutes)"

  # Foreground overlap: clone + deps. On clone failure, stop the namespace job too.
  log "  -> cloning sample into '$target'..."
  if ! git clone --branch money-transfer-project-cloud-setup --single-branch "$url" "$target" >&2; then
    kill "$nspid" 2>/dev/null; wait "$nspid" 2>/dev/null; rm -f "$nsout"
    die clone-failed "git clone failed for $url"
  fi
  log "  [ok] cloned"
  log "  -> installing dependencies ($sdk via $mgr)..."
  install_deps_for "$sdk" "$mgr" "$target" >&2
  log "  [ok] dependencies ready"

  # Join: wait for the namespace create to finish, then resolve the handle.
  log "  -> joining: waiting for the namespace to finish provisioning..."
  wait "$nspid"; rc=$?
  out="$(cat "$nsout")"; rm -f "$nsout"
  if [ "$rc" -ne 0 ]; then
    printf '%s\n' "$out" | tail -n 20 >&2
    die create-rejected "namespace create exited $rc — usually region or name format; see output above."
  fi
  # Resolve from the authoritative, pagination-immune `--name` filter and wait until the
  # namespace is ACTIVE (a just-submitted create is ACTIVATING for a few minutes). The
  # filtered result carries the real endpoint, so read the address from it.
  local info
  info="$(await_active_namespace "$name" "${NS_AWAIT_MAX_SECS:-600}")"; rc=$?
  case "$rc" in
    0) ;;
    2) die namespace-not-provisioning "namespace '$name' never appeared after ~${NS_PHANTOM_GRACE_SECS:-75}s — the create was accepted but isn't provisioning (commonly an unavailable region, e.g. azure-centralus). Re-run with an AWS/GCP region." ;;
    *) die handle-not-found "namespace '$name' created but not ACTIVE after retries (still provisioning?). Re-run provision-and-scaffold or await-namespace to resume the wait." ;;
  esac
  handle="${info%%|*}"
  log "  [ok] namespace active: $handle"

  result_open ok
  result_kv namespace_handle "$handle"
  result_kv address "${info#*|}"
  result_kv repo_path "$target"
  result_kv manager "$mgr"
  result_close
}

# cleanup PRINTS the destructive commands; it never runs them.
cmd_cleanup_info() {
  local handle="" keyid=""
  while [ $# -gt 0 ]; do
    case "$1" in
      --handle) handle="$2"; shift 2 ;;
      --key-id) keyid="$2"; shift 2 ;;
      *) die bad-args "unknown arg: $1" ;;
    esac
  done
  local cfg; cfg="$(config_path)"
  result_open ok
  result_kv config_path "$cfg"
  result_kv revoke_key_cmd "temporal cloud apikey delete --key-id ${keyid:-<key-id>}"
  result_kv delete_namespace_cmd "temporal cloud namespace delete --namespace ${handle:-<handle>}"
  result_kv note "Run these yourself; never --auto-confirm a delete. Then remove the [profile.cloud-setup] block from the config file."
  result_close
}

# repair-config: strip ALL [profile.cloud-setup] block(s) from the shared TOML,
# preserving [profile.default]. Use when prior/partial runs left duplicate
# cloud-setup sections that made the file unparseable. Never reads the file into
# the agent's context (the work is in awk); never touches other profiles. After
# repair, re-run create-key to write one fresh, valid profile.
cmd_repair_config() {
  local cfg; cfg="$(config_path)"
  if [ ! -f "$cfg" ]; then
    result_open ok; result_kv config_path "$cfg"; result_kv removed_blocks 0
    result_kv note "no config file present — nothing to repair"; result_close; return
  fi
  local before dir tmp
  before="$(grep -cE '^[[:space:]]*\[profile\.cloud-setup\]' "$cfg" 2>/dev/null || true)"
  dir="$(dirname "$cfg")"
  tmp="$(mktemp "${dir}/temporal.toml.XXXXXX")"
  chmod 600 "$tmp"
  strip_cloud_setup "$cfg" > "$tmp"
  mv "$tmp" "$cfg"
  chmod 600 "$cfg"
  log "Stripped cloud-setup profile(s) from $cfg (was ${before:-0}); [profile.default] preserved. Re-run create-key to write a fresh one."
  result_open ok
  result_kv config_path "$cfg"
  result_kv removed_blocks "${before:-0}"
  result_kv note "cloud-setup stripped; default preserved. Run create-key to recreate the profile."
  result_close
}

# emit_gate <heading> [<comment> <command>]...  (determinism lever)
# Prints a READY-TO-RENDER gate block on stdout, AFTER the RESULT block, delimited
# by `=== GATE ===` / `=== END GATE ===`. The agent renders the text between the
# markers VERBATIM (then appends the numbered choices) instead of assembling the
# fenced block from prose rules — which is error-prone to hand-assemble
# (dropped fence, comment-only, glued rules). ASCII only; comment goes ABOVE its
# command (green-comment style). Same idea as pinning the CLI flags: move the exact
# format into the script so the model only has to print it.
emit_gate() {
  # Plain bold heading, then a fenced ```bash``` block: a "# " comment above each
  # command. Keep the fence — without it a leading "#" renders as a markdown
  # heading. The agent renders this verbatim, so it can't be mangled by hand.
  local heading="$1"; shift
  printf '=== GATE ===\n'
  printf '**%s**\n\n' "$heading"
  printf '```bash\n'
  local first=1
  while [ $# -ge 2 ]; do
    [ "$first" -eq 1 ] || printf '\n'
    first=0
    printf '# %s\n%s\n' "$1" "$2"
    shift 2
  done
  printf '```\n'
  printf '=== END GATE ===\n'
}

# cmd_preview: dry-run. Resolve and PRINT the concrete command(s) a
# subcommand would run, plus the resolved user-facing parameters, with NO side
# effects (no clone, no temporal calls, no installs) — this powers the per-command
# confirm gate and the "Edit" path in SKILL.md. Deterministic and offline: the
# namespace NAME is shown as its template (it is randomized at real run time), so
# preview never needs the network or auth. Emits cmd_1/cmd_2/... in the RESULT AND
# a ready-to-render `=== GATE ===` block (see emit_gate) the agent prints verbatim.
cmd_preview() {
  local sub="${1:-}"; shift || true
  [ -n "$sub" ] || die bad-args "preview needs a subcommand, e.g. preview run-workflow --sdk python --dir D"
  local sdk="" region="" dir="" mgr="" handle="" address="" demo="" maxsecs=""
  while [ $# -gt 0 ]; do
    case "$1" in
      --sdk) sdk="$2"; shift 2 ;;
      --region) region="$2"; shift 2 ;;
      --dir) dir="$2"; shift 2 ;;
      --manager) mgr="$2"; shift 2 ;;
      --handle) handle="$2"; shift 2 ;;
      --address) address="$2"; shift 2 ;;
      --demo-failure) demo="$2"; shift 2 ;;
      --max-secs) maxsecs="$2"; shift 2 ;;
      *) die bad-args "unknown arg: $1" ;;
    esac
  done

  local url target candidates wcmd scmd tq wf_name icmd
  case "$sub" in
    clone)
      [ -n "$sdk" ] || die bad-args "preview clone needs --sdk"
      url="$(repo_url_for_sdk "$sdk")"; [ -n "$url" ] || die unknown-sdk "no repo mapping for SDK '$sdk'"
      target="${dir:-$(basename "$url")}"
      result_open ok
      result_kv preview clone
      result_kv repo_url "$url"
      result_kv clone_dir "$target"
      result_kv cmd_1 "git clone --branch money-transfer-project-cloud-setup --single-branch $url $target"
      result_close
      emit_gate "Downloading the sample app (clone only)" \
        "clone the Cloud-ready sample" "git clone --branch money-transfer-project-cloud-setup --single-branch $url $target" ;;
    install-deps)
      [ -n "$sdk" ] || die bad-args "preview install-deps needs --sdk"
      resolve_pkg_inputs "$sdk" "$mgr" || die_pkg $? "$sdk"
      mgr="$PKG_MGR"; icmd="$PKG_ICMD"; candidates="$PKG_CANDIDATES"
      result_open ok
      result_kv preview install-deps
      result_kv manager "$mgr"
      result_kv supported_managers "$(printf '%s' "$candidates" | tr ' ' ',')"
      result_kv clone_dir "${dir:-<repo_path>}"
      result_kv cmd_1 "(cd ${dir:-<repo_path>} && $icmd)"
      result_close
      emit_gate "Installing the sample's dependencies ($mgr)" \
        "install dependencies with $mgr $(install_location_note "$sdk")" "(cd ${dir:-<repo_path>} && $icmd)" ;;
    provision-and-scaffold)
      [ -n "$sdk" ] || die bad-args "preview provision-and-scaffold needs --sdk"
      url="$(repo_url_for_sdk "$sdk")"; [ -n "$url" ] || die unknown-sdk "no repo mapping for SDK '$sdk'"
      resolve_pkg_inputs "$sdk" "$mgr" || die_pkg $? "$sdk"
      mgr="$PKG_MGR"; icmd="$PKG_ICMD"
      target="${dir:-$(basename "$url")}"
      result_open ok
      result_kv preview provision-and-scaffold
      result_kv region "${region:-<region>}"
      result_kv manager "$mgr"
      result_kv clone_dir "$target"
      result_kv namespace_name_template "quickstartai-<user>-<random> (randomized at run; <=39 chars)"
      result_kv cmd_1 "temporal cloud namespace create --name <name> --region ${region:-<region>} --api-key-auth-enabled --retention-days 30 --auto-confirm"
      result_kv cmd_2 "git clone --branch money-transfer-project-cloud-setup --single-branch $url $target"
      result_kv cmd_3 "(cd $target && $icmd)"
      result_close
      emit_gate "Creating your namespace & downloading the sample app" \
        "create your Cloud namespace - billable; provisions server-side (~a few min)" "temporal cloud namespace create --name <name> --region ${region:-<region>} --api-key-auth-enabled --retention-days 30 --auto-confirm" \
        "clone the Cloud-ready sample" "git clone --branch money-transfer-project-cloud-setup --single-branch $url $target" \
        "install dependencies with $mgr $(install_location_note "$sdk")" "(cd $target && $icmd)" ;;
    run-workflow)
      [ -n "$sdk" ] || die bad-args "preview run-workflow needs --sdk"
      wcmd="$(worker_cmd_for "$sdk")"; scmd="$(starter_cmd_for "$sdk")"; tq="$(task_queue_for "$sdk")"
      { [ -n "$wcmd" ] && [ -n "$scmd" ] && [ -n "$tq" ]; } || die unknown-sdk "no run-command mapping for SDK '$sdk'"
      wf_name="money-transfer-demo"; [ "$demo" = "transient" ] && wf_name="money-transfer-demo-recovery"
      result_open ok
      result_kv preview run-workflow
      result_kv repo_dir "${dir:-<repo_path>}"
      result_kv workflow_id "$wf_name"
      result_kv task_queue "$tq"
      result_kv max_secs "${maxsecs:-180}"
      [ "$demo" = "transient" ] && result_kv demo_failure transient
      result_kv cmd_1 "(cd ${dir:-<repo_path>} && ${demo:+DEMO_FAILURE=transient }$wcmd)   # Worker (background)"
      result_kv cmd_2 "(cd ${dir:-<repo_path>} && WORKFLOW_ID=$wf_name $scmd)   # starter"
      result_close
      local rw_head="Run your first Workflow"
      [ "$demo" = "transient" ] && rw_head="Run the recovery Workflow (inject a failure)"
      emit_gate "$rw_head" \
        "Worker - runs in the background, polls the task queue, stopped when done" "(cd ${dir:-<repo_path>} && ${demo:+DEMO_FAILURE=transient }$wcmd)" \
        "starter - submits the Workflow, waits for COMPLETED, exits" "(cd ${dir:-<repo_path>} && WORKFLOW_ID=$wf_name $scmd)" ;;
    create-namespace)
      result_open ok
      result_kv preview create-namespace
      result_kv region "${region:-<region>}"
      result_kv namespace_name_template "quickstartai-<user>-<random> (randomized at run; <=39 chars)"
      result_kv cmd_1 "temporal cloud namespace create --name <name> --region ${region:-<region>} --api-key-auth-enabled --retention-days 30 --auto-confirm"
      result_close
      emit_gate "Creating your Cloud namespace" \
        "create your Cloud namespace - billable; provisions server-side (~a few min)" "temporal cloud namespace create --name <name> --region ${region:-<region>} --api-key-auth-enabled --retention-days 30 --auto-confirm" ;;
    start-namespace)
      result_open ok
      result_kv preview start-namespace
      result_kv region "${region:-<region>}"
      result_kv namespace_name_template "quickstartai-<user>-<random> (randomized at run; <=39 chars)"
      result_kv cmd_1 "temporal cloud namespace create --name <name> --region ${region:-<region>} --api-key-auth-enabled --retention-days 30 --auto-confirm --async   # submits, returns immediately; provisions server-side"
      result_close
      emit_gate "Creating your Cloud namespace" \
        "create your Cloud namespace - billable; submits async, provisions server-side (~a few min)" \
        "temporal cloud namespace create --name <name> --region ${region:-<region>} --api-key-auth-enabled --retention-days 30 --auto-confirm --async" ;;
    await-namespace)
      result_open ok
      result_kv preview await-namespace
      result_kv cmd_1 "temporal cloud namespace list --name <namespace-name> -o json   # polled until the namespace is ACTIVE (bounded)"
      result_close
      emit_gate "Waiting for the namespace to provision" \
        "poll until the namespace exists and is ACTIVE (bounded)" \
        "temporal cloud namespace list --name <namespace-name> -o json" ;;
    scaffold)
      [ -n "$sdk" ] || die bad-args "preview scaffold needs --sdk"
      url="$(repo_url_for_sdk "$sdk")"; [ -n "$url" ] || die unknown-sdk "no repo mapping for SDK '$sdk'"
      resolve_pkg_inputs "$sdk" "$mgr" || die_pkg $? "$sdk"
      mgr="$PKG_MGR"; icmd="$PKG_ICMD"
      target="${dir:-$(basename "$url")}"
      result_open ok
      result_kv preview scaffold
      result_kv manager "$mgr"
      result_kv clone_dir "$target"
      result_kv cmd_1 "git clone --branch money-transfer-project-cloud-setup --single-branch $url $target"
      result_kv cmd_2 "(cd $target && $icmd)"
      result_close
      emit_gate "Downloading the sample app (clone + dependencies)" \
        "clone the Cloud-ready sample" "git clone --branch money-transfer-project-cloud-setup --single-branch $url $target" \
        "install dependencies with $mgr $(install_location_note "$sdk")" "(cd $target && $icmd)" ;;
    create-key)
      # Never show the token; the command itself is fine (it carries no secret).
      result_open ok
      result_kv preview create-key
      result_kv handle "${handle:-<namespace-handle>}"
      result_kv address "${address:-<address>}"
      result_kv cmd_1 "temporal cloud apikey create-for-me --display-name money-transfer-cloud-setup --expiry-duration 25h --auto-confirm -o json   # token captured to a 0600 file, never printed"
      result_close
      emit_gate "Creating your API key and saving the config" \
        "mint the key + write the cloud-setup profile to temporal.toml (token captured to a 0600 file, never printed)" \
        "temporal cloud apikey create-for-me --display-name money-transfer-cloud-setup --expiry-duration 25h --auto-confirm -o json" ;;
    install-cli)
      # Presence-aware (read-only): disclose what cmd_install_cli will do. BETA
      # stopgap (PE-79): a present CLI is updated to the latest prerelease (not
      # skipped) while the CLI has no real versions; an absent CLI is installed.
      if cloud_cli_present; then
        local uc_cmd uc_note
        case "$(uname -s)" in
          Darwin) uc_cmd="brew upgrade temporalio/prerelease/temporal-cloud"; uc_note="update the Temporal CLI to the latest prerelease via Homebrew (adds/updates software)" ;;
          *)      uc_cmd="# already installed; update temporal-cloud manually from https://github.com/temporalio/cloud-cli/releases/latest"; uc_note="the Temporal CLI is already installed; on this OS, update temporal-cloud manually from the releases page" ;;
        esac
        result_open ok
        result_kv preview install-cli
        result_kv cli_present true
        result_kv cmd_1 "$uc_cmd"
        result_close
        emit_gate "Updating the Temporal CLI to the latest" "$uc_note" "$uc_cmd"
      else
        local ic_cmd ic_note
        case "$(uname -s)" in
          Darwin) ic_cmd="brew install temporalio/prerelease/temporal-cloud"; ic_note="install the Temporal CLI via Homebrew (adds software)" ;;
          *)      ic_cmd="# download temporal-cloud from https://github.com/temporalio/cloud-cli/releases/latest and put it on PATH"; ic_note="install the Temporal CLI (download the binary, put it on PATH)" ;;
        esac
        result_open ok
        result_kv preview install-cli
        result_kv cli_present false
        result_kv cmd_1 "$ic_cmd"
        result_close
        emit_gate "Installing the Temporal CLI" "$ic_note" "$ic_cmd"
      fi ;;
    login)
      result_open ok
      result_kv preview login
      result_kv cmd_1 "temporal cloud login   # opens a browser on your machine; blocks until you finish"
      result_kv cmd_2 "temporal cloud whoami   # confirms the signed-in identity"
      result_close
      emit_gate "Sign in to Temporal Cloud" \
        "open a browser to sign in (blocks until you finish)" "temporal cloud login" \
        "confirm the signed-in identity" "temporal cloud whoami" ;;
    regions)
      result_open ok
      result_kv preview regions
      result_kv cmd_1 "temporal cloud region list"
      result_close
      emit_gate "Listing your Cloud regions" \
        "list the regions your account can use" \
        "temporal cloud region list" ;;
    await-auth)
      result_open ok
      result_kv preview await-auth
      result_kv cmd_1 "temporal --profile cloud-setup workflow list --limit 1   # polled until the new API key is accepted (bounded ~90s)"
      result_close
      emit_gate "Waiting for the API key to be accepted" \
        "poll an authorized call until the new key is accepted (bounded ~90s)" \
        "temporal --profile cloud-setup workflow list --limit 1" ;;
    verify-config)
      result_open ok
      result_kv preview verify-config
      result_kv cmd_1 "temporal --profile cloud-setup config list   # lists profile fields; api_key is redacted, never shown"
      result_close
      emit_gate "Verifying the saved config" \
        "list the cloud-setup profile fields (api_key redacted, never shown)" \
        "temporal --profile cloud-setup config list" ;;
    preflight)
      result_open ok
      result_kv preview preflight
      result_close
      emit_gate "Checking your environment" \
        "check git / jq / brew are available (read-only, local)" "command -v git jq brew" \
        "flag any stray TEMPORAL_* env vars that would override your saved config" "env | grep '^TEMPORAL_' || true" ;;
    detect-tools)
      [ -n "$sdk" ] || die bad-args "preview detect-tools needs --sdk"
      candidates="$(managers_for "$sdk")"; [ -n "$candidates" ] || die unknown-sdk "no package-manager mapping for SDK '$sdk'"
      local dt_bins="" dt_m dt_b
      for dt_m in $candidates; do dt_b="$(manager_bin "$dt_m")"; dt_bins="${dt_bins:+$dt_bins }$dt_b"; done
      result_open ok
      result_kv preview detect-tools
      result_close
      emit_gate "Detecting your local tools" \
        "detect which package managers are installed for $sdk (read-only, local)" "command -v $dt_bins" \
        "read each tool's version to flag anything below the minimum" "$(runtime_for "$sdk") --version" ;;
    repair-config)
      result_open ok
      result_kv preview repair-config
      result_kv cmd_1 "strip [profile.cloud-setup] block(s) from temporal.toml (keeps [profile.default])"
      result_close
      emit_gate "Repairing temporal.toml (strip duplicate cloud-setup profiles)" \
        "remove any existing [profile.cloud-setup] block(s); keeps [profile.default]" "scripts/provision.sh repair-config" ;;
    *)
      die bad-args "preview does not support subcommand '$sub' (try: preflight, detect-tools, install-cli, login, regions, start-namespace, scaffold, await-namespace, provision-and-scaffold, install-deps, clone, create-namespace, create-key, await-auth, run-workflow, verify-config, repair-config)" ;;
  esac
}

# announce_gate <subcommand> [args...]   (deterministic disclosure floor)
# Before an effectful subcommand acts, print ITS gate to STDERR so the tool block
# ALWAYS records what the command runs — even when the agent skips rendering the
# chat-side gate from §Gate templates. Reuses cmd_preview's single-source gate
# text, so the disclosure can never drift from the real command; strips the
# machine markers and frames it as
# plain human disclosure. Never blocks the real work: any preview failure (an arg
# preview doesn't know, an unsupported sub) is swallowed and the command proceeds.
# Opt out with TCLOUD_DISCLOSE=0 (the test harness sets this to keep stderr clean).
# It is a RECORD, not a pre-consent prompt — in the request/response tool model it
# surfaces bundled with the result; the chat-side gate remains the pre-action one.
announce_gate() {
  [ "${TCLOUD_DISCLOSE:-1}" = "0" ] && return 0
  local sub="$1"; shift
  # Forward only the flags cmd_preview understands; drop the rest (--name,
  # --display-name, …) so disclosure never dies on an arg the real command accepts
  # but preview doesn't. (--display-name carries no secret; it's dropped for parity
  # with the preview create-key gate, which already shows the command tokenless.)
  local pv=()
  while [ $# -gt 0 ]; do
    case "$1" in
      --sdk|--region|--dir|--manager|--handle|--address|--demo-failure|--max-secs)
        if [ $# -ge 2 ]; then pv+=("$1" "$2"); shift 2; else shift; fi ;;
      *) shift ;;
    esac
  done
  local block
  # cmd_preview is offline + side-effect-free; a `die` inside it exits only this
  # command-substitution subshell, so `|| return 0` keeps the real work going.
  block="$(cmd_preview "$sub" ${pv[@]+"${pv[@]}"} 2>/dev/null)" || return 0
  # Keep only the lines between the GATE markers (drop the markers + the RESULT block).
  block="$(printf '%s\n' "$block" | sed -n '/^=== GATE ===$/,/^=== END GATE ===$/{/^=== GATE ===$/d;/^=== END GATE ===$/d;p;}')"
  [ -n "$block" ] || return 0
  {
    printf 'disclosure — the command(s) this step runs (auto-recorded):\n'
    printf '%s\n' "$block"
  } >&2
}

usage() {
  cat >&2 <<'EOF'
provision.sh — deterministic executor for temporal-cloud-setup

Usage: provision.sh <command> [args]

Commands:
  preflight [--sdk S]                       Environment checks + config path + stray TEMPORAL_* vars
  detect-tools --sdk S                      Detect supported+installed package managers, pick a default,
                                            report versions, surface discrepancies
  preview <subcommand> [args]               Dry-run: print the concrete command(s) a subcommand would run
                                            (no side effects) — powers the per-command confirm/Edit gate
  install-cli                               Install the prerelease temporal-cloud CLI (updates it to latest if already present)
  login                                     temporal cloud login (browser) + whoami
  regions                                   List Cloud regions (raw list to stderr)
  start-namespace --sdk S --region R        Fire-and-forget: submit `namespace create --async` (server-side); emits namespace_name
  await-namespace --name N [--max-secs M]   Join: poll `namespace list` until N provisions; emits namespace_handle + address
  scaffold --sdk S [--dir D] [--manager M]  Set up the app only: clone the cloud-ready sample + install deps; emits repo_path, manager
  create-namespace --sdk S --region R       Blocking wrapper = start + await (non-parallel fallback); emits namespace_handle + address
  create-key --handle H --address A         Mint API key, write [profile.cloud-setup] TOML; emits key_id (token never printed)
  verify-config                             temporal --profile cloud-setup config list
  await-auth                                Wait until the new API key is accepted (poll `workflow list`) before the Worker
  run-workflow --sdk S --dir D [--demo-failure transient] [--max-secs N]
                                            Single synchronous call: start Worker (bg) -> wait until polling -> run starter -> stop Worker; emits workflow_status, workflow_id, run_id
  clone --sdk S [--dir D]                   Clone the cloud-ready sample for the SDK
  install-deps --sdk S --dir D [--manager M]  Install sample deps with a chosen/default package manager
  provision-and-scaffold --sdk S --region R [--dir D] [--manager M]
                                            Single synchronous call: namespace create (bg) || clone+deps; emits namespace_handle, address, repo_path, manager
  repair-config                             Strip duplicate/old [profile.cloud-setup] blocks (keeps default)
  cleanup-info --handle H --key-id K        Print (do not run) the cleanup commands

All commands print a delimited "=== RESULT ===" block on stdout for the agent to parse.
EOF
}

main() {
  local cmd="${1:-}"; shift || true
  # Deterministic disclosure floor: echo the subcommand's gate to stderr before it
  # acts, so the tool block records what ran even if the agent skipped the chat-side
  # gate. `preview` only prints, `cleanup-info` runs nothing, help/empty don't act —
  # none need an echo (and cmd_preview can't render them anyway).
  case "$cmd" in
    preview|cleanup-info|""|-h|--help|help) : ;;
    *) announce_gate "$cmd" "$@" ;;
  esac
  case "$cmd" in
    preflight)        cmd_preflight "$@" ;;
    detect-tools)     cmd_detect_tools "$@" ;;
    preview)          cmd_preview "$@" ;;
    install-deps)     cmd_install_deps "$@" ;;
    install-cli)      cmd_install_cli "$@" ;;
    login)            cmd_login "$@" ;;
    regions)          cmd_regions "$@" ;;
    start-namespace)  cmd_start_namespace "$@" ;;
    await-namespace)  cmd_await_namespace "$@" ;;
    scaffold)         cmd_scaffold "$@" ;;
    create-namespace) cmd_create_namespace "$@" ;;
    create-key)       cmd_create_key "$@" ;;
    verify-config)    cmd_verify_config "$@" ;;
    await-auth)       cmd_await_auth "$@" ;;
    run-workflow)     cmd_run_workflow "$@" ;;
    clone)            cmd_clone "$@" ;;
    provision-and-scaffold) cmd_provision_and_scaffold "$@" ;;
    repair-config)    cmd_repair_config "$@" ;;
    cleanup-info)     cmd_cleanup_info "$@" ;;
    ""|-h|--help|help) usage ;;
    *) usage; die bad-command "unknown command: $cmd" ;;
  esac
}

main "$@"

SHA-256: 0ae7d2892766b42010d3a8d978299d0893d24f0ec458fbef11771e594d0e1199