#!/bin/bash
# run-paths.sh  -  the single resolver for pipeline run-state paths (shell side).
#
# Shell twin of pipeline/scripts/_run-paths.mjs. The two must agree; the
# contract is asserted by pipeline/scripts/smoke-run-path-canonical.sh, which
# runs both over the same fixture tree and diffs their answers.
#
# A run's files live under the log root in ONE of two layouts:
#
#   nested   <root>/<project>/<task_id>/     documented canonical
#   flat     <root>/<task_id>/               what phase-tracker.sh writes
#
# Both are populated on real machines, so every reader accepts both. This file
# does NOT change where anything is written: measured on a real install, 90 of
# 103 runs are flat and every tracker-state.json is. A task id present in both
# layouts is ONE run; the most recently written directory wins. Relocation is
# opt-in and explicit: `migrate-state.mjs --relocate`.
#
# No `set -e` on purpose: this is a sourced library and changing the caller's
# shell options is not this file's business.
#
# Usage:
#   . "$(dirname "$0")/../lib/run-paths.sh"
#   root=$(ma_logs_root)
#   dir=$(ma_resolve_run_dir "{JIRA_KEY}-123") || echo "unknown run"
#   f=$(ma_resolve_run_file "{JIRA_KEY}-123" tracker-state.json)
#   ma_list_runs   # task_id<TAB>project<TAB>dir<TAB>layout, one per run
#
# Bash 3.2 compatible: no associative arrays, no `mapfile`, no `grep -P`.

# Marker files and reserved names are spelled out at each test rather than held
# in a space-separated variable iterated with `for x in $VAR`. bash word-splits
# an unquoted variable; zsh does not, so there the loop would run ONCE with the
# whole string as a single word, every test would fail, and the library would
# answer "no runs" instead of erroring. This file is sourced, so the caller's
# shell decides - and answering zero is the failure mode that looks like data.

ma_logs_root() {
  printf '%s\n' "${LOGS_ROOT:-$HOME/.claude/logs/multi-agent}"
}

# ma_is_run_dir <dir> -> 0 when the directory holds at least one run marker
# Phase 4 removes the worktree and salvages the run's files into `artifacts/`
# inside the same run directory, so a shipped run keeps its state one level
# deeper. A reader that only looks at the top level reports it as stateless.
MA_ARTIFACTS_SUBDIR=artifacts

ma_is_run_dir() {
  local d="$1"
  [ -f "$d/agent-state.json" ] && return 0
  [ -f "$d/tracker-state.json" ] && return 0
  [ -f "$d/agent-log.md" ] && return 0
  [ -f "$d/$MA_ARTIFACTS_SUBDIR/agent-state.json" ] && return 0
  [ -f "$d/$MA_ARTIFACTS_SUBDIR/tracker-state.json" ] && return 0
  [ -f "$d/$MA_ARTIFACTS_SUBDIR/agent-log.md" ] && return 0
  return 1
}

ma_is_reserved() {
  case "$1" in
    review-watch | jira-backups | shadow-git | _analysis-jira | prompts | contract-server) return 0 ;;
  esac
  return 1
}

# Newest mtime across a run directory's markers, as epoch seconds. GNU-first
# `stat -c` ordering is required: `stat -f` is a valid GNU flag
# (--file-system) that SUCCEEDS, so BSD-first would silently win whenever GNU
# coreutils is first on PATH (Homebrew gnubin).
ma_run_mtime() {
  local d="$1" m t newest=0 b
  for b in "$d" "$d/$MA_ARTIFACTS_SUBDIR"; do
    for m in agent-state.json tracker-state.json agent-log.md; do
      [ -f "$b/$m" ] || continue
      # -L follows symlinks. Some nested run directories are symlink bridges to
      # the flat copy of the same run (3 such pairs on the install this was
      # measured on); without -L the shell reads the LINK's mtime while the
      # .mjs twin's fs.statSync reads the TARGET's, and the two picked
      # different winners for one record. Both must read the same clock.
      t=$(stat -L -c %Y "$b/$m" 2>/dev/null || stat -L -f %m "$b/$m" 2>/dev/null || echo 0)
      [ -n "$t" ] || t=0
      [ "$t" -gt "$newest" ] && newest="$t"
    done
  done
  printf '%s\n' "$newest"
}

# ma_canonical_run_dir <task_id> [project] -> where a NEW run should be written
ma_canonical_run_dir() {
  local task_id="$1" project="${2:-}" root
  root=$(ma_logs_root)
  if [ -n "$project" ]; then
    printf '%s\n' "$root/$project/$task_id"
  else
    printf '%s\n' "$root/$task_id"
  fi
}

# ma_run_dir_candidates <task_id> [project] -> candidate dirs, most specific
# first. Existence is not checked here.
ma_run_dir_candidates() {
  local task_id="$1" project="${2:-}" root d n
  root=$(ma_logs_root)
  [ -n "$project" ] && printf '%s\n' "$root/$project/$task_id"
  printf '%s\n' "$root/$task_id"
  [ -n "$project" ] && return 0
  # No project given: the run may still be nested under one.
  #
  # `find` rather than a `*/` glob on purpose: an unmatched glob is a literal
  # string in bash and a hard error in zsh, and this file is sourced, so the
  # caller's shell decides. find behaves identically in both.
  while IFS= read -r d; do
    [ -n "$d" ] || continue
    n=$(basename "$d")
    ma_is_reserved "$n" && continue
    [ "$n" = "$task_id" ] && continue
    ma_is_run_dir "$root/$n/$task_id" && printf '%s\n' "$root/$n/$task_id"
  done <<EOF
$(find "$root" -mindepth 1 -maxdepth 1 -type d 2>/dev/null)
EOF
  return 0
}

# ma_realpath <path> -> the path with every symlink resolved, or the input when
# it cannot be resolved. BSD realpath first, python3 as the fallback (both are
# already hard requirements of this tree).
ma_realpath() {
  realpath "$1" 2>/dev/null ||
    python3 -c 'import os,sys; print(os.path.realpath(sys.argv[1]))' "$1" 2>/dev/null ||
    printf '%s\n' "$1"
}

# ma_is_bridge_dir <dir> -> 0 when any marker in it is a symlink elsewhere.
# Some nested run directories are symlink bridges into the flat copy of the
# same run. They are one record seen twice, not two copies.
ma_is_bridge_dir() {
  local d="$1" m
  for m in agent-state.json tracker-state.json agent-log.md; do
    [ -L "$d/$m" ] && return 0
  done
  return 1
}

# ma_same_record <dir_a> <dir_b> -> 0 when both are views of ONE record.
ma_same_record() {
  local a="$1" b="$2" m ra rb
  for m in agent-state.json tracker-state.json agent-log.md; do
    [ -e "$a/$m" ] && [ -e "$b/$m" ] || continue
    ra=$(ma_realpath "$a/$m")
    rb=$(ma_realpath "$b/$m")
    [ "$ra" = "$rb" ] && return 0
  done
  return 1
}

# ma_rank_fields <dir> -> "<mtime>\t<has_state>\t<depth>", the three ranking
# columns used to pick between candidate directories for one task id.
#
# ma_sort_key packs the same three into one space-separated key for callers
# that sort a single column; both must match compareCandidates in
# _run-paths.mjs exactly.
#
# mtime alone is not an order: two directories written in the same second tie,
# and each implementation then fell back to its own traversal order - which is
# how the two came to disagree about a run present in both layouts. The tail of
# the key defines the answer: richer record first (agent-state.json is what
# every reader wants), then the documented nested layout, then the path.
ma_rank_fields() {
  local d="$1" t has_state depth real
  t=$(ma_run_mtime "$d")
  has_state=0
  [ -f "$d/agent-state.json" ] && has_state=1
  depth=$(printf '%s' "$d" | awk -F/ '{print NF}')
  real=1
  ma_is_bridge_dir "$d" && real=0
  printf '%s\t%s\t%s\t%s\n' "$real" "$t" "$has_state" "$depth"
}

ma_sort_key() {
  local d="$1" t has_state depth real
  t=$(ma_run_mtime "$d")
  has_state=0
  [ -f "$d/agent-state.json" ] && has_state=1
  depth=$(printf '%s' "$d" | awk -F/ '{print NF}')
  real=1
  ma_is_bridge_dir "$d" && real=0
  printf '%d %012d %d %04d %s\n' "$real" "$t" "$has_state" "$depth" "$d"
}

# ma_resolve_run_dir <task_id> [project] -> the directory, or rc 1 when unknown.
ma_resolve_run_dir() {
  local task_id="$1" project="${2:-}" c best
  best=$(
    while IFS= read -r c; do
      [ -n "$c" ] || continue
      ma_is_run_dir "$c" || continue
      ma_sort_key "$c"
    done <<EOF
$(ma_run_dir_candidates "$task_id" "$project")
EOF
  )
  [ -n "$best" ] || return 1
  # Descending on mtime, then has-state, then depth; ascending on path.
  printf '%s\n' "$best" | sort -k1,1nr -k2,2nr -k3,3nr -k4,4nr -k5,5 | head -1 | cut -d' ' -f5-
}

# ma_task_id_variants <id> -> the spellings one task id has been written under,
# in preference order. `#316` arrives from a GitHub issue reference, `316` from
# the bare-number input class, `task-316` from an older directory convention.
# Callers used to inline this list; dropping a spelling silently stops old runs
# from resolving.
ma_task_id_variants() {
  local raw="$1" bare
  bare="${raw#\#}"
  printf '%s\n' "$raw"
  [ "$bare" != "$raw" ] && printf '%s\n' "$bare"
  printf '%s\n' "task-$bare"
}

# ma_resolve_run_dir_any <task_id> [project] -> dir for any spelling, or rc 1
ma_resolve_run_dir_any() {
  local task_id="$1" project="${2:-}" v dir
  while IFS= read -r v; do
    [ -n "$v" ] || continue
    dir=$(ma_resolve_run_dir "$v" "$project") && {
      printf '%s\n' "$dir"
      return 0
    }
  done <<EOF
$(ma_task_id_variants "$task_id")
EOF
  return 1
}

# ma_resolve_run_file <task_id> <filename> [project] -> path, or rc 1.
# Every spelling of the id is tried.
ma_resolve_run_file() {
  local task_id="$1" filename="$2" project="${3:-}" v dir
  while IFS= read -r v; do
    [ -n "$v" ] || continue
    dir=$(ma_resolve_run_dir "$v" "$project") || continue
    if [ -f "$dir/$filename" ]; then
      printf '%s\n' "$dir/$filename"
      return 0
    fi
    # The salvaged copy Phase 4 leaves behind.
    if [ -f "$dir/$MA_ARTIFACTS_SUBDIR/$filename" ]; then
      printf '%s\n' "$dir/$MA_ARTIFACTS_SUBDIR/$filename"
      return 0
    fi
  done <<EOF
$(ma_task_id_variants "$task_id")
EOF
  return 1
}

# ma_list_runs -> one TAB-separated row per run, deduplicated by task id:
#   task_id<TAB>project<TAB>dir<TAB>layout
# `project` is "-" when the layout does not name one.
ma_list_runs() {
  local root d n kd kn rows row id
  root=$(ma_logs_root)
  [ -d "$root" ] || return 0
  rows=""
  while IFS= read -r d; do
    [ -n "$d" ] || continue
    n=$(basename "$d")
    ma_is_reserved "$n" && continue
    if ma_is_run_dir "$root/$n"; then
      rows="$rows$n	-	$root/$n	flat	$(ma_rank_fields "$root/$n")
"
      continue
    fi
    while IFS= read -r kd; do
      [ -n "$kd" ] || continue
      kn=$(basename "$kd")
      ma_is_run_dir "$root/$n/$kn" || continue
      rows="$rows$kn	$n	$root/$n/$kn	nested	$(ma_rank_fields "$root/$n/$kn")
"
    done <<EOF
$(find "$root/$n" -mindepth 1 -maxdepth 1 -type d 2>/dev/null)
EOF
  done <<EOF
$(find "$root" -mindepth 1 -maxdepth 1 -type d 2>/dev/null)
EOF
  # Dedup by task id using the same total order as resolveRunDir and as
  # compareCandidates in the .mjs twin: mtime desc, has-state desc, depth desc,
  # path ASC. The sub-keys are separate columns on purpose - a single reverse
  # sort over a composite key also reverses the path, which is the one
  # component that has to ascend.
  printf '%s' "$rows" | grep -v '^$' |
    sort -t'	' -k1,1 -k5,5nr -k6,6nr -k7,7nr -k8,8nr -k3,3 | awk -F'	' '
    !seen[$1]++ { print $1 "\t" $2 "\t" $3 "\t" $4 }
  ' | sort -t'	' -k1,1
}

# ma_duplicate_run_ids -> task ids that exist as two SEPARATE records in the two
# layouts, one per line. A symlink bridge is excluded: it is one record seen
# twice, and reporting it overstates the drift.
ma_duplicate_run_ids() {
  local root d n kd kn pairs id dirs a b
  root=$(ma_logs_root)
  [ -d "$root" ] || return 0
  pairs=$(
    while IFS= read -r d; do
      [ -n "$d" ] || continue
      n=$(basename "$d")
      ma_is_reserved "$n" && continue
      if ma_is_run_dir "$root/$n"; then
        printf '%s\t%s\n' "$n" "$root/$n"
        continue
      fi
      while IFS= read -r kd; do
        [ -n "$kd" ] || continue
        kn=$(basename "$kd")
        ma_is_run_dir "$root/$n/$kn" && printf '%s\t%s\n' "$kn" "$root/$n/$kn"
      done <<EOF
$(find "$root/$n" -mindepth 1 -maxdepth 1 -type d 2>/dev/null)
EOF
    done <<EOF
$(find "$root" -mindepth 1 -maxdepth 1 -type d 2>/dev/null)
EOF
  )
  # Ids seen more than once, minus the ones whose two directories resolve to the
  # same underlying files (a symlink bridge is one record, not drift).
  while IFS= read -r id; do
    [ -n "$id" ] || continue
    dirs=$(printf '%s\n' "$pairs" | awk -F'\t' -v k="$id" '$1==k {print $2}')
    a=$(printf '%s\n' "$dirs" | sed -n 1p)
    b=$(printf '%s\n' "$dirs" | sed -n 2p)
    [ -n "$b" ] || continue
    ma_same_record "$a" "$b" || printf '%s\n' "$id"
  done <<EOF
$(printf '%s\n' "$pairs" | awk -F'\t' '{print $1}' | sort | uniq -d)
EOF
}
