# shellcheck shell=sh
# Self-measurement (FR-49). Reports whether the discipline actually
# happened, from session transcripts, instead of asserting that it did.
#
# Usage:  sh .claude/skills/skill-router/measure-discipline.sh [transcript-dir]
#         sh .claude/skills/skill-router/measure-discipline.sh --otel <export.jsonl>
#
# The --otel form (CAP-17) reads an OpenTelemetry export instead of
# transcripts: one JSON object per line as the console exporter or a
# collector file exporter writes it. It counts the tool decisions the
# harness attributed to a hook (event claude_code.tool_decision with
# source "hook"): accepted, rejected, and how many carried a skill
# name. Enable the export with CLAUDE_CODE_ENABLE_TELEMETRY=1,
# OTEL_LOGS_EXPORTER=console (or otlp to a collector) and
# OTEL_LOG_TOOL_DETAILS=1, redirecting the output to a file.
#
# This is the same method that produced the finding this whole system
# exists to answer: a 668-message session with 65 installed skills and
# zero skill invocations. It is therefore known to work, and it is the
# only honest source for M1, M2 and M5.
#
# Stated limits, because a measurement that hides its limits is worse
# than none (NFR-6, FR-50):
#   * The transcript is written asynchronously and may lag the current
#     turn, so a very recent invocation can be missing. Counts are
#     therefore a LOWER bound on skill use.
#   * "A skill was loaded" is not "the skill was followed". Nothing here
#     measures obedience, and no report from it may claim that.
#   * Sessions are counted, not turns: a session that loaded the router
#     once counts as routed even if a later turn ignored it.

set -u

if [ "${1:-}" = "--otel" ]; then
  otel_="${2:-}"
  if [ -z "${otel_}" ] || [ ! -f "${otel_}" ]; then
    echo "usage: measure-discipline.sh --otel <export.jsonl>" >&2
    exit 1
  fi
  events_=$(LC_ALL=C grep -a -c 'claude_code.tool_decision' "${otel_}" || true)
  hook_=$(LC_ALL=C grep -a 'claude_code.tool_decision' "${otel_}" \
    | LC_ALL=C grep -a -E -c '"source"[[:space:]]*:[[:space:]]*"hook"' || true)
  hook_reject_=$(LC_ALL=C grep -a 'claude_code.tool_decision' "${otel_}" \
    | LC_ALL=C grep -a -E '"source"[[:space:]]*:[[:space:]]*"hook"' \
    | LC_ALL=C grep -a -E -c '"decision"[[:space:]]*:[[:space:]]*"reject"' || true)
  hook_accept_=$((hook_ - hook_reject_))
  skill_=$(LC_ALL=C grep -a 'claude_code.tool_decision' "${otel_}" \
    | LC_ALL=C grep -a -E -c 'skill_name' || true)
  echo "measured over ${events_} tool decision(s) in ${otel_}"
  echo "M6  decisions made by a hook         : ${hook_}/${events_}"
  echo "M7  of those, hook rejections        : ${hook_reject_} (accepted ${hook_accept_})"
  echo "M8  decisions naming a skill         : ${skill_}"
  echo ""
  echo "Hook decisions show the gate acted. They do not show that the"
  echo "router was followed. Nothing here measures obedience."
  exit 0
fi

dir_="${1:-}"
if [ -z "${dir_}" ]; then
  slug_="$(pwd | sed 's|/|-|g')"
  dir_="${HOME}/.claude/projects/${slug_}"
fi

if [ ! -d "${dir_}" ]; then
  echo "no transcript directory at ${dir_}" >&2
  echo "pass one explicitly: sh measure-discipline.sh <dir>" >&2
  exit 1
fi

sessions_=0
routed_=0
gated_=0
skills_total_=0

for file_ in "${dir_}"/*.jsonl; do
  [ -f "${file_}" ] || continue
  sessions_=$((sessions_ + 1))

  if LC_ALL=C grep -a -E -q '"skill"[[:space:]]*:[[:space:]]*"skill-router"' \
      "${file_}" 2>/dev/null; then
    routed_=$((routed_ + 1))
  fi

  # Every distinct skill invoked in this session.
  count_=$(LC_ALL=C grep -a -o -E '"skill"[[:space:]]*:[[:space:]]*"[a-z0-9-]+"' \
    "${file_}" 2>/dev/null | LC_ALL=C sort -u | LC_ALL=C grep -c . || true)
  skills_total_=$((skills_total_ + count_))

  # A write-gate denial is itself a transcript event, so the gate needs
  # no separate instrumentation (M5).
  if LC_ALL=C grep -a -F -q 'Write gate: this session has no skill-router' \
      "${file_}" 2>/dev/null; then
    gated_=$((gated_ + 1))
  fi
done

if [ "${sessions_}" = "0" ]; then
  echo "no transcripts found in ${dir_}" >&2
  exit 1
fi

echo "measured over ${sessions_} session(s) in ${dir_}"
echo "M1  sessions that loaded the router : ${routed_}/${sessions_}"
echo "M2  distinct skills loaded, total   : ${skills_total_}"
echo "M5  sessions where the gate fired   : ${gated_}/${sessions_}"
echo ""
echo "Lower bounds only: the transcript may lag the current turn, and a"
echo "loaded skill is not a followed skill. Nothing here measures obedience."
exit 0
