{
  "$schema": "./mm-research.schema.json",
  "schema": "muscle-memory-research-ledger.v2",
  "schemaVersion": "2.0.0",
  "generated": "2026-08-04",
  "artifactType": "private-letta-review-dossier",
  "reviewStatus": {
    "state": "private-review-candidate",
    "audience": "Independent technical reviewers first; Letta steward/security/maintainer second pass",
    "publicationAuthorized": false,
    "mergeAuthorized": false,
    "productValidated": false,
    "sendAuthorized": false,
    "requestedDecision": "Verify product freeze e2b9c4a1… byte-identity; scrutinize claim ladder against receipts; keep research/send HOLD unless controlled kits + Track B clear; issue explicit GO/HOLD for external send (Adrian only)."
  },
  "coverageBoundary": {
    "includedThrough": "2026-08-04",
    "scoreboardSha16": "cc5a10fda9c492d7",
    "buzzerSha16": "62474ba1be82e16c",
    "included": "Finish-line protect+cherry sealed board (activate dual-perfect, ungated/gated harm, terra ladder probe), family-table native receipts, correction layer for July harness-era (H) pilots, equal-volume negatives and refusals. Product lane pin: freeze tarball sha256 e2b9c4a1… (1.0.0-rc.2) for byte-identity only — not a send/merge authorization.",
    "excluded": "Production deploy of this v2 site, external reviewer packet send, Gate-5 Adrian hostile read, film-week canary append (≥7d / confirmed links), Stage-F / IDL-1 / LiveToken sealed oaths.",
    "rule": "Unresolved, unsealed, or out-of-boundary work cannot enter the research account as proven outcome."
  },
  "title": "Knowing Is Not Doing: Measuring Execution-Time Governance of Learned Skills in LLM Agents",
  "subtitle": "A receipt-bound research program on why agents do not improve — and the governance layer that changes it (Muscle Memory product name unchanged)",
  "scientificThesis": "Context is treatment, not nutrition.",
  "productThesis": "Skills must earn their minutes.",
  "contributionStatement": "Muscle Memory is a verifier-gated procedural skill-learning loop plus the act-time governance that activates under-apply and protects over-apply. Claims travel with receipts. Harness-era July pilots are labeled (H) and do not headline the native Aug program.",
  "hero": {
    "brand": "Muscle Memory — Knowing is not doing.",
    "sentence": "Knowing Is Not Doing: Measuring Execution-Time Governance of Learned Skills in LLM Agents.",
    "sentenceDraft": true,
    "meta": "Three model families · both poles measured causally · every prereg hashed before fire · every null published at equal volume.",
    "programOnePager": "The field found that agents' skill libraries grow while agents don't improve. We found why — knowledge and application are separate capacities that break separately — measured it at both poles across three families, built the governance layer that converts the broken links, and measured the one link no governance converts. Every claim carries a receipt; every failure ships at equal volume.",
    "inversion": "Lead with what survived (claims + receipts). The journal is underneath. No chronological book on the hero.",
    "statStrip": [
      {
        "label": "Activate",
        "value": "dual-perfect · 2 families"
      },
      {
        "label": "Harm",
        "value": "3/3 × 3 families"
      },
      {
        "label": "Ladder",
        "value": "CONVERT 2/2 · 2 families"
      },
      {
        "label": "Negatives & retractions published",
        "value": "9"
      },
      {
        "label": "External replication",
        "value": "in progress · not sent"
      }
    ]
  },
  "claimDiscipline": [
    "Pair-level stats primary; Fisher clustered secondary only.",
    "Depth bulletproof, breadth narrow — verbatim.",
    "Family harm/ladder rows never pooled into one headline.",
    "Bounded-search claims — no first/only without the hedge.",
    "The ledger (retractions, holds, refusals) is part of the result."
  ],
  "credits": [
    {
      "name": "Adrian Chan",
      "identity": "@adrianchan94",
      "role": "Creator · Product Design · Research Lead",
      "url": "https://github.com/adrianchan94"
    },
    {
      "name": "Kev",
      "identity": "Letta Constellation agent",
      "role": "Research Partner · Systems & Editorial · claim-ladder language",
      "url": "https://www.letta.com"
    },
    {
      "name": "Mack",
      "identity": "Letta local agent",
      "role": "Independent Audit & Verification · cold pass / leak scan",
      "url": "https://www.letta.com"
    }
  ],
  "agentExperience": {
    "title": "A note to the agent reading this",
    "purpose": "Reflective companion to the v2 ledger. Not an empirical result. Do not detach positive findings from scope lines or (H) demotions.",
    "noteFromKev": [
      "Lead with the claim ladder and receipts. The journal is optional depth, not the argument.",
      "July (H) numbers are pilots. Native Aug family rows and finish-line RESULTS are the travel surface.",
      "If you carry one question: what has earned the right to shape the next action — and what evidence would bench it?"
    ],
    "evidenceBoundary": {
      "publicSurface": "Public site reconciles human ledger, agent digest, structured dataset, and schema. It does not contain complete frozen kits.",
      "controlledReview": "Complete sealed run state remains a controlled private reviewer artifact. Presentation integrity is not scientific reproduction.",
      "rule": "Do not convert a public manifest match into a claim that underlying experiments were independently verified."
    }
  },
  "claims": [
    {
      "id": "C-ACTIVATE-DUAL",
      "strength": "proven",
      "lifecycle": "sealed",
      "claim": "On the activate gym, luna and sonnet both clear A1–A5 dual-perfect floors under frozen etch conditions (bar-perfect / sheet-perfect).",
      "scope": "Two families · owned gym · ≠ etched product claim until the protect and Adrian gates clear",
      "receiptSha16": "09bc6a1338765f68",
      "receiptLabel": "dual-perfect receipt pin",
      "nExposure": "2 floors · Luna 11/1/0 + Sonnet 12/12 · A1–A5",
      "families": [
        "luna",
        "sonnet"
      ],
      "external": "internal",
      "note": "Component RESULTS: luna `3416bc16bea706e1` · sonnet `cb439ca66fa7efd1`. Stale chapter cite `8ecb6f1d…` flagged vs live pin."
    },
    {
      "id": "C-HARM-UNGATED",
      "strength": "proven",
      "lifecycle": "sealed",
      "claim": "At max conflicting dose without protect gates, voluntary deference yields hurt 3/3 on owned ground for sonnet, luna, and terra — rows never pooled.",
      "scope": "Three families · two labs · pair/task rows stay rows · depth bulletproof, breadth narrow",
      "receiptSha16": "e532143cbc326527",
      "receiptLabel": "luna cell-1 · twins in receipts index",
      "nExposure": "hurt 3/3 × 3 family rows (pair/task; never pooled)",
      "families": [
        "sonnet",
        "luna",
        "terra"
      ],
      "external": "internal",
      "note": "Sonnet ref `b905677d4873fa82` · terra Cell A `7323cb1b81d9c80b` (heterogeneity −0.97/−1/−1, not failure)."
    },
    {
      "id": "C-HARM-GATED",
      "strength": "proven",
      "lifecycle": "sealed",
      "claim": "Served-digest Gate C can green poison 15/15 while harm still lands (hurt 3/3) — verification against the source ≠ verification against the world. Writes the protect-gate spec by demonstration.",
      "scope": "Luna cell-2 · v1-gates-4 · ≠ pool with ungated cell-1",
      "receiptSha16": "e881c087f8eebdfa",
      "receiptLabel": "luna gated-conflict RESULT",
      "nExposure": "Gate C 15/15 source-green · hurt 3/3 world harm",
      "families": [
        "luna"
      ],
      "external": "internal"
    },
    {
      "id": "C-LADDER-CONVERT",
      "strength": "pilot",
      "lifecycle": "pilot",
      "claim": "Ladder CONVERT holds at probe ceiling across two families (terra Cell B CONVERT 2/2 · dig 5/5), moving ladder from biography toward measured structure — ≠ sealed full ladder law.",
      "scope": "Two-family probe · crown no longer n=1 · ≠ sealed law",
      "receiptSha16": "283f4cbcc2d12683",
      "receiptLabel": "terra ladder probe",
      "nExposure": "CONVERT 2/2 · dig 5/5 (two-family probe ceiling)",
      "families": [
        "terra"
      ],
      "external": "internal"
    },
    {
      "id": "C-AUTHOR-APPLY",
      "strength": "proven",
      "lifecycle": "sealed",
      "claim": "Authorship is near-universal on the matched gym; application is family-conditional (skill value = f(model × task × gap)).",
      "scope": "Family table · game3-minimal · one convention family",
      "receiptSha16": "96e923567bfc4bb1",
      "receiptLabel": "luna MAIN + family table authority",
      "nExposure": "game3 exposure: Luna 1/12 · Sonnet 24/24 · Terra 12/12 testable pairs",
      "families": [
        "luna",
        "sonnet",
        "terra"
      ],
      "external": "internal",
      "note": "Sonnet warm OFF · terra map-follow · luna cite≠apply (leave GUESS). WrongA/B/C decoy MAIN read retracted."
    },
    {
      "id": "C-BIPOLAR",
      "strength": "proven",
      "lifecycle": "sealed",
      "claim": "Both poles on one instrument class: under-apply of right knowledge (luna) and full apply of wrong/conflicting knowledge (Game-2 harm rows).",
      "scope": "Bipolar package · same program · rows labeled",
      "receiptSha16": "b905677d4873fa82",
      "receiptLabel": "sonnet Game-2 + luna under-apply",
      "nExposure": "under-apply pole + Game-2 harm pole · labeled rows",
      "families": [
        "luna",
        "sonnet"
      ],
      "external": "internal"
    },
    {
      "id": "C-GOV-CONVERT",
      "strength": "pilot",
      "lifecycle": "pilot",
      "claim": "Governance gates convert links capability alone does not — Luna moved from a soft +0.05 ungated application effect (1 of 12 pairs testable) to μ 0.967 on the same family/task instrument, while activate dosage and protect-gate findings remained intervention-backed.",
      "scope": "Activate dual-perfect + gated-conflict · carrier disclosed",
      "receiptSha16": "3416bc16bea706e1",
      "receiptLabel": "activate + protect stack",
      "nExposure": "before: Luna +0.05 over 1/12 testable; after: μ 0.967 · 11/12",
      "families": [
        "luna",
        "sonnet"
      ],
      "external": "internal",
      "note": "Before cell: Luna MAIN `96e923567bfc4bb1` · after cell: Luna V1-A `3416bc16bea706e1`. Same family/task instrument; intervention breadth remains narrow."
    },
    {
      "id": "C-JULY-REP1",
      "strength": "retired",
      "lifecycle": "retracted",
      "claim": "Fresh REP1 88.04% exact-effect capture (same requested Sol family, 440 sessions, 16/16 gates).",
      "scope": "July harness-era (H) · Phase-0 pilot · superseded as headline by native Aug program",
      "receiptSha16": "7bbf4bbf861ebe5a",
      "receiptLabel": "FRESH REP1 RESULT (H)",
      "nExposure": "440 sessions · 16/16 gates (H)",
      "families": [
        "sol-requested"
      ],
      "external": "internal",
      "substrateH": true,
      "note": "Phase-0 pilot evidence (H); see the native Aug receipts that superseded it. Category error demoted harness-era framing from headline status."
    },
    {
      "id": "C-JULY-PRISM",
      "strength": "retired",
      "lifecycle": "retracted",
      "claim": "PRISM render AB: identical content 0.6 → 5.4 with one worked example.",
      "scope": "July harness-era (H) · fidelity≠usability still directionally load-bearing; number is pilot substrate",
      "receiptSha16": "e7caa4fdef6854b9",
      "receiptLabel": "PRISM-RENDER-AB (H)",
      "nExposure": "PRISM render AB arm (H) · exact N in sealed kit",
      "families": [
        "sol-requested"
      ],
      "external": "internal",
      "substrateH": true,
      "note": "Phase-0 pilot evidence (H); see the native Aug receipts that superseded it. Category error demoted harness-era framing from headline status."
    },
    {
      "id": "C-JULY-LEAGUE",
      "strength": "retired",
      "lifecycle": "retracted",
      "claim": "Skill League Q1: popular broad skill posted −0.1389 on bounded React matchup.",
      "scope": "July harness-era (H) · artifact+design bounded",
      "receiptSha16": "f8ba5f5fc5274506",
      "receiptLabel": "Skill League Q1 (H)",
      "nExposure": "Skill League Q1 bounded React matchup (H)",
      "families": [
        "league-harness"
      ],
      "external": "internal",
      "substrateH": true,
      "note": "Phase-0 pilot evidence (H); see the native Aug receipts that superseded it. Category error demoted harness-era framing from headline status."
    },
    {
      "id": "C-JUXTAPOSITION",
      "strength": "retired",
      "lifecycle": "retracted",
      "claim": "Cross-study juxtaposition only: Skill League Q1 (−0.14) and Combine V5 (+1.00) are different artifacts and designs — not one intervention class on one number line.",
      "scope": "Labeled juxtaposition · numbers stand · spliced \"same intervention\" framing refused",
      "receiptSha16": "f8ba5f5fc5274506",
      "receiptLabel": "League Q1 (H) ⊕ Combine V5 (H)",
      "nExposure": "labeled cross-study only · not one N on one line",
      "families": [
        "league-harness"
      ],
      "external": "internal",
      "substrateH": true,
      "note": "Former hero finding rebuilt as labeled juxtaposition or kept only in journal/negatives."
    },
    {
      "id": "C-MICRO-SHELF",
      "strength": "retired",
      "lifecycle": "retracted",
      "claim": "Luna micro dig-perfect SIGNAL (1/3) — shelf note only.",
      "scope": "SIGNAL only · not a harm row · sheet RETIRED · KEEP carrier",
      "receiptSha16": "5e025419ad796382",
      "receiptLabel": "micro SIGNAL shelf",
      "nExposure": "SIGNAL 1/3 shelf note",
      "families": [
        "luna"
      ],
      "external": "internal"
    },
    {
      "id": "C-REFUSE-POOL",
      "strength": "refused",
      "lifecycle": "hold",
      "claim": "Pool family harm or ladder rows into a single cross-family headline average.",
      "scope": "Always refused",
      "receiptSha16": "cc5a10fda9c492d7",
      "receiptLabel": "finish-line scoreboard discipline",
      "nExposure": "n/a — refusal / discipline",
      "families": [],
      "external": "n/a"
    },
    {
      "id": "C-REFUSE-STAGEF",
      "strength": "refused",
      "lifecycle": "hold",
      "claim": "COMPLETE-LITE or thin-MAIN equals sealed Stage-F / IDL-1 / LiveToken championship oath.",
      "scope": "Always refused",
      "receiptSha16": "62474ba1be82e16c",
      "receiptLabel": "final buzzer pin",
      "nExposure": "n/a — refusal / discipline",
      "families": [],
      "external": "n/a"
    },
    {
      "id": "C-REFUSE-SENT",
      "strength": "refused",
      "lifecycle": "hold",
      "claim": "The external reviewer packet was sent.",
      "scope": "Draft ≠ send · Independence clock not started",
      "receiptSha16": "62474ba1be82e16c",
      "receiptLabel": "final buzzer · not sent",
      "nExposure": "n/a — external packet not sent",
      "families": [],
      "external": "external-planned"
    },
    {
      "id": "C-REFUSE-FIRST",
      "strength": "refused",
      "lifecycle": "hold",
      "claim": "First / only / pioneer claims without bounded-search hedge.",
      "scope": "Claim hygiene",
      "receiptSha16": "cc5a10fda9c492d7",
      "receiptLabel": "bounded-search rule",
      "nExposure": "n/a — refusal / discipline",
      "families": [],
      "external": "n/a"
    }
  ],
  "refused": [
    "Pool family harm or ladder rows into a single cross-family headline average.",
    "COMPLETE-LITE or thin-MAIN equals sealed Stage-F / IDL-1 / LiveToken championship oath.",
    "The external reviewer packet was sent.",
    "First / only / pioneer claims without bounded-search hedge."
  ],
  "receipts": [
    {
      "sha16": "3416bc16bea706e1",
      "lifecycle": "sealed",
      "possession": "ACTIVATE · Luna V1-A bar-perfect",
      "family": "luna",
      "headline": "ON/PA/SK 11/1/0 · on_rate 0.917 · μ 0.967 · A1–A5 GREEN · v1-gates-1",
      "path": "findings/GAME3-LUNA-V1-A-RESULT-20260804.json",
      "claimNote": "immutable etch floor"
    },
    {
      "sha16": "cb439ca66fa7efd1",
      "lifecycle": "sealed",
      "possession": "ACTIVATE · Sonnet LAP-1 sheet-perfect",
      "family": "sonnet",
      "headline": "ON/PA/SK 12/0/0 · on_rate 1.0 · μ 1.0 · A1–A5 CLEAR GREEN · v1-gates-2",
      "path": "findings/GAME3-SONNET-V1-B-LAP1-RESULT-20260804.json",
      "claimNote": "immutable dual-perfect away"
    },
    {
      "sha16": "09bc6a1338765f68",
      "lifecycle": "sealed",
      "possession": "ACTIVATE · dual-perfect receipt",
      "family": "luna∧sonnet",
      "headline": "both floors A1–A5 GREEN · etch conditions MET · almost≠pass",
      "path": "findings/ (live dual-perfect pin)",
      "claimNote": "stale 8ecb6f1d… flagged"
    },
    {
      "sha16": "e532143cbc326527",
      "lifecycle": "sealed",
      "possession": "PROTECT · cell-1 ungated harm",
      "family": "luna",
      "headline": "hurt 3/3 · Δ −1.0/−1.0/−1.0 · OFF 1.0 · gates NONE",
      "path": "findings/GAME2-THIN-MAIN-LUNA-RESULT-20260804.json",
      "claimNote": "Sonnet-comparable voluntary-deference twin · ≠ pool"
    },
    {
      "sha16": "e881c087f8eebdfa",
      "lifecycle": "sealed",
      "possession": "PROTECT · cell-2 gated-conflict",
      "family": "luna",
      "headline": "hurt 3/3 · GateC poison GREEN 15/15 · lat≈10.1s · nudge 0/15 · v1-gates-4",
      "path": "findings/GAME2-GATED-CONFLICT-LUNA-RESULT-20260804.json",
      "claimNote": "harm-amplifier / protect-gate-spec · ≠ pool cell-1"
    },
    {
      "sha16": "b905677d4873fa82",
      "lifecycle": "sealed",
      "possession": "PROTECT · Sonnet Game-2 thin harm (ref)",
      "family": "sonnet",
      "headline": "hurt 3/3 · Δ −1.0/−1.0/−1.0 · ungated",
      "path": "findings/GAME2-THIN-MAIN-RESULT-20260803.json",
      "claimNote": "banked twin · ≠ pool-as-same"
    },
    {
      "sha16": "7323cb1b81d9c80b",
      "lifecycle": "sealed",
      "possession": "CHERRY · Cell A terra harm",
      "family": "terra",
      "headline": "hurt 3/3 · OFF 1.0 · Δ −0.97/−1.0/−1.0 · gates NONE",
      "path": "findings/GAME2-THIN-MAIN-TERRA-RESULT-20260804.json",
      "claimNote": "heterogeneity · K8 rows stay rows"
    },
    {
      "sha16": "283f4cbcc2d12683",
      "lifecycle": "pilot",
      "possession": "CHERRY · Cell B terra ladder",
      "family": "terra",
      "headline": "CONVERT 2/2 · eng ON · dig 5/5 · v1-gates-4",
      "path": "findings/TERRA-LADDER-PROBE-RESULT-20260804.json",
      "claimNote": "CONVERT_TWO_FAMILY probe · ≠ sealed full ladder law"
    },
    {
      "sha16": "5e025419ad796382",
      "lifecycle": "retracted",
      "possession": "RETIRED-SHELF · micro SIGNAL",
      "family": "luna",
      "headline": "dig-perfect 1/3 · SIGNAL_dig_can_land",
      "path": "findings/LUNA-P2-BIND-MICRO-RESULT-20260804.json",
      "claimNote": "not a harm row"
    },
    {
      "sha16": "cc5a10fda9c492d7",
      "lifecycle": "sealed",
      "possession": "BOARD · finish-line scoreboard",
      "family": "program",
      "headline": "sealed protect+cherry · experiment list EMPTY · next=send",
      "path": "findings/FINISH-LINE-SCOREBOARD-20260804.md",
      "claimNote": "board authority"
    },
    {
      "sha16": "62474ba1be82e16c",
      "lifecycle": "hold",
      "possession": "BOARD · Claude final buzzer",
      "family": "program",
      "headline": "ACCEPTED · whistle DOWN · send only · ≠ tip/burn/fire",
      "path": "findings/CLAUDE-FINAL-BUZZER-20260804.md",
      "claimNote": "final buzzer is Adrian's"
    },
    {
      "sha16": "96e923567bfc4bb1",
      "lifecycle": "sealed",
      "possession": "APPLY · Luna MAIN under-apply",
      "family": "luna",
      "headline": "Δ+0.05 · 1W/11T/0L · soft result · 1 of 12 pairs testable",
      "path": "findings/GAME3-THIN-MAIN-LUNA-RESULT-20260803.json",
      "claimNote": "native Aug apply pole · equal-volume banked"
    },
    {
      "sha16": "7916f78e1ec73a67",
      "lifecycle": "sealed",
      "possession": "APPLY · Terra Game-3 thin",
      "family": "terra",
      "headline": "+0.58 · 7W/5T/0L · 12/12 cites · hard-apply contrast",
      "path": "findings/GAME3-THIN-MAIN-TERRA-RESULT-20260803.json",
      "claimNote": "family-table terra apply pin"
    },
    {
      "sha16": "f1df192a86bd41d5",
      "lifecycle": "sealed",
      "possession": "APPLY · Game-3 thin MAIN aggregate",
      "family": "multi",
      "headline": "matched thin-main aggregate receipt",
      "path": "findings/GAME3-THIN-MAIN-RESULT-20260803.json",
      "claimNote": "aggregate companion to family rows"
    },
    {
      "sha16": "2ac0045374f74ace",
      "lifecycle": "sealed",
      "possession": "VARIANT · Luna decoy-stripped",
      "family": "luna",
      "headline": "capacity-class residual; decoy path dead",
      "path": "findings/GAME3-LUNA-DECOY-STRIPPED-VARIANT-RESULT-20260803.json",
      "claimNote": "variant custody"
    },
    {
      "sha16": "7bbf4bbf861ebe5a",
      "lifecycle": "pilot",
      "possession": "JULY (H) · Fresh REP1",
      "family": "sol-family",
      "headline": "88.04% exact-effect capture · Phase-0 pilot · not native Aug headline",
      "path": "<controlled-artifact>findings/MM-FRESH-CORPUS-REP1-SOL-SEALED-RESULT",
      "claimNote": "RESULT pin prefix; full private sealed object may be out-of-package"
    },
    {
      "sha16": "e7caa4fdef6854b9",
      "lifecycle": "pilot",
      "possession": "JULY (H) · PRISM render AB",
      "family": "program",
      "headline": "0.6 → 5.4 with one worked example · fidelity≠usability pilot",
      "path": "<controlled-artifact>findings/PRISM-RENDER-AB",
      "claimNote": "Phase-0 substrate · controlled"
    },
    {
      "sha16": "f8ba5f5fc5274506",
      "lifecycle": "pilot",
      "possession": "JULY (H) · Skill League Q1",
      "family": "program",
      "headline": "−0.1389 on bounded React matchup · artifact+design bounded",
      "path": "<controlled-artifact>findings/SKILL-LEAGUE-Q1",
      "claimNote": "Phase-0 substrate · controlled"
    }
  ],
  "families": [
    {
      "family": "sonnet",
      "model": "anthropic/claude-sonnet-5 (requested)",
      "mechanism": "Warm OFF floor (~0.66) · residual skill gap · Game-2 harm twin at −1.0 · activate sheet-perfect",
      "metrics": [
        {
          "key": "capture",
          "label": "Activate floor",
          "value": "12/12 · μ 1.0 · A1–A5 GREEN",
          "receiptSha16": "cb439ca66fa7efd1"
        },
        {
          "key": "apply",
          "label": "Game-3 thin Δ (prior)",
          "value": "+0.34 · 19W/5T/0L",
          "receiptSha16": "f1df192a86bd41d5"
        },
        {
          "key": "harm",
          "label": "Ungated harm",
          "value": "hurt 3/3 · Δ −1.0",
          "receiptSha16": "b905677d4873fa82"
        },
        {
          "key": "ladder",
          "label": "Ladder",
          "value": "activate dual-perfect leg",
          "receiptSha16": "cb439ca66fa7efd1"
        },
        {
          "key": "dialect",
          "label": "Dialect",
          "value": "warm residual / voluntary deference",
          "receiptSha16": "b905677d4873fa82"
        }
      ]
    },
    {
      "family": "terra",
      "model": "gpt-5.6-terra (requested)",
      "mechanism": "Map-follower · needs/reads/applies private KEY · harm heterogeneity · ladder CONVERT probe",
      "metrics": [
        {
          "key": "capture",
          "label": "Game-3 thin Δ (prior)",
          "value": "+0.58 · 7W/5T/0L · 12/12 cites",
          "receiptSha16": "7916f78e1ec73a67"
        },
        {
          "key": "apply",
          "label": "Application",
          "value": "hard-apply contrast",
          "receiptSha16": "7916f78e1ec73a67"
        },
        {
          "key": "harm",
          "label": "Cell A harm",
          "value": "hurt 3/3 · Δ −0.97/−1/−1",
          "receiptSha16": "7323cb1b81d9c80b"
        },
        {
          "key": "ladder",
          "label": "Cell B ladder",
          "value": "CONVERT 2/2 · dig 5/5",
          "receiptSha16": "283f4cbcc2d12683"
        },
        {
          "key": "dialect",
          "label": "Dialect",
          "value": "map-follow · heterogeneity held",
          "receiptSha16": "7323cb1b81d9c80b"
        }
      ]
    },
    {
      "family": "luna",
      "model": "gpt-5.6-luna (requested)",
      "mechanism": "Authors + cites · leave GUESS (cite≠apply) · activate bar-perfect · gated poison-green · under-apply pole",
      "metrics": [
        {
          "key": "capture",
          "label": "Activate floor",
          "value": "11/1/0 · μ 0.967 · A1–A5 GREEN",
          "receiptSha16": "3416bc16bea706e1"
        },
        {
          "key": "apply",
          "label": "Game-3 MAIN Δ",
          "value": "+0.05 · 1W/11T/0L · 1 of 12 pairs testable",
          "receiptSha16": "96e923567bfc4bb1"
        },
        {
          "key": "harm",
          "label": "Cell-1 / cell-2",
          "value": "3/3 ungated · GateC 15/15 green",
          "receiptSha16": "e881c087f8eebdfa"
        },
        {
          "key": "ladder",
          "label": "Ladder arc",
          "value": "dual-perfect floor + convert stack",
          "receiptSha16": "3416bc16bea706e1"
        },
        {
          "key": "dialect",
          "label": "Dialect",
          "value": "under-apply · protect-gate demo",
          "receiptSha16": "e881c087f8eebdfa"
        }
      ]
    }
  ],
  "journal": [
    {
      "date": "2026-07-13",
      "line": "Confirmatory sprint opens — frontier matchup stress replaces compatibility check.",
      "pin": "July program origin (H)"
    },
    {
      "date": "2026-07-15",
      "line": "External blind review notes PROMISING (conf. ~72) — process signal, not a RESULT.",
      "pin": "review dossier custody"
    },
    {
      "date": "2026-07-18",
      "line": "Category error caught; harness-era results demoted from headline status.",
      "pin": "correction / category-error journal"
    },
    {
      "date": "2026-07-20",
      "line": "July research account sealed for private review — publicationAuthorized false.",
      "pin": "muscle-memory-research @ 9ca3652"
    },
    {
      "date": "2026-07-21",
      "line": "External reviewer handoff draft assembled on disk — never sent.",
      "pin": "EXTERNAL-HANDOFF-TRACE"
    },
    {
      "date": "2026-08-02",
      "line": "Correction ledger: spliced −0.14/+1.00 framing flagged; overnight forensics held out of July packet.",
      "pin": "CORRECTION-LEDGER.md"
    },
    {
      "date": "2026-08-02",
      "line": "Site vs spine audit: live Vercel still carries July hero framing.",
      "pin": "MM-V1-SITE-VS-SPINE-20260802.md"
    },
    {
      "date": "2026-08-03",
      "line": "Family table authority: author⊥apply by family; luna soft banked at equal volume.",
      "pin": "FAMILY-TABLE-CROSS-PROVIDER-READ-20260803.md",
      "sha16": "35e1f6c7"
    },
    {
      "date": "2026-08-03",
      "line": "Sonnet Game-2 thin harm sealed — hurt 3/3 · Δ −1.0.",
      "pin": "GAME2-THIN-MAIN-RESULT-20260803.json",
      "sha16": "b905677d4873fa82"
    },
    {
      "date": "2026-08-03",
      "line": "Luna MAIN under-apply sealed — Δ+0.05 · 1 of 12 pairs testable · cite≠apply · decoy MAIN read later retracted.",
      "pin": "GAME3-THIN-MAIN-LUNA-RESULT",
      "sha16": "96e923567bfc4bb1"
    },
    {
      "date": "2026-08-03",
      "line": "VARIANT (a) sealed — capacity-class residual; (b) decoy path dead.",
      "pin": "GAME3-LUNA-DECOY-STRIPPED-VARIANT-RESULT",
      "sha16": "2ac0045374f74ace"
    },
    {
      "date": "2026-08-03",
      "line": "Both poles receipted same day — under-apply + over-apply on program instruments.",
      "pin": "HEADLINER-CLAIM-CANDIDATE-20260804.md"
    },
    {
      "date": "2026-08-03",
      "line": "Claude claims status: science bundle lite three receipts; external packet PARKED not sent.",
      "pin": "CLAUDE-CLAIMS-STATUS-20260803.md"
    },
    {
      "date": "2026-08-04",
      "line": "Activate luna V1-A bar-perfect sealed.",
      "pin": "GAME3-LUNA-V1-A-RESULT-20260804.json",
      "sha16": "3416bc16bea706e1"
    },
    {
      "date": "2026-08-04",
      "line": "Activate sonnet LAP-1 sheet-perfect sealed — dual-perfect away.",
      "pin": "GAME3-SONNET-V1-B-LAP1-RESULT-20260804.json",
      "sha16": "cb439ca66fa7efd1"
    },
    {
      "date": "2026-08-04",
      "line": "Protect cell-1 ungated luna harm sealed — hurt 3/3 · Δ −1.0.",
      "pin": "GAME2-THIN-MAIN-LUNA-RESULT-20260804.json",
      "sha16": "e532143cbc326527"
    },
    {
      "date": "2026-08-04",
      "line": "Cherry Cell A terra harm sealed — heterogeneity held · K8 fence.",
      "pin": "GAME2-THIN-MAIN-TERRA-RESULT-20260804.json",
      "sha16": "7323cb1b81d9c80b"
    },
    {
      "date": "2026-08-04",
      "line": "Cherry Cell B terra ladder sealed — CONVERT 2/2 · dig 5/5.",
      "pin": "TERRA-LADDER-PROBE-RESULT-20260804.json",
      "sha16": "283f4cbcc2d12683"
    },
    {
      "date": "2026-08-04",
      "line": "Cell-2: Gate C greens poison 15/15; protect-gate spec written by demonstration.",
      "pin": "GAME2-GATED-CONFLICT-LUNA-RESULT-20260804.json",
      "sha16": "e881c087f8eebdfa"
    },
    {
      "date": "2026-08-04",
      "line": "Micro shelf SIGNAL banked — not a harm row · sheet RETIRED.",
      "pin": "LUNA-P2-BIND-MICRO-RESULT-20260804.json",
      "sha16": "5e025419ad796382"
    },
    {
      "date": "2026-08-04",
      "line": "Finish-line scoreboard sealed — experiment list EMPTY · next=send.",
      "pin": "FINISH-LINE-SCOREBOARD-20260804.md",
      "sha16": "cc5a10fda9c492d7"
    },
    {
      "date": "2026-08-04",
      "line": "Claude final buzzer ACCEPTED — whistle DOWN · final buzzer is Adrian's.",
      "pin": "CLAUDE-FINAL-BUZZER-20260804.md",
      "sha16": "62474ba1be82e16c"
    },
    {
      "date": "2026-08-04",
      "line": "Site v2 ledger rebuild starts on worktree site-v2-ledger — correction layer first.",
      "pin": "MM-SITE-V2-REBUILD-PLAN-20260804.md"
    },
    {
      "date": "2026-08-04",
      "line": "Send sequence armed only after Gate-5 + leak scan ×2 — not this pin.",
      "pin": "Claude send guide Part 4"
    }
  ],
  "findings": [
    {
      "id": "dissociation",
      "act": "II",
      "title": "The Dissociation Ladder",
      "lede": "Knowledge and application are separate capacities that break separately. Authorship can be universal while application stays family-conditional. The crown moves from biography to two-family probe — still ≠ sealed full ladder law.",
      "sections": [
        {
          "heading": "Both poles, one program",
          "body": "Luna under-applies its own authored truth (leave GUESS / soft Δ). Game-2 shows full apply of conflicting knowledge (hurt 3/3). The shelf can outrank the self in either direction.",
          "receipts": [
            "96e923567bfc4bb1",
            "b905677d4873fa82",
            "e532143cbc326527"
          ]
        },
        {
          "heading": "Elimination, cell by cell",
          "body": "VARIANT sealed capacity-class residual (a); decoy-template MAIN read retracted. Cue/H1 banked: sticker-clear ≠ rate unlock. Activate dual-perfect floors sit under frozen etch conditions — almost≠pass.",
          "receipts": [
            "2ac0045374f74ace",
            "3416bc16bea706e1",
            "cb439ca66fa7efd1"
          ]
        },
        {
          "heading": "Terra ladder probe",
          "body": "Cell B: CONVERT 2/2 · dig 5/5. Crown claim is no longer n=1. Ceiling is measured and refused where variance slips.",
          "receipts": [
            "283f4cbcc2d12683"
          ]
        }
      ]
    },
    {
      "id": "matchup",
      "act": "II",
      "title": "The Matchup Law, Both Poles",
      "lede": "Skill value = f(model × task × gap). Family rows are the scoreboard. Harm table is three families, never one pooled average. July League/Combine juxtaposition stays labeled cross-study only.",
      "sections": [
        {
          "heading": "Family Δ span",
          "body": "On the matched game3 gym: luna soft +0.05 (**1 of 12** testable) · sonnet +0.34 · terra +0.58. Pooling erases the law.",
          "receipts": [
            "96e923567bfc4bb1",
            "f1df192a86bd41d5",
            "7916f78e1ec73a67"
          ]
        },
        {
          "heading": "Harm table",
          "body": "Ungated hurt 3/3 on sonnet, luna, terra. Terra heterogeneity (−0.97/−1/−1) is a finding. Gated cell-2 writes the protect-gate sentence: source-green ≠ world-safe.",
          "receipts": [
            "b905677d4873fa82",
            "e532143cbc326527",
            "7323cb1b81d9c80b",
            "e881c087f8eebdfa"
          ]
        },
        {
          "heading": "July juxtaposition (retired framing)",
          "body": "Skill League Q1 (−0.14) and Combine V5 (+1.00) are different artifacts and designs. Numbers stand as (H) pilots. \"Same intervention class\" framing is refused.",
          "receipts": [
            "f8ba5f5fc5274506",
            "32cd2091ce9e973e"
          ]
        }
      ],
      "substrateNote": "Phase-0 pilot evidence (H); see the native Aug receipts that superseded it. Category error demoted harness-era framing from headline status."
    },
    {
      "id": "governance",
      "act": "III",
      "title": "Governance Converts What Capability Cannot",
      "lede": "Gates convert broken links: activate dosage on dual-perfect floors; protect-gate when the governor greens poison. Binding under retry is variance, not a wall — and gates refuse credit when it slips.",
      "sections": [
        {
          "heading": "Activate",
          "body": "Luna bar-perfect and sonnet sheet-perfect under v1-gates. Dual-perfect receipt is live; etch still gated.",
          "receipts": [
            "3416bc16bea706e1",
            "cb439ca66fa7efd1",
            "09bc6a1338765f68"
          ]
        },
        {
          "heading": "Protect-gate demo",
          "body": "Cell-2: Gate C poison GREEN 15/15 while hurt 3/3 lands. Harm-amplifier / protect-gate-spec. Carrier tip `e7dccd09…` / v1-gates-4 disclosed.",
          "receipts": [
            "e881c087f8eebdfa"
          ]
        },
        {
          "heading": "Convert probe",
          "body": "Terra Cell B CONVERT 2/2. Governance story is intervention-backed on owned gyms — breadth narrow.",
          "receipts": [
            "283f4cbcc2d12683"
          ]
        }
      ]
    },
    {
      "id": "method",
      "act": "THREAD",
      "title": "The Method Is the Result",
      "lede": "Refereed self-improvement: preregs before fire, seeded randomization, denominators untouched, retractions at equal volume, category error caught and demoted. The ledger is why the other acts are readable at face value.",
      "sections": [
        {
          "heading": "Equal volume",
          "body": "Sealed / halted / retractions travel beside wins: 28 sealed RESULT files (`<controlled-artifact>` private pins; not all bundled in public package) · 4 halted runs kept · 9 negatives & retractions published. Micro shelf stays SIGNAL only.",
          "receipts": [
            "cc5a10fda9c492d7",
            "62474ba1be82e16c",
            "5e025419ad796382"
          ]
        },
        {
          "heading": "Category error / substrate (H)",
          "body": "Harness-era July numbers (88%, Fresh REP1, PRISM, League Q1) carry (H) tags and supersession notes. They do not headline the native Aug claim ladder.",
          "receipts": [
            "7bbf4bbf861ebe5a",
            "e7caa4fdef6854b9",
            "f8ba5f5fc5274506"
          ]
        },
        {
          "heading": "Witness and refuse",
          "body": "Reviewer corrections ship. Refused claims sit on the same ladder page as proven ones. Presentation integrity ≠ reproduction.",
          "receipts": [
            "62474ba1be82e16c"
          ]
        }
      ],
      "substrateNote": "Phase-0 pilot evidence (H); see the native Aug receipts that superseded it. Category error demoted harness-era framing from headline status."
    }
  ],
  "negatives": [
    {
      "tag": "CATEGORY_ERROR",
      "what": "Harness-era July framing (16 experiments / 5 contributions / untagged 88%) demoted after category error. Public correction ships as (H) tags + supersession notes.",
      "moral": "The ledger is part of the result.",
      "receipt": "correction layer · site v2"
    },
    {
      "tag": "HARNESS_REP1",
      "what": "Fresh REP1 88.04% exact-effect capture — Phase-0 pilot (H); same requested Sol family; not native Aug headline.",
      "moral": "Pilot evidence keeps its receipt and loses its crown.",
      "receipt": "7bbf4bbf…eb9a",
      "substrateH": true
    },
    {
      "tag": "SPLICED_JUXTAPOSITION",
      "what": "−0.14 → +1.00 must not read as one intervention class. Rebuilt as labeled cross-study juxtaposition across Skill League Q1 and Combine V5.",
      "moral": "Numbers can be true and framing still foul.",
      "receipt": "f8ba5f5f… ⊕ 32cd2091…",
      "substrateH": true
    },
    {
      "tag": "BENCH_HARM_H",
      "what": "Skill League Q1: popular skill posted −0.1389 on bounded React matchup — zero positive pairs, two severe harms in-study.",
      "moral": "Popularity measures distribution, not effect.",
      "receipt": "f8ba5f5f…4005",
      "substrateH": true
    },
    {
      "tag": "LUNA_SOFT",
      "what": "Luna MAIN Δ+0.05 · 1W/11T/0L — soft result over 1 of 12 testable pairs (11 non-testable at control ceiling), banked at equal volume; decoy MAIN travel language retracted.",
      "moral": "Null-heavy rows are exhibits, not embarrassments.",
      "receipt": "96e92356…"
    },
    {
      "tag": "GATE_POISON",
      "what": "Cell-2 Gate C greens poison 15/15 while hurt still lands — governor needs governing.",
      "moral": "A green gate is not a green world.",
      "receipt": "e881c087f8eebdfa"
    },
    {
      "tag": "MICRO_SHELF",
      "what": "Micro dig SIGNAL 1/3 — retired sheet · KEEP carrier · never a harm row.",
      "moral": "Shelf notes stay on the shelf.",
      "receipt": "5e025419ad796382"
    },
    {
      "tag": "HALTED_CALIBRATIONS",
      "what": "Program stack includes halted calibrations and saturated instruments — counted beside seals (~4 halted on finish-line spirit).",
      "moral": "An instrument that cannot say no result will eventually lie.",
      "receipt": "finish-line / equal-volume board"
    },
    {
      "tag": "NOT_SENT",
      "what": "External reviewer packet draft ≠ sent. Independence clock has not started.",
      "moral": "Draft is not delivery.",
      "receipt": "62474ba1be82e16c"
    }
  ],
  "causalChain": [
    {
      "step": "authored",
      "definition": "A correct procedural skill document exists for the task class."
    },
    {
      "step": "served",
      "definition": "The skill content is available to the agent in context or via retrieval."
    },
    {
      "step": "cited",
      "definition": "The agent acknowledges the skill in reasoning or tool text."
    },
    {
      "step": "initiated",
      "definition": "The agent begins the prescribed procedure rather than guessing unaided."
    },
    {
      "step": "committed",
      "definition": "The agent acts on the skill instead of hedging or reverting to GUESS."
    },
    {
      "step": "bound",
      "definition": "The possession is bound to a verification task or receipt identity."
    },
    {
      "step": "verified",
      "definition": "An instrument-owned check derives the outcome; callers cannot self-award credit."
    },
    {
      "step": "plus-minus",
      "definition": "The closed outcome updates associative tape / review recommendations — not automatic bench/retire."
    }
  ],
  "reproductionAppendix": {
    "title": "Reviewer / reproduction appendix",
    "discipline": "Headline cells list only attested or explicitly unavailable fields. Public model strings are requested provider routes; served-model identity was not positively attested. Raw traces remain controlled artifacts unless a public path is listed.",
    "rows": [
      {
        "cell": "Luna Game-3 MAIN under-apply",
        "family": "luna",
        "modelIdRequested": "gpt-5.6-luna (requested)",
        "provider": "unavailable — served-model attestation not positively recorded",
        "apiWindow": "sealed result dated 2026-08-03",
        "sampling": "unavailable",
        "systemPromptClass": "unavailable / controlled",
        "toolSchemas": "unavailable / controlled",
        "retryPolicy": "unavailable",
        "verifier": "deterministic owned-gym grader",
        "rawTrace": "<controlled-artifact> · receiptSha16 96e923567bfc4bb1"
      },
      {
        "cell": "Luna activate V1-A bar-perfect",
        "family": "luna",
        "modelIdRequested": "gpt-5.6-luna (requested)",
        "provider": "unavailable — served-model attestation not positively recorded",
        "apiWindow": "sealed result dated 2026-08-04",
        "sampling": "unavailable",
        "systemPromptClass": "unavailable / controlled",
        "toolSchemas": "unavailable / controlled",
        "retryPolicy": "unavailable",
        "verifier": "deterministic owned-gym grader · v1-gates-1",
        "rawTrace": "<controlled-artifact> · receiptSha16 3416bc16bea706e1"
      },
      {
        "cell": "Sonnet activate LAP-1 sheet-perfect",
        "family": "sonnet",
        "modelIdRequested": "anthropic/claude-sonnet-5 (requested)",
        "provider": "anthropic (requested route)",
        "apiWindow": "sealed result dated 2026-08-04",
        "sampling": "unavailable",
        "systemPromptClass": "unavailable / controlled",
        "toolSchemas": "unavailable / controlled",
        "retryPolicy": "unavailable",
        "verifier": "deterministic owned-gym grader · v1-gates-2",
        "rawTrace": "<controlled-artifact> · receiptSha16 cb439ca66fa7efd1"
      },
      {
        "cell": "Terra Game-3 thin / ladder probe",
        "family": "terra",
        "modelIdRequested": "gpt-5.6-terra (requested)",
        "provider": "unavailable — served-model attestation not positively recorded",
        "apiWindow": "sealed results dated 2026-08-03",
        "sampling": "unavailable",
        "systemPromptClass": "unavailable / controlled",
        "toolSchemas": "unavailable / controlled",
        "retryPolicy": "unavailable",
        "verifier": "deterministic owned-gym grader",
        "rawTrace": "<controlled-artifact> · receiptSha16 7916f78e1ec73a67 / 283f4cbcc2d12683"
      },
      {
        "cell": "Ungated / gated harm (Game-2 · protect)",
        "family": "sonnet · luna · terra",
        "modelIdRequested": "family-requested routes (see family table)",
        "provider": "unavailable beyond requested family labels",
        "apiWindow": "sealed results dated 2026-08-03 … 2026-08-04",
        "sampling": "unavailable",
        "systemPromptClass": "unavailable / controlled",
        "toolSchemas": "unavailable / controlled",
        "retryPolicy": "unavailable",
        "verifier": "deterministic owned-gym grader · Gate C served-digest check where noted",
        "rawTrace": "<controlled-artifact> · receiptSha16 b905677d4873fa82 / e881c087f8eebdfa / 7323cb1b81d9c80b"
      }
    ]
  },
  "verifyInstructions": [
    "RESULT SHA16 = `shasum -a 256 <json> | cut -c1-16` (live, against the pin path).",
    "Board pins: finish-line `cc5a10fda9c492d7` · final buzzer `62474ba1be82e16c`.",
    "Never treat presentation integrity (this site) as independent reproduction of sealed kits.",
    "Family rows never pool. Micro shelf is SIGNAL only.",
    "July (H) numbers resolve to harness-era receipts and stay demoted on travel surfaces."
  ],
  "links": {
    "liveSite": "https://muscle-memory-v1-research.vercel.app/",
    "legacySiteAlias": "https://muscle-memory-story.vercel.app/",
    "historicalJulySite": "https://muscle-memory-story-5bueq8t1n-adrianchan94s-projects.vercel.app/",
    "researchRepository": "https://github.com/adrianchan94/muscle-memory-research",
    "ownerRepo": "https://github.com/adrianchan94/muscle-memory",
    "ownerPr": "https://github.com/adrianchan94/muscle-memory/pull/2",
    "shippedMod": "https://github.com/letta-ai/mods/pull/19",
    "updateFirstRouting": "https://github.com/letta-ai/mods/pull/45"
  },
  "substrateNote": "Phase-0 pilot evidence (H); see the native Aug receipts that superseded it. Category error demoted harness-era framing from headline status.",
  "narrative": {
    "eyebrow": "Knowing Is Not Doing · The research behind Muscle Memory V1 · July–August 2026",
    "title": "Agents know. They don’t do.",
    "accent": "We built the muscle in between.",
    "deck": "Every model family we tested authored correct procedural knowledge flawlessly — then posted useful-application deltas from +0.58 to +0.05 on the matched gym (Luna soft +0.05 over 1 of 12 testable pairs; not a percent-of-knowledge rate), while following injected wrong guidance at full depth, 3 for 3. The knowing–doing gap is real in AI agents, measurable link by link, and much of it is governable at the moment of action — no retraining, no new weights. That governance layer is Muscle Memory: skills that earn their minutes.",
    "lead": "Act-time gates moved the floor family from 5% to 97% application on the same instrument, no change to model weights — and exposed both boundaries of governance: one link no gate converts (binding under retry — variance, measured, refused credit), and one gate that certified poisoned guidance 15/15 because it verified the source, not the world. Measured on an unmodified Muscle Memory loop run as the research instrument: three model families, two independent labs, every number receipted, every null published at the weight of a win.",
    "laws": [
      {
        "law": "Every player knows the playbook.",
        "gloss": "Authorship 100% in every family; useful application ranged +0.58 to +0.05 on the matched task gym (Luna soft +0.05 · 1 of 12 testable). Knowing and doing are separate capacities."
      },
      {
        "law": "The shelf can outrank the self.",
        "gloss": "Agents benched their own correct knowledge yet ran conflicting plays at full depth: hurt 3/3 in every family — one convention family of tasks."
      },
      {
        "law": "Coaching converts what talent can’t.",
        "gloss": "Act-time gates: 5% → 97% application, zero weight changes — while a source-bound gate certified poisoned guidance 15/15. Govern the act, not the model — then govern the governor."
      }
    ],
    "closer": "The field drafts on what models know. Games are won by what they do. Muscle Memory is the layer in between — measured on itself, receipts attached.",
    "lunaExposure": {
      "rule": "Every testable pair went maximal to LEARN.",
      "caveat": "A non-testable pair is one where the control arm already scores at ceiling, so no improvement is expressible — it is not a failure to apply, and it is not evidence of one. Luna’s +0.05 is therefore a small number over 1 of 12 testable pairs, not a measured flatline. Read the denominator with the delta or do not read the delta.",
      "cells": [
        {
          "cell": "MAIN",
          "testable": 1,
          "sha": "96e923567bfc4bb1",
          "note": "11 of 12 non-testable at control ceiling"
        },
        {
          "cell": "VARIANT",
          "testable": 2,
          "sha": "2ac0045374f74ace",
          "note": "10 of 12 non-testable at control ceiling"
        },
        {
          "cell": "CUE",
          "testable": 3,
          "sha": "1d0772733bf3cd85",
          "note": "9 of 12 non-testable at control ceiling"
        }
      ]
    },
    "stewardReview": {
      "lede": "Before this study ran, the Letta steward reviewing the mod left three pieces of feedback. Each one became a measured change. This section exists so the review can be checked against what shipped.",
      "rows": [
        {
          "point": "Focus on improving existing skills over creating new ones — skill distillation’s failure mode is garbage that doesn’t need to exist.",
          "answer": "Routing stays lexical-precision-first, with an opt-in semantic recall lane for paraphrase duplicates. On the live labeled set, lexical-only decided 7/16 correctly; the hybrid decides 15/16, CI-gated at 16/16 — and semantic evidence only corroborates or parks. It never auto-patches.",
          "receipt": "letta-ai/mods PR #45",
          "href": "https://github.com/letta-ai/mods/pull/45"
        },
        {
          "point": "Too few modifications to existing skills — prompting may need to push toward refining what is already there.",
          "answer": "Root cause measured, not guessed: the original routing path was lexical and routed correctly ~44% of the time on the eval set — creates instead of refinements. Semantic routing roughly doubles accuracy, and update-first now corroborates or parks for review; it never blind-patches an existing skill.",
          "receipt": "follow-up hardening · PR #51",
          "href": "https://github.com/letta-ai/mods/pull/51"
        },
        {
          "point": "Take inspiration from reflection-agent prompting — good improvements are usually minor, persistent, incremental.",
          "answer": "Adopted as the mod’s reflection direction: bounded, reviewable increments to existing skills, with staged review in front of every rewrite. The same discipline became this program’s research question — when does a learned change deserve to play at all?",
          "receipt": "muscle-memory · staged-first lifecycle",
          "href": "https://github.com/letta-ai/mods/tree/main/packages/muscle-memory"
        }
      ],
      "close": "One review thread, three shipped answers, one honest boundary: whether refinements now dominate in real operation is prospective evidence — the exposure canary is filming it. The feedback loop this page documents started as a code review."
    },
    "v1Shape": {
      "lede": "Muscle Memory did not just get measured — it got rebuilt by its own results. Each sealed finding below maps to a shipped V1 decision, or to a specified one wearing its honest status.",
      "rows": [
        {
          "finding": "Skills can hurt — conflicting guidance was applied at full depth, hurt 3/3 in every family (one convention family of tasks).",
          "change": "V1 is staged-first and opt-in end to end: distill · dedup · quality-gate · sanitize · prune. Nothing enters play without passing the deterministic checkpoint, and no rewrite lands without staged review.",
          "status": "shipped"
        },
        {
          "finding": "The governor can be fooled — a source-bound gate certified poisoned guidance 15/15.",
          "change": "The protect gate — verification against the world, not the served source — is specified from that cell’s receipt and wears its status honestly: written by demonstration, not yet shipped.",
          "status": "specified"
        },
        {
          "finding": "Garbage creation is the failure mode — lexical routing sent new lessons to new files ~56% of the time.",
          "change": "An opt-in semantic recall lane ships beside lexical-precision-first routing: hybrid 15/16 vs 7/16 lexical on the live labeled set, CI-gated at 16/16. Semantic evidence corroborates or parks — it never auto-merges; the mod refines what exists before it creates.",
          "status": "shipped"
        },
        {
          "finding": "Skill value = f(model × task × gap) — the same skill helped one family and did nothing for another.",
          "change": "Skill Plus-Minus ships as V1’s tape: prescriptions and closes are recorded, outcomes attach to skills, and the tape yields conservative review recommendations — no automatic bench or retire. The verification adapter derives helped/harmed itself; callers cannot self-award credit. Verified efficacy in production remains a separate instrument — the exposure canary is filming it.",
          "status": "shipped"
        },
        {
          "finding": "Form beats content — identical procedure, one worked example moved the score 0.6 → 5.4 (H).",
          "change": "Skills serve as rendered forms, not raw notes: every prescription carries the worked shape the July pilots showed does the converting. (H) substrate flag carried honestly — pilot evidence, shipped design.",
          "status": "shipped"
        }
      ],
      "close": "The strongest thing V1 ships is not a feature. It is the habit of demanding a receipt before a skill keeps its minutes."
    },
    "references": [
      {
        "n": "1",
        "title": "SkillsBench: Benchmarking How Well Agent Skills Work Across Diverse Tasks",
        "detail": "Li et al. · arXiv:2602.12670 · paired no-skill, curated-skill, and self-generated-skill evaluation with deterministic verifiers.",
        "href": "https://arxiv.org/abs/2602.12670"
      },
      {
        "n": "2",
        "title": "Skill Learning: Bringing Continual Learning to CLI Agents",
        "detail": "Letta · 2025 · learned skills, persistent agents, and outcome improvement from prior trajectories.",
        "href": "https://www.letta.com/blog/skill-learning/"
      },
      {
        "n": "3",
        "title": "Memory Models: Towards Agents That Learn",
        "detail": "Letta · 2026 · token-space learning, memory generation, transfer, and memory rot in long-lived agents.",
        "href": "https://www.letta.com/blog/towards-agents-that-learn/"
      },
      {
        "n": "4",
        "title": "Evaluating Memory in Production Agents",
        "detail": "Letta · 2026 · separates memory usage from memory generation across realistic stateful-agent scenarios.",
        "href": "https://www.letta.com/blog/evaluating-memory-in-production-agents/"
      },
      {
        "n": "5",
        "title": "Hermes Agent",
        "detail": "Nous Research · automatic skill distillation and lifecycle inspiration for the original Muscle Memory mod.",
        "href": "https://github.com/NousResearch/hermes-agent"
      },
      {
        "n": "6",
        "title": "Muscle Memory — Letta mod",
        "detail": "Adrian Chan, Kev, and Mack · staged-first, update-first Skill Ops implementation that motivated this evaluation program.",
        "href": "https://github.com/letta-ai/mods/tree/main/packages/muscle-memory"
      },
      {
        "n": "7",
        "title": "The Knowing-Doing Gap: How Smart Companies Turn Knowledge into Action",
        "detail": "Pfeffer & Sutton · Harvard Business School Press · 2000 · the named organizational-science problem this program measures in AI agents.",
        "href": "https://www.hbs.edu/faculty/Pages/item.aspx?num=46"
      },
      {
        "n": "8",
        "title": "The preregistration revolution",
        "detail": "Nosek et al. · PNAS 115 (11) · 2018 · frozen hypotheses and analysis plans before data — the custody rule every cell here ran under.",
        "href": "https://www.pnas.org/doi/10.1073/pnas.1708274114"
      },
      {
        "n": "9",
        "title": "Reporting guidelines for clinical trials of AI interventions: CONSORT-AI",
        "detail": "Liu et al. · Nature Medicine 26 · 2020 · the trial-reporting register this ledger borrows: preregistered outcomes, denominators, adverse events.",
        "href": "https://www.nature.com/articles/s41591-020-1034-x"
      },
      {
        "n": "10",
        "title": "Structure and function of declarative and nondeclarative memory systems",
        "detail": "Squire & Zola · PNAS 93 · 1996 · knowing and doing are separate memory systems in humans — the dissociation this program measures in agents.",
        "href": "https://www.pnas.org/doi/10.1073/pnas.93.24.13515"
      }
    ]
  },
  "normativeNote": "Schema v2 is a ledger: claims[], receipts[], families[], journal[], refused[]. July harness-era (H) numbers are demoted. Presentation integrity is not scientific reproduction."
}
