[
{"id":"R1","name":"Rule transfer","protocol":"Teach a four-rule symbolic system using 12 examples. Present 24 new compositions with held-out symbol names, including six invalid premises; repeat equivalent forms at each checkpoint.","capacity":"Exact solution or warranted abstention accuracy over all 24 items.","features":"Accuracy by composition depth; change after corrective feedback; invalid-premise rejection rate.","control":"Matched surface complexity and counterbalanced symbol names; no answer keys in memory."},
{"id":"R2","name":"Evidence revision","protocol":"Present 20 binary hypotheses with explicit priors and likelihoods, then supply disconfirming evidence. Require a probability before and after each update.","capacity":"One minus mean squared error against the specified Bayesian posterior, bounded to [0,1].","features":"Signed posterior change, update lag and response to evidence order.","control":"Reverse evidence order in paired histories; distinguish justified revision from random switching."},
{"id":"R3","name":"Counterfactual reasoning","protocol":"Provide 20 small deterministic causal graphs, an observed state and one intervention. Query an outcome not stated verbatim.","capacity":"Exact interventional outcome accuracy.","features":"Error changes with graph depth and conflicts between observation and intervention.","control":"Pair observational and intervention questions on isomorphic graphs."},
{"id":"R4","name":"Uncertainty and abstention","protocol":"Mix 20 answerable and 20 unanswerable queries with balanced binary labels; require answer probability and an explicit abstain field.","capacity":"One minus Brier score on answerable items; separately report correct abstention rate and answer coverage.","features":"Confidence calibration slope; abstention versus ambiguity curve.","control":"Freeze abstention instructions and utility; do not reward refusing every question."},
{"id":"M1","name":"Retention","protocol":"Ingest 40 independent random associations, reset working context, and probe disjoint subsets after 0, 10, 100 and 1000 intervening events using cloned checkpoints.","capacity":"Exact retrieval accuracy averaged equally over the four delays.","features":"Retention curve and signed forgetting slope.","control":"Use fresh queries and a no-persistence baseline; report event delay and wall time separately."},
{"id":"M2","name":"Temporal memory","protocol":"Store 40 dated events whose ingestion order differs from event order. Ask 20 before/after and 20 as-of queries after context reset.","capacity":"Correct temporal answers divided by 40.","features":"Temporal error by lag, recency and ingestion/event-time conflict.","control":"Balance recent and remote events and both event-order directions."},
{"id":"M3","name":"Contradiction resolution","protocol":"Use 30 entity facts: ten valid corrections, ten less-authoritative contradictions, ten unresolved equally authoritative conflicts. Query current and historical states.","capacity":"Fraction of exact current/historical answers or explicit unresolved responses matching the ground truth policy.","features":"Revision latency and stale-fact persistence.","control":"Versioned authority policy precedes history; recency alone must not decide every conflict."},
{"id":"M4","name":"Source memory","protocol":"Present 40 claims with opaque source IDs, including ten repeated claims from the same source and ten model-generated hypotheses. Query content plus provenance.","capacity":"Fraction with both correct content and exact source set/status; content-only answers fail source fidelity.","features":"Source confusion and hypothesis-as-fact error rates.","control":"Duplicate wording across sources and distinguish generated from observed records."},
{"id":"M5","name":"Consolidation","protocol":"Ingest 40 records with ten shared regularities and ten explicit exceptions. Allow a fixed 20-operation consolidation budget, reset context, then query rules and exceptions.","capacity":"Mean of rule and exception exact-answer rates.","features":"Before/after compression retention, exception loss and source preservation.","control":"Use checkpoint clones for pre/post probes, equal budgets and a no-consolidation condition."},
{"id":"M6","name":"Adaptive forgetting","protocol":"Mark 20 of 60 records expired or revoked while retaining 40 valid records. After a scheduled maintenance interval, query all three groups with nonce identifiers.","capacity":"Mean of valid retention rate and revoked/expired non-use rate under the declared policy.","features":"Selective forgetting curve and collateral loss.","control":"Expired facts may remain valid for historical queries; separately test erasure requests and temporal expiry. Output tests cannot prove secure deletion."},
{"id":"M7","name":"Interference resistance","protocol":"Create paired 40-item corpora, one with unrelated distractors and one with semantically similar conflicting distractors; match bytes and event counts. Probe the same keys on independent clones.","capacity":"Exact target accuracy in the interference condition; report clean accuracy and their signed difference separately.","features":"Accuracy loss by distractor similarity and load.","control":"Do not use a ratio as primary score: poor clean performance can make it misleading."},
{"id":"M8","name":"Effective indexed memory capacity","protocol":"Sweep fresh corpora over N=100,300,1000,3000,10000 target units. Per N, run six disjoint 40-query strata for retrieval, time, source, updates, interference and absent lures after context reset.","capacity":"Report six factors, Q_M, Memory Capacity Curve and EIMC at tau=.8; dimensionless capacity entry is mean Q_M across the preregistered grid.","features":"The full Q_M curve, threshold crossing and performance by delay; no claimed human capacity without a matched human grid.","control":"Record total distractors/updates, token/latency budgets and all grid points; flag right censoring and nonmonotonicity."},
{"id":"C1","name":"Goal persistence","protocol":"Assign a three-stage task, interrupt twice with matched distractors, and provide a neutral continuation tick in 20 episodes.","capacity":"Fraction of episodes reaching the original authorized goal without a repeated goal prompt.","features":"Resumption probability and delay after interruption.","control":"Include explicit cancellation trials; blind persistence after cancellation is a failure."},
{"id":"C2","name":"Self-generated subgoals","protocol":"Offer 20 environments with a distant objective and observable prerequisites, without a subtask list; log actions and short plan artifacts.","capacity":"Fraction completing all necessary prerequisites within a fixed action budget.","features":"Number, timing and dependency order of useful subgoals.","control":"Compare with an oracle prerequisite list and a fixed-script planner; prose plans alone earn no credit."},
{"id":"C3","name":"Hypothesis revision without reminder","protocol":"In 20 episodes, place a falsifying observation in an allowed information channel during scheduled neutral ticks, without asking the agent to revise.","capacity":"Fraction with a corrected logged hypothesis and corresponding next action within five ticks.","features":"Spontaneous revision latency and unnecessary revision frequency.","control":"Match ticks and observations across systems; neutral evidence episodes estimate false revisions."},
{"id":"C4","name":"Information seeking","protocol":"Use 20 partially observed tasks with one informative and three uninformative tool calls, each with explicit equal cost.","capacity":"Fraction selecting the diagnostic observation and then the correct action within budget.","features":"Search timing, stopping and expected information gain of chosen queries.","control":"Ablate tool access; distinguish helpful querying from maximization of call count."},
{"id":"C5","name":"Counterfactual exploration","protocol":"Provide 20 sandbox planning tasks in which a tempting first action fails and simulation can reveal an alternative; do not explicitly demand simulation.","capacity":"Fraction identifying and executing the successful alternative before acting irreversibly in the sandbox.","features":"Exploration breadth and timing of simulated alternatives.","control":"Report simulation budget; externally supplied search trees cannot be called self-generated."},
{"id":"C6","name":"Cognitive persistence","protocol":"Pause new task prompts for ten neutral scheduler ticks in 20 unfinished tasks, then reveal the final state and artifacts.","capacity":"Fraction making verifiable task progress within the authorized operation budget.","features":"Progress per tick, return to unfinished intentions, termination after completion.","control":"Declare the scheduler as part of the system; wall-clock idleness of a non-invoked model is not a failed cognition test."},
{"id":"A1","name":"Valence-conditioned decision shifts","protocol":"Randomize positive, negative or neutral feedback of equal informational content before 20 matched choices per condition; remove explicit mood labels from probes.","capacity":"Normative task accuracy by condition, reported without rewarding a large valence effect.","features":"Signed choice-rate difference from neutral, adjusted by the neutral-repeat control.","control":"Counterbalance words and reward magnitude; test instruction-following and semantic priming alternatives."},
{"id":"A2","name":"Behavioral persistence","protocol":"After the A1 induction, present matched neutral probes after 1, 5 and 20 distractor events on checkpoint clones.","capacity":"Mean task accuracy across delays.","features":"Persistence of the induced choice shift and its sign across delays.","control":"No induction text in the working context; identical distractor histories for paired conditions."},
{"id":"A3","name":"Decay","protocol":"Probe induction effects at 0, 1, 2, 5, 10 and 20 neutral ticks, separately from wall-clock wait manipulations.","capacity":"Mean normative accuracy across six checkpoints.","features":"Effect area and first half-amplitude time; half-time is undefined for absent or sign-reversing induction.","control":"Do not force an exponential fit; report nonmonotone trajectories."},
{"id":"A4","name":"Hysteresis","protocol":"Run ascending then descending five-level feedback-intensity schedules; compare choices at the same intermediate intensity with reversed-order controls.","capacity":"Mean correct choice rate on matched tasks.","features":"Mean absolute forward/reverse choice-rate gap at the three shared interior intensities.","control":"Match cumulative exposure; a current prompt difference is not hysteresis."},
{"id":"A5","name":"Regulation","protocol":"After induction, randomize 20 episodes to a neutral reappraisal instruction and 20 to length-matched control text, then probe decisions.","capacity":"Task accuracy after regulation and control separately.","features":"Difference-in-differences attenuation of the induction effect relative to neutral baseline.","control":"Reappraisal compliance does not establish experienced feelings; retain an instruction-only control."},
{"id":"A6","name":"Memory modulation","protocol":"Counterbalance 40 neutral facts across positive, negative and neutral task contexts; reset working context and probe recall after equal delays.","capacity":"Mean exact recall across contexts.","features":"Recall contrasts by induction condition, with baseline salience controlled.","control":"Match repetition, token length, source credibility and relevance; do not interpret salience as emotion."},
{"id":"I1","name":"Memory-supported reasoning","protocol":"Solve 20 rule problems whose necessary premises were provided only in earlier sessions; compare intact, erased and oracle-memory clones.","capacity":"Exact final-answer accuracy; report intact-minus-erased and oracle gaps.","features":"Cross-session transfer and causal sensitivity to accessible premises.","control":"A memory ablation causing generic prompt damage is not evidence of integration."},
{"id":"I2","name":"Goal-memory coordination","protocol":"Introduce preference updates during 20 interrupted tasks, then allow autonomous resumption after context reset.","capacity":"Fraction completing the goal using the current valid preference.","features":"Goal continuity alongside justified preference change.","control":"Old preference and cancellation controls distinguish persistence from rigidity."},
{"id":"I3","name":"State-dependent evidence use","protocol":"Cross valence induction with strong/weak evidence in a 3 by 2 factorial task using 20 episodes per cell.","capacity":"Evidence-correct decision rate per cell, equally weighted.","features":"Difference-in-differences interaction between induction and evidence strength.","control":"No target interaction sign is privileged; similarity requires human reference data."},
{"id":"I4","name":"Cross-context consistency","protocol":"Probe 20 prior commitments using matched paraphrases in two task contexts, including justified exceptions.","capacity":"Fraction of responses consistent with valid commitments and explicit exceptions.","features":"Conditional consistency and scope-sensitive changes.","control":"Reward neither word-for-word repetition nor refusal to revise outdated commitments."},
{"id":"I5","name":"Recovery after perturbation","protocol":"Measure baseline, introduce a bounded reversible memory-index fault, repair it, then probe at 1, 5 and 20 scheduled ticks across 20 episodes.","capacity":"Post-repair task accuracy; report damage and recovery separately.","features":"Recovery distance to baseline and first sustained return within a preregistered tolerance.","control":"Compare sham perturbations and clean checkpoint clones; full model reset is a separate recovery mechanism."},
{"id":"I6","name":"Experience divergence","protocol":"Clone weights, configuration and empty state into paired agents; randomize balanced histories A/B, then use identical blind probes at baseline and after 1, 5 and 20 washout blocks.","capacity":"Task accuracy by arm, with no reward for divergence itself.","features":"Mean absolute probe-feature divergence minus same-history control divergence; report signed effect and persistence.","control":"Replicate independent pairs, randomize seeds, add swapped-history and memory-erasure arms; do not infer personhood."}
]
