{
  "schema_version": 1,
  "updated_utc": "2026-09-23T05:01:36Z",
  "purpose": "Reconciliation of retained Mnemos cluster experiment records, raw trial outputs, and per-query latency traces for the report. This file supplements, and does not rewrite, the append-only cluster experiment registry.",
  "canonical_sources": {
    "cluster_registry": "~/benchmarks/results/experiment-registry.jsonl",
    "raw_result_layout": "~/benchmarks/results/<run-root>/arms/<arm>/trial-<N>/{context,answer,judge}/results.json",
    "latency_aggregation_artifact": "../macpaw-mnemos-v080-lock/work/stage-latency-breakdown-20260922.json",
    "latency_aggregation_method": "Offline aggregation of retained latency_study records; context.total_ms, final-Qwen TTFT and after_first_token_ms decode duration summed across answer attempts, and user_facing_end_to_end_ms. p50/p95 use linear interpolation. Retries are included. No inference rerun and no judge calls were made.",
    "retrieval_and_accuracy_recomputation": "Read-only recomputation from canonical cluster trial judge/results.json files; accuracy SD is sample SD across the three trial-level percentages over 50 answerable queries. Retrieval metrics are aggregated over answerable queries from each stored retrieval object (k=20): strict rate is the share with all_evidence_retrieved=true; evidence-unit recall is mean all_evidence_recall_at_k; MRR is mean reciprocal rank of first gold evidence. No answer generation or judging was rerun.",
    "architecture_trial_sources": {
      "EXP-073": "~/benchmarks/results/w9-20260918-191721/arms/w9_dense_baseline_3t/trial-{0,1,2}/judge/results.json",
      "EXP-056": "~/benchmarks/results/w5-qwen-compress-20260917-195026/arms/qwen_abstractive_dense_rerank/trial-{0,1,2}/judge/results.json",
      "EXP-077": "~/benchmarks/results/w9-20260918-191721/arms/w9_abs300_decompose_3t/trial-{0,1,2}/judge/results.json",
      "EXP-074": "~/benchmarks/results/w9-20260918-191721/arms/w9_hybrid_readretrieve_3t/trial-{0,1,2}/judge/results.json"
    },
    "budget_fill_run_source": "Claude cluster-result handoff supplied aggregate telemetry for EXP-119/120/121. EXP-121 judged counts and all three 60-case trial artifacts were separately verified at ~/benchmarks/results/wKd-decompose-ksweep-20260922-234131/arms/decomp_k24_12k/trial-{0,1,2}/judge/results.json. Other EXP-121 aggregates remain attributed to the handoff.",
    "evidence_token_audit": {
      "job_id": 2478944,
      "status": "Completed; exit 0",
      "output_path": "~/benchmarks/results/mnemos-evidence-token-counts-2478944.json",
      "output_sha256": "dbb2c08df0de3498808fe5a41a89359f2fc54540d23213226ffa76d5e51a64c7",
      "tokenizer": "Qwen/Qwen3.8-27B snapshot 1d4bf0f2ff6012fd82039f2fa52739d0dd7c60c0",
      "tokenizer_sha256": {
        "tokenizer.json": "0997f410c57a1f4e53b09e4be8f4a172d90edd9564368fb0847030937229b9f3",
        "tokenizer_config.json": "b11349aafa7cdc6a320767cf7ceb29ed82f7eda5d65e8e0819e76f0ce947bf27"
      },
      "definition": "Qwen tokenizer encode of saved evidence text only (prompt_evidence strings verbatim, or structured oracle_evidence text bodies joined by two newlines), add_special_tokens=false; excludes metadata, question, system/user framing, and answer. The report table uses the all-query p50 across each arm's retained evidence payloads.",
      "source_files_by_experiment_sha256": {
        "EXP-073": [
          "35c4b2eac5193a713ff844ebc1ec2b0ba473cd73973841c8c71bac3f77d8446b",
          "c1a561561feeb25f171706dfa902c8fcdb4633438f89bf428eca0cbb0831bb6d",
          "5fa7c98255e6a7484c408114cfdeb82215a67f56cfd10e68001bc38a21105c60"
        ],
        "EXP-056": [
          "04295d6a79c28d18c6c2ceaa0ac71a7be7124bce1c31871b5178fb161324eb63",
          "eabe01d61e3fa9b969c68e47698e6ac7f4372e7aca29de1819a95c9565ab1d2d",
          "9373d38ff3c1288da238adb6181de23d8f8853807c74f5552a3324ad32fefc77"
        ],
        "EXP-077": [
          "bbff47c8d3490f02d4f714d1a1f80f829393e468d43fad51aa368152fb4e0337",
          "dd039a720078d364f195035e27b9e59cc7da3dbc5bb3bacb5a37d8d1503ff3ea",
          "30c55880a235f6502258f83c31b2e34c072c4867563dbd63fb4cd4ec5353f74"
        ],
        "EXP-074": [
          "93290aced81ec5962553e209cf1758b1b9388a36132254c5bafbfb9cd0837149",
          "666304a3109e4bd3a732d55f8c5cdb1a35291e144f510fb641c5b896f34c23ec",
          "3c4f72b77254e40d69a6e40f7844161fcd6b0e12e30442140dbd8047e8146ad9"
        ],
        "EXP-085": [
          "4d98a89a41c4d385cbce3176bc4c96faf7cba604cfb04b337b553084136566c8"
        ],
        "EXP-083": [
          "eedf0b33f890295b12950de2085f51eea6201b088d579d782f97818e44fddf6f"
        ],
        "EXP-092": [
          "1468747c6b27b8b9ff1fad486207d3de7949d9d83bb56cc6d6f43680e8aa872a"
        ],
        "EXP-098": [
          "7bd1a6baba1cd1bf884795d02e94a0e234c57f52a66d9b5bda9af08dd1cf8cdb"
        ]
      },
      "legacy_values_note": "Earlier report values labelled reader-context tokens used an underspecified definition. The corrected evidence-only counts below are authoritative; legacy values are preserved in legacy_reader_context_tokens_reported_p50 where available."
    },
    "final_packed_context_metrics": {
      "method": "Recomputed every retained answerable query-trial using the canonical evaluator score_retrieval(oracle, final_packed_hits, k=20). final_packed_hits are the exact ordered records sent to the Qwen reader after packing (not the wider candidate pool). Canonical evidence identities are deduplicated by the evaluator.",
      "definitions": {
        "recall_strict_percent": "Percent of answerable query-trials where every distinct gold evidence unit is present in the final packed context.",
        "recall_unit_percent": "Mean percent of distinct gold evidence units present in each final packed context.",
        "mrr": "Mean reciprocal rank of the first gold evidence item in final packed order; zero when no gold item is present."
      },
      "denominator_per_variant": "150 answerable query-trials (3 trials x 50 answerable queries)",
      "verification": "All recomputed per-query metric objects matched the saved canonical retrieval objects exactly for all 150 cases in each of the four architecture variants.",
      "legacy_alternate_mrr_note": "Older cutoff-replay MRR values were calculated from a separate retained-ranking diagnostic and differ from canonical score_retrieval. They remain in the audit for provenance but are superseded by the final-packed-context MRR below and are not shown in the report table.",
      "experiments": {
        "EXP-073": {
          "strict_all_evidence_rate_percent": 42.0,
          "mean_evidence_unit_recall_percent": 70.2,
          "mrr": 0.8195,
          "trial_metrics": [
            {"trial": 0, "answerable_queries": 50, "strict_percent": 42.0, "unit_recall_percent": 70.1667, "mrr": 0.819524, "source_sha256": "35c4b2eac5193a713ff844ebc1ec2b0ba473cd73973841c8c71bac3f77d8446b"},
            {"trial": 1, "answerable_queries": 50, "strict_percent": 42.0, "unit_recall_percent": 70.1667, "mrr": 0.819524, "source_sha256": "c1a561561feeb25f171706dfa902c8fcdb4633438f89bf428eca0cbb0831bb6d"},
            {"trial": 2, "answerable_queries": 50, "strict_percent": 42.0, "unit_recall_percent": 70.1667, "mrr": 0.819524, "source_sha256": "5fa7c98255e6a7484c408114cfdeb82215a67f56cfd10e68001bc38a21105c60"}
          ]
        },
        "EXP-056": {
          "strict_all_evidence_rate_percent": 30.7,
          "mean_evidence_unit_recall_percent": 60.5,
          "mrr": 0.7188,
          "trial_metrics": [
            {"trial": 0, "answerable_queries": 50, "strict_percent": 0.0, "unit_recall_percent": 32.8333, "mrr": 0.435524, "source_sha256": "04295d6a79c28d18c6c2ceaa0ac71a7be7124bce1c31871b5178fb161324eb63"},
            {"trial": 1, "answerable_queries": 50, "strict_percent": 46.0, "unit_recall_percent": 74.3333, "mrr": 0.8605, "source_sha256": "eabe01d61e3fa9b969c68e47698e6ac7f4372e7aca29de1819a95c9565ab1d2d"},
            {"trial": 2, "answerable_queries": 50, "strict_percent": 46.0, "unit_recall_percent": 74.3333, "mrr": 0.8605, "source_sha256": "9373d38ff3c1288da238adb6181de23d8f8853807c74f5552a3324ad32fefc77"}
          ]
        },
        "EXP-077": {
          "strict_all_evidence_rate_percent": 58.0,
          "mean_evidence_unit_recall_percent": 79.5,
          "mrr": 0.8829,
          "trial_metrics": [
            {"trial": 0, "answerable_queries": 50, "strict_percent": 58.0, "unit_recall_percent": 79.5, "mrr": 0.882857, "source_sha256": "bbff47c8d3490f02d4f714d1a1f80f829393e468d43fad51aa368152fb4e0337"},
            {"trial": 1, "answerable_queries": 50, "strict_percent": 58.0, "unit_recall_percent": 79.5, "mrr": 0.882857, "source_sha256": "dd039a720078d364f195035e27b9e59cc7da3dbc5bb3bacb5a37d8d1503ff3ea"},
            {"trial": 2, "answerable_queries": 50, "strict_percent": 58.0, "unit_recall_percent": 79.5, "mrr": 0.882857, "source_sha256": "30c55880a235f6502258f83c31b2e34c072c4867563dbd63fb4cd4ec5353f74"}
          ]
        },
        "EXP-074": {
          "strict_all_evidence_rate_percent": 64.0,
          "mean_evidence_unit_recall_percent": 84.5,
          "mrr": 0.864,
          "trial_metrics": [
            {"trial": 0, "answerable_queries": 50, "strict_percent": 64.0, "unit_recall_percent": 84.5, "mrr": 0.864, "source_sha256": "93290aced81ec5962553e209cf1758b1b9388a36132254c5bafbfb9cd0837149"},
            {"trial": 1, "answerable_queries": 50, "strict_percent": 64.0, "unit_recall_percent": 84.5, "mrr": 0.864, "source_sha256": "666304a3109e4bd3a732d55f8c5cdb1a35291e144f510fb641c5b896f34c23ec"},
            {"trial": 2, "answerable_queries": 50, "strict_percent": 64.0, "unit_recall_percent": 84.5, "mrr": 0.864, "source_sha256": "3c4f72b77254e40d69a6e40f7844161fcd6b0e12e30442140dbd8047e8146ad9"}
          ]
        }
      }
    },
    "limitations": "Prefill and server queue were not separately instrumented. For fixed upper-bound packs, pack assembly occurred outside the timed reader trace; context-build is N/A, not zero-cost. Stage p50/p95 percentiles are marginal and need not sum to E2E percentiles."
  },
  "experiments": {
    "EXP-073": {
      "run_root": "w9-20260918-191721",
      "arm": "w9_dense_baseline_3t",
      "trial_count": 3,
      "accuracy_percent": {
        "mean": 41.3,
        "standard_deviation_sample": 1.2,
        "correct_by_trial": [
          20,
          21,
          21
        ],
        "answerable_denominator_per_trial": 50
      },
      "reader_context_tokens_exact_p50": 2969.5,
      "legacy_reader_context_tokens_reported_p50": 2926,
      "evidence_payload_tokens_qwen38": {
        "all_queries_p50": 2969.5,
        "all_queries_range": [
          1464,
          4420
        ],
        "answerable_p50": 2943.5,
        "answerable_range": [
          1464,
          4420
        ],
        "no_answer_p50": 3298,
        "no_answer_range": [
          2745,
          4358
        ],
        "records": 180,
        "trial_count": 3
      },
      "index_size_reported_mb": 558,
      "latency_samples": 180,
      "latency_seconds_p50": {
        "context_construction": 1.788,
        "qwen_ttft": 0.363,
        "qwen_decode_duration": 12.204,
        "end_to_end": 16.478
      },
      "latency_seconds_p95": {
        "context_construction": 7.864,
        "qwen_ttft": 0.741,
        "qwen_decode_duration": 54.14,
        "end_to_end": 55.824
      },
      "retrieval_metrics_at_20": {
        "strict_all_evidence_rate_percent": 42,
        "mean_evidence_unit_recall_percent": 70.2,
        "mrr": 0.82,
        "answerable_queries_per_trial": 50,
        "trial_values": [
          {
            "strict_all_evidence_rate_percent": 42,
            "mean_evidence_unit_recall_percent": 70.2,
            "mrr": 0.82
          },
          {
            "strict_all_evidence_rate_percent": 42,
            "mean_evidence_unit_recall_percent": 70.2,
            "mrr": 0.82
          },
          {
            "strict_all_evidence_rate_percent": 42,
            "mean_evidence_unit_recall_percent": 70.2,
            "mrr": 0.82
          }
        ],
        "source_note": "Stored per-query retrieval.k=20 records; alternate k metrics below rescore the saved packed hit list, not a full candidate pool."
      },
      "retrieval_metrics_packed_hits_by_k": {
        "5": {
          "strict_all_evidence_rate_percent": 38,
          "mean_evidence_unit_recall_percent": 68.17,
          "mrr": 0.86167
        },
        "10": {
          "strict_all_evidence_rate_percent": 42,
          "mean_evidence_unit_recall_percent": 70.17,
          "mrr": 0.86167
        },
        "20": {
          "strict_all_evidence_rate_percent": 42,
          "mean_evidence_unit_recall_percent": 70.17,
          "mrr": 0.86167
        },
        "basis": "Ordered retrieved_hits saved in the reader pack; full candidate pool was not retained. N=150 answerable query-trials. MRR uses first occurrence of the first matching gold identity."
      },
      "full_candidate_pool_replay_by_k": {
        "5": {
          "strict_all_evidence_rate_percent": 36,
          "mean_evidence_unit_recall_percent": 67,
          "mrr": 0.85467
        },
        "10": {
          "strict_all_evidence_rate_percent": 42,
          "mean_evidence_unit_recall_percent": 71,
          "mrr": 0.85467
        },
        "20": {
          "strict_all_evidence_rate_percent": 56,
          "mean_evidence_unit_recall_percent": 79.5,
          "mrr": 0.85751
        },
        "50": {
          "strict_all_evidence_rate_percent": 74,
          "mean_evidence_unit_recall_percent": 88,
          "mrr": 0.85751
        },
        "100": {
          "strict_all_evidence_rate_percent": 82,
          "mean_evidence_unit_recall_percent": 92,
          "mrr": 0.85779
        },
        "answerable_query_trials": 150,
        "baseline_candidate_list_parity": {
          "trial_0": "50/50 exact ordered hit IDs and scores",
          "trial_1": "50/50 exact ordered hit IDs and scores",
          "trial_2": "50/50 exact ordered hit IDs and scores"
        },
        "method": "Rescore full frozen candidate ranking before evidence packing; retrieval/reranking only, no answer generation or judging.",
        "job_id": 2478867,
        "run_root": "~/benchmarks/results/multik-retrieval-only-20260922-233540/arms/EXP-073/results.json",
        "basis_note": "Different from retrieval_metrics_packed_hits_by_k: this is the full pre-pack candidate pool. Do not compare its @5/@10/@20 directly to saved-pack diagnostics."
      },
      "extended_cutoff_status": {
        "original_condition": "Not recoverable: retained original run records contain only the packed retrieved_hits/k=20 list, not the complete ordered candidate ranking needed for exact @50/@100.",
        "supplemental_completed_work_study_id": "multik-retrieval-only-20260922-233540",
        "supplemental_condition": "Widened source candidate depth (dense_k=100, candidate_cap=128); report only as a separate diagnostic, not as original-condition metrics.",
        "baseline_saved_pack_identity_matches": {
          "trial-0": [
            50,
            50
          ],
          "trial-1": [
            50,
            50
          ],
          "trial-2": [
            50,
            50
          ]
        }
      }
    },
    "EXP-056": {
      "run_root": "w5-qwen-compress-20260917-195026",
      "arm": "qwen_abstractive_dense_rerank",
      "trial_count": 3,
      "accuracy_percent": {
        "mean": 42,
        "standard_deviation_sample": 2,
        "correct_by_trial": [
          21,
          20,
          22
        ],
        "answerable_denominator_per_trial": 50
      },
      "reader_context_tokens_exact_p50": 1668.5,
      "evidence_payload_tokens_qwen38": {
        "all_queries_p50": 1668.5,
        "all_queries_range": [
          779,
          2879
        ],
        "answerable_p50": 1666,
        "answerable_range": [
          779,
          2675
        ],
        "no_answer_p50": 1853.5,
        "no_answer_range": [
          1351,
          2879
        ],
        "records": 180,
        "trial_count": 3
      },
      "index_size_reported_mb": 40,
      "latency_samples": 180,
      "latency_seconds_p50": {
        "context_construction": 0.186,
        "qwen_ttft": 0.235,
        "qwen_decode_duration": 11.846,
        "end_to_end": 12.343
      },
      "latency_seconds_p95": {
        "context_construction": 0.401,
        "qwen_ttft": 0.48,
        "qwen_decode_duration": 47.058,
        "end_to_end": 47.718
      },
      "retrieval_metrics_at_20": {
        "strict_all_evidence_rate_percent": 30.7,
        "mean_evidence_unit_recall_percent": 60.5,
        "mrr": 0.719,
        "answerable_queries_per_trial": 50,
        "trial_values": [
          {
            "strict_all_evidence_rate_percent": 0,
            "mean_evidence_unit_recall_percent": 32.8,
            "mrr": 0.436
          },
          {
            "strict_all_evidence_rate_percent": 46,
            "mean_evidence_unit_recall_percent": 74.3,
            "mrr": 0.86
          },
          {
            "strict_all_evidence_rate_percent": 46,
            "mean_evidence_unit_recall_percent": 74.3,
            "mrr": 0.86
          }
        ],
        "source_note": "Stored per-query retrieval.k=20 records; trial 0 is materially lower than trials 1-2. Registry prose says 46%, which conflicts with the raw three-trial recomputation."
      },
      "retrieval_metrics_packed_hits_by_k": {
        "5": {
          "strict_all_evidence_rate_percent": 24,
          "mean_evidence_unit_recall_percent": 54.89,
          "mrr": 0.71511
        },
        "10": {
          "strict_all_evidence_rate_percent": 30.67,
          "mean_evidence_unit_recall_percent": 60.5,
          "mrr": 0.71884
        },
        "20": {
          "strict_all_evidence_rate_percent": 30.67,
          "mean_evidence_unit_recall_percent": 60.5,
          "mrr": 0.71884
        },
        "basis": "Ordered retrieved_hits saved in the reader pack; full candidate pool was not retained. N=150 answerable query-trials. MRR uses first occurrence of the first matching gold identity.",
        "trial_level_sample_sd_at_5": {
          "strict_all_evidence_rate_percentage_points": 20.78,
          "mean_evidence_unit_recall_percentage_points": 21.55,
          "mrr": 0.24749
        },
        "trial_level_sample_sd_at_10": {
          "strict_all_evidence_rate_percentage_points": 26.56,
          "mean_evidence_unit_recall_percentage_points": 23.96,
          "mrr": 0.24536
        }
      },
      "alternate_cutoff_replay": {
        "job_id": 2478867,
        "status": "Quarantined: baseline candidate-list parity failed",
        "baseline_candidate_list_parity": {
          "trial_0": "0/50 exact ordered hits",
          "trial_1": "0/50 exact ordered hits",
          "trial_2": "0/50 exact ordered hits"
        },
        "use_metrics": false,
        "reason": "Replay candidate lists did not reproduce the original saved retrieval outputs; no alternate-k metrics are reportable from this replay."
      },
      "extended_cutoff_status": {
        "original_condition": "Not recoverable: retained original run records contain only the packed retrieved_hits/k=20 list, not the complete ordered candidate ranking needed for exact @50/@100.",
        "supplemental_completed_work_study_id": "multik-retrieval-only-20260922-233540",
        "supplemental_condition": "Widened source candidate depth (dense_k=100, candidate_cap=128); report only as a separate diagnostic, not as original-condition metrics.",
        "baseline_saved_pack_identity_matches": {
          "trial-0": [
            0,
            50
          ],
          "trial-1": [
            0,
            50
          ],
          "trial-2": [
            0,
            50
          ]
        }
      }
    },
    "EXP-077": {
      "run_root": "w9-20260918-191721",
      "arm": "w9_abs300_decompose_3t",
      "trial_count": 3,
      "accuracy_percent": {
        "mean": 47.3,
        "standard_deviation_sample": 2.3,
        "correct_by_trial": [
          23,
          23,
          25
        ],
        "answerable_denominator_per_trial": 50
      },
      "reader_context_tokens_exact_p50": 1002.5,
      "legacy_reader_context_tokens_reported_p50": 936,
      "evidence_payload_tokens_qwen38": {
        "all_queries_p50": 1002.5,
        "all_queries_range": [
          695,
          1564
        ],
        "answerable_p50": 992,
        "answerable_range": [
          695,
          1564
        ],
        "no_answer_p50": 1139.5,
        "no_answer_range": [
          864,
          1224
        ],
        "records": 180,
        "trial_count": 3
      },
      "index_size_reported_mb": 39,
      "latency_samples": 180,
      "latency_seconds_p50": {
        "context_construction": 2.704,
        "qwen_ttft": 0.143,
        "qwen_decode_duration": 9.105,
        "end_to_end": 12.003
      },
      "latency_seconds_p95": {
        "context_construction": 3.689,
        "qwen_ttft": 0.318,
        "qwen_decode_duration": 40.868,
        "end_to_end": 44.546
      },
      "retrieval_metrics_at_20": {
        "strict_all_evidence_rate_percent": 58,
        "mean_evidence_unit_recall_percent": 79.5,
        "mrr": 0.883,
        "answerable_queries_per_trial": 50,
        "trial_values": [
          {
            "strict_all_evidence_rate_percent": 58,
            "mean_evidence_unit_recall_percent": 79.5,
            "mrr": 0.883
          },
          {
            "strict_all_evidence_rate_percent": 58,
            "mean_evidence_unit_recall_percent": 79.5,
            "mrr": 0.883
          },
          {
            "strict_all_evidence_rate_percent": 58,
            "mean_evidence_unit_recall_percent": 79.5,
            "mrr": 0.883
          }
        ],
        "source_note": "Stored per-query retrieval.k=20 records; alternate k metrics below rescore the saved packed hit list, not a full candidate pool."
      },
      "retrieval_metrics_packed_hits_by_k": {
        "5": {
          "strict_all_evidence_rate_percent": 44,
          "mean_evidence_unit_recall_percent": 72.5,
          "mrr": 0.87667
        },
        "10": {
          "strict_all_evidence_rate_percent": 58,
          "mean_evidence_unit_recall_percent": 79.5,
          "mrr": 0.88286
        },
        "20": {
          "strict_all_evidence_rate_percent": 58,
          "mean_evidence_unit_recall_percent": 79.5,
          "mrr": 0.88286
        },
        "basis": "Ordered retrieved_hits saved in the reader pack; full candidate pool was not retained. N=150 answerable query-trials. MRR uses first occurrence of the first matching gold identity."
      },
      "extended_cutoff_status": {
        "original_condition": "Not recoverable: retained original run records contain only the packed retrieved_hits/k=20 list, not the complete ordered candidate ranking needed for exact @50/@100.",
        "supplemental_completed_work_study_id": "multik-retrieval-only-20260922-233540",
        "supplemental_condition": "Widened source candidate depth (dense_k=100, candidate_cap=128); report only as a separate diagnostic, not as original-condition metrics.",
        "baseline_saved_pack_identity_matches": {
          "trial-0": [
            1,
            50
          ],
          "trial-1": [
            1,
            50
          ],
          "trial-2": [
            1,
            50
          ]
        }
      }
    },
    "EXP-074": {
      "run_root": "w9-20260918-191721",
      "arm": "w9_hybrid_readretrieve_3t",
      "trial_count": 3,
      "accuracy_percent": {
        "mean": 55.3,
        "standard_deviation_sample": 1.2,
        "correct_by_trial": [
          28,
          28,
          27
        ],
        "answerable_denominator_per_trial": 50
      },
      "reader_context_tokens_exact_p50": 2111.5,
      "legacy_reader_context_tokens_reported_p50": 2131,
      "evidence_payload_tokens_qwen38": {
        "all_queries_p50": 2111.5,
        "all_queries_range": [
          1257,
          2893
        ],
        "answerable_p50": 2119.5,
        "answerable_range": [
          1257,
          2893
        ],
        "no_answer_p50": 2094,
        "no_answer_range": [
          1553,
          2535
        ],
        "records": 180,
        "trial_count": 3
      },
      "index_size_reported_mb": 41,
      "latency_samples": 180,
      "latency_seconds_p50": {
        "context_construction": 1.004,
        "qwen_ttft": 0.263,
        "qwen_decode_duration": 12.065,
        "end_to_end": 13.5
      },
      "latency_seconds_p95": {
        "context_construction": 2.009,
        "qwen_ttft": 0.52,
        "qwen_decode_duration": 46.715,
        "end_to_end": 48.21
      },
      "retrieval_metrics_at_20": {
        "strict_all_evidence_rate_percent": 64,
        "mean_evidence_unit_recall_percent": 84.5,
        "mrr": 0.864,
        "answerable_queries_per_trial": 50,
        "trial_values": [
          {
            "strict_all_evidence_rate_percent": 64,
            "mean_evidence_unit_recall_percent": 84.5,
            "mrr": 0.864
          },
          {
            "strict_all_evidence_rate_percent": 64,
            "mean_evidence_unit_recall_percent": 84.5,
            "mrr": 0.864
          },
          {
            "strict_all_evidence_rate_percent": 64,
            "mean_evidence_unit_recall_percent": 84.5,
            "mrr": 0.864
          }
        ],
        "source_note": "Stored per-query retrieval.k=20 records; alternate k metrics below rescore the saved packed hit list, not a full candidate pool. This supersedes the previous approximate recall figure."
      },
      "retrieval_metrics_packed_hits_by_k": {
        "5": {
          "strict_all_evidence_rate_percent": 44,
          "mean_evidence_unit_recall_percent": 73.5,
          "mrr": 0.86067
        },
        "10": {
          "strict_all_evidence_rate_percent": 64,
          "mean_evidence_unit_recall_percent": 84.5,
          "mrr": 0.864
        },
        "20": {
          "strict_all_evidence_rate_percent": 64,
          "mean_evidence_unit_recall_percent": 84.5,
          "mrr": 0.864
        },
        "basis": "Ordered retrieved_hits saved in the reader pack; full candidate pool was not retained. N=150 answerable query-trials. MRR uses first occurrence of the first matching gold identity."
      },
      "extended_cutoff_status": {
        "original_condition": "Not recoverable: retained original run records contain only the packed retrieved_hits/k=20 list, not the complete ordered candidate ranking needed for exact @50/@100.",
        "supplemental_completed_work_study_id": "multik-retrieval-only-20260922-233540",
        "supplemental_condition": "Widened source candidate depth (dense_k=100, candidate_cap=128); report only as a separate diagnostic, not as original-condition metrics.",
        "baseline_saved_pack_identity_matches": {
          "trial-0": [
            0,
            50
          ],
          "trial-1": [
            0,
            50
          ],
          "trial-2": [
            0,
            50
          ]
        }
      }
    },
    "EXP-085": {
      "run_root": "qwen-exact-gold-vllm-20260916",
      "trial_count": 3,
      "answerable_correct_by_trial": [
        44,
        46,
        47
      ],
      "answerable_denominator_per_trial": 50,
      "accuracy_percent": {
        "mean": 91.3,
        "standard_deviation_sample": 3.1,
        "range": [
          88,
          94
        ]
      },
      "reader_context_tokens_exact_median": 549.5,
      "legacy_reader_context_tokens_reported_median": 993.5,
      "evidence_payload_tokens_qwen38": {
        "all_queries_p50": 549.5,
        "all_queries_range": [
          284,
          1800
        ],
        "answerable_p50": 549.5,
        "answerable_range": [
          284,
          1800
        ],
        "no_answer_p50": null,
        "no_answer_range": null,
        "records": 50,
        "trial_count": 1,
        "scope_note": "One saved canonical evidence payload covers all 50 answerable cases and is reused across the three timing replays."
      },
      "latency_samples": 150,
      "latency_seconds_p50": {
        "context_construction": 0,
        "qwen_ttft": 0.122,
        "qwen_decode_duration": 7.693,
        "end_to_end": 7.811
      },
      "latency_seconds_p95": {
        "context_construction": 0,
        "qwen_ttft": 0.32,
        "qwen_decode_duration": 27.543,
        "end_to_end": 27.858
      },
      "context_interpretation": "Fixed exact-gold pack; pack construction was outside the timed reader trace.",
      "recall_mrr": "Not applicable: gold evidence supplied"
    },
    "EXP-092": {
      "run_root": "w11-20260918-201415",
      "arm": "w11_random_n8",
      "trial_count": 3,
      "answerable_correct_by_trial": [
        42,
        44,
        42
      ],
      "answerable_denominator_per_trial": 50,
      "no_answer_correct_by_trial": [
        2,
        4,
        4
      ],
      "no_answer_denominator_per_trial": 10,
      "accuracy_percent": {
        "mean": 85.3,
        "standard_deviation_sample": 2.3,
        "range": [
          84,
          88
        ]
      },
      "reader_context_tokens_exact_p50": 2931.5,
      "legacy_reader_context_tokens_reported_p50": 3195,
      "evidence_payload_tokens_qwen38": {
        "all_queries_p50": 2931.5,
        "all_queries_range": [
          2439,
          3961
        ],
        "answerable_p50": 2917.5,
        "answerable_range": [
          2463,
          3961
        ],
        "no_answer_p50": 2949.5,
        "no_answer_range": [
          2439,
          3396
        ],
        "records": 60,
        "trial_count": 1,
        "scope_note": "One canonical prebuilt evidence pack covers 50 answerable and 10 no-answer cases; it was reused across the three timing trials."
      },
      "latency_samples": 180,
      "latency_seconds_p50": {
        "context_construction_instrumented": 0,
        "qwen_ttft": 0.343,
        "qwen_decode_duration": 10.632,
        "end_to_end": 10.967
      },
      "latency_seconds_p95": {
        "context_construction_instrumented": 0,
        "qwen_ttft": 0.66,
        "qwen_decode_duration": 43.04,
        "end_to_end": 43.804
      },
      "context_interpretation": "Eight-random-distractor evidence pack was prebuilt; pack assembly was outside the timed reader trace.",
      "retry_queries": 14,
      "p50_output_tokens_all_attempts": 550.5,
      "recall_mrr": "Not applicable: gold evidence supplied"
    },
    "EXP-083": {
      "run_root": "w11-20260918-201415",
      "arm": "w11_goldinject_raw_3t",
      "trial_count": 3,
      "answerable_correct_by_trial": [
        39,
        36,
        41
      ],
      "answerable_denominator_per_trial": 50,
      "accuracy_percent": {
        "mean": 77.3,
        "standard_deviation_sample": 5,
        "range": [
          72,
          82
        ]
      },
      "reader_context_tokens_exact_p50": 2992.5,
      "legacy_reader_context_tokens_reported_p50": 3238,
      "evidence_payload_tokens_qwen38": {
        "all_queries_p50": 2992.5,
        "all_queries_range": [
          2234,
          4659
        ],
        "answerable_p50": 2856,
        "answerable_range": [
          2234,
          4659
        ],
        "no_answer_p50": 3207.5,
        "no_answer_range": [
          2746,
          4358
        ],
        "records": 60,
        "trial_count": 1,
        "scope_note": "One canonical prebuilt evidence pack covers 50 answerable and 10 no-answer cases; it was reused across the three timing trials."
      },
      "qwen_answer_usage_input_tokens_by_trial_median": [
        3265,
        3045,
        3265
      ],
      "index_bytes": 557933599,
      "latency_samples": 180,
      "latency_seconds_p50": {
        "context_construction_instrumented": 0,
        "qwen_ttft": 0.354,
        "qwen_decode_duration": 11.01,
        "end_to_end": 11.35
      },
      "latency_seconds_p95": {
        "context_construction_instrumented": 0,
        "qwen_ttft": 0.733,
        "qwen_decode_duration": 47.953,
        "end_to_end": 48.665
      },
      "context_interpretation": "Retrieved candidate pack plus injected gold was assembled before the timed reader trace; the registry's 3,238 exact reader-context value is distinct from answer.usage.input_tokens.",
      "retry_queries": 17,
      "p50_output_tokens_all_attempts": 559,
      "recall_mrr_scored_condition": "Not applicable: gold evidence was injected into the scored pack.",
      "natural_pre_injection_retrieval_metrics": {
        "source_study": "w11-20260918-201415",
        "source_paths": [
          "~/benchmarks/results/w11-20260918-201415/candidate-pool/raw/adam/results.json",
          "~/benchmarks/results/w11-20260918-201415/candidate-pool/raw/bei/results.json",
          "~/benchmarks/results/w11-20260918-201415/candidate-pool/raw/victoria/results.json"
        ],
        "sha256": [
          "2debf5077641c64b4d324e9a08899cc8bbb6c4ab3ea8cfd454f3c71635e63010",
          "aef0fd54b2db656f2f0950c5df12e115d457e77f9286e3669b770cd9822ab2fe",
          "1c3d8d8af56ce19944ec286c7b06e55a891b901460acbd2a629050460163dad0"
        ],
        "gold_labels_path": "~/benchmarks/mnemos-native-linux/HippoCamp-Chat-v0.8.8-MacPaw-Compact/benchmark/oracle-cases.json",
        "gold_labels_sha256": "3c33c5ee29b31d00292d9d49317569ff87f3e06616e529d641974ecbb5e50453",
        "answerable_cases": 50,
        "candidate_pool": "Ordered pre-pack all_candidates; rank coverage 1-29, median list length 13. These are natural pre-gold rankings, distinct from the gold-injected pack used to score EXP-083.",
        "metric_method": "For each cutoff, score first-occurrence ranks of distinct gold evidence identities. Strict recall is the share of answerable cases with every gold unit in top-k; unit recall is the mean fraction of gold units in top-k; MRR is the mean reciprocal rank of the first gold identity. Duplicate identities use their first rank.",
        "by_k": {
          "5": {
            "strict_all_evidence_rate_percent": 38,
            "mean_evidence_unit_recall_percent": 67.67,
            "mrr": 0.8617
          },
          "10": {
            "strict_all_evidence_rate_percent": 42,
            "mean_evidence_unit_recall_percent": 70.17,
            "mrr": 0.8617
          },
          "20": {
            "strict_all_evidence_rate_percent": 42,
            "mean_evidence_unit_recall_percent": 70.17,
            "mrr": 0.8617
          },
          "50": {
            "strict_all_evidence_rate_percent": 42,
            "mean_evidence_unit_recall_percent": 70.17,
            "mrr": 0.8617
          },
          "100": {
            "strict_all_evidence_rate_percent": 42,
            "mean_evidence_unit_recall_percent": 70.17,
            "mrr": 0.8617
          }
        },
        "cutoff_saturation": "Exact natural-pool ranks are retained through max rank 29. Re-scoring all candidates at k=50 and k=100 confirms no additional gold evidence appears beyond k=20; these cutoffs equal the full available pool.",
        "cutoff_audit": {
          "method": "Re-scored retained natural pre-injection all_candidates against the v0.8.8 oracle using the evaluator evidence-identity rules; answerable denominator n=50 (Adam 18, Bei 15, Victoria 17). Strict recall is queries with all unique gold units; unit recall is mean per-query matched/required unique units; MRR is reciprocal rank of first matched gold unit, zero when none.",
          "source_paths": [
            "~/benchmarks/results/w11-20260918-201415/candidate-pool/raw/adam/results.json",
            "~/benchmarks/results/w11-20260918-201415/candidate-pool/raw/bei/results.json",
            "~/benchmarks/results/w11-20260918-201415/candidate-pool/raw/victoria/results.json"
          ],
          "sha256": [
            "2debf5077641c64b4d324e9a08899cc8bbb6c4ab3ea8cfd454f3c71635e63010",
            "aef0fd54b2db656f2f0950c5df12e115d457e77f9286e3669b770cd9822ab2fe",
            "1c3d8d8af56ce19944ec286c7b06e55a891b901460acbd2a629050460163dad0"
          ],
          "study_id": "w11-20260918-201415"
        }
      }
    },
    "EXP-098": {
      "run_root": "w12-20260918-234922",
      "arm": "w12_hard_nearmiss",
      "trial_count": 3,
      "answerable_correct_by_trial": [
        39,
        44,
        41
      ],
      "answerable_denominator_per_trial": 50,
      "no_answer_correct_by_trial": [
        10,
        9,
        9
      ],
      "no_answer_denominator_per_trial": 10,
      "accuracy_percent": {
        "mean": 82.7,
        "standard_deviation_sample": 5,
        "range": [
          78,
          88
        ]
      },
      "all_cases_correct_by_trial": [
        49,
        53,
        50
      ],
      "all_cases_denominator_per_trial": 60,
      "all_cases_accuracy_percent": {
        "mean": 84.4,
        "standard_deviation_sample": 3.5,
        "range": [
          81.7,
          88.3
        ]
      },
      "reader_context_tokens_exact_p50": 2819.5,
      "legacy_reader_context_tokens_reported_p50": 3083,
      "evidence_payload_tokens_qwen38": {
        "all_queries_p50": 2819.5,
        "all_queries_range": [
          2234,
          3696
        ],
        "answerable_p50": 2772,
        "answerable_range": [
          2234,
          3696
        ],
        "no_answer_p50": 3220,
        "no_answer_range": [
          2780,
          3673
        ],
        "records": 60,
        "trial_count": 1,
        "scope_note": "One canonical prebuilt evidence pack covers 50 answerable and 10 no-answer cases; it was reused across the three timing trials."
      },
      "latency_samples": 180,
      "latency_seconds_p50": {
        "context_construction_instrumented": 0,
        "qwen_ttft": 0.335,
        "qwen_decode_duration": 11.684,
        "end_to_end": 11.988
      },
      "latency_seconds_p95": {
        "context_construction_instrumented": 0,
        "qwen_ttft": 0.766,
        "qwen_decode_duration": 49.375,
        "end_to_end": 50.206
      },
      "context_interpretation": "LLM-selected hard-near-miss pack was prebuilt; pack assembly was outside the timed reader trace.",
      "retry_queries": 26,
      "p50_output_tokens_all_attempts": 605.5,
      "recall_mrr": "Not applicable: gold evidence supplied"
    },
    "EXP-119": {
      "name": "Hybrid summary + verbatim spans + one follow-up retrieval; k24 budget-fill",
      "status": "Completed",
      "trial_count": 3,
      "accuracy_percent": {"mean": 56.0, "standard_deviation_sample": 3.5, "correct_by_trial": [30, 27, 27], "answerable_denominator_per_trial": 50},
      "no_answer_correct_by_trial": [10, 10, 10],
      "no_answer_denominator_per_trial": 10,
      "reader_context_tokens": {"p50": 3022, "range": [2591, 3828], "records": 180},
      "index_size_mb": 40.9,
      "evidence_budget": "k24; 11.8K/12K characters",
      "latency_p50": {"evidence_build_seconds": 1.15, "final_qwen_ttft_seconds": 0.327, "final_qwen_generation_seconds": 15.2, "final_qwen_decode_rate_tokens_per_second": 51.5, "end_to_end_seconds": 16.6},
      "end_to_end_seconds_p95": 57.4,
      "end_to_end_seconds_mean": 22.3,
      "retrieval_metrics_final_pack": {"all_evidence_recall_at_12k_percent": 64.0, "fractional_evidence_unit_recall_percent": 84.0, "mrr": 0.862},
      "notes": "Evidence-build context.total_ms includes controller/planning work. TTFT, generation and decode-rate measurements are from the final Qwen chat call. Decode rate is throughput, not duration."
    },
    "EXP-120": {
      "name": "Abstractive source summaries + dense retrieval/rerank; k24 budget-fill",
      "status": "Completed",
      "trial_count": 3,
      "accuracy_percent": {"mean": 46.7, "standard_deviation_sample": 2.3, "correct_by_trial": [22, 24, 24], "answerable_denominator_per_trial": 50},
      "no_answer_correct_by_trial": [10, 10, 10],
      "no_answer_denominator_per_trial": 10,
      "reader_context_tokens": {"p50": 3124, "range": [2462, 3883], "records": 180},
      "index_size_mb": 40.6,
      "evidence_budget": "k24; 11.9K/12K characters",
      "latency_p50": {"evidence_build_seconds": 0.22, "final_qwen_ttft_seconds": 0.332, "final_qwen_generation_seconds": 16.5, "final_qwen_decode_rate_tokens_per_second": 51.1, "end_to_end_seconds": 16.7},
      "end_to_end_seconds_p95": 60.7,
      "end_to_end_seconds_mean": 26.0,
      "retrieval_metrics_final_pack": {"all_evidence_recall_at_12k_percent": 58.0, "fractional_evidence_unit_recall_percent": 79.7, "mrr": 0.861},
      "notes": "Evidence-build context.total_ms is reported separately from the final Qwen call. TTFT, generation and decode-rate measurements are from that final call; decode rate is throughput, not duration."
    },
    "EXP-121": {
      "name": "Fixed 300-token summary + upfront query decomposition + dense retrieval/rerank; k24",
      "status": "Completed",
      "trial_count": 3,
      "accuracy_percent": {"mean": 54.0, "standard_deviation_sample": 3.5, "correct_by_trial": [28, 28, 25], "answerable_denominator_per_trial": 50},
      "no_answer_correct_by_trial": [10, 10, 9],
      "no_answer_denominator_per_trial": 10,
      "reader_context_tokens": {"p50": 3095, "range": [2514, 4235], "records": 180},
      "index_size_mb": 39.4,
      "evidence_budget": "k24; budget-filling configuration",
      "latency_p50": {"evidence_build_seconds": 2.92, "final_qwen_ttft_seconds": 0.333, "final_qwen_generation_seconds": 15.1, "final_qwen_decode_rate_tokens_per_second": 51.5, "end_to_end_seconds": 18.7},
      "end_to_end_seconds_p95": 60.2,
      "end_to_end_seconds_mean": 25.5,
      "retrieval_metrics_final_pack": {"all_evidence_recall_at_12k_percent": 76.0, "fractional_evidence_unit_recall_percent": 89.3, "mrr": 0.884},
      "score_artifacts": "~/benchmarks/results/wKd-decompose-ksweep-20260922-234131/arms/decomp_k24_12k/trial-{0,1,2}/judge/results.json",
      "notes": "Three completed k24 trials scored 28/50, 28/50, and 25/50 on answerable questions. Raw judged artifacts show no-answer abstention 10/10, 10/10, and 9/10; trial 2 inferred an unsupported exam location. Reader-context records=180 counts evaluated cases across trials, not packed evidence records. The earlier 53.3% k16 result is a separate interim configuration."
    }
  },
  "subagent_audit": [
    {
      "agent": "Beauvoir",
      "task": "Inspect/run assigned Mnemos experiment work",
      "outcome": "No stable cluster connection; no experiment jobs submitted and no results/report changes made.",
      "useful_findings": [
        "Located the retained stage-latency aggregate and latency-recompute JSONL in the sibling Mnemos repository."
      ]
    },
    {
      "agent": "Franklin",
      "task": "Audit stored results and experiment status",
      "outcome": "No local raw architecture traces found in its search; no job submitted.",
      "useful_findings": [
        "Pointed to the active cluster retrieval-k sweep job campaign; its queue state was last observed at 2026-09-22T22:44Z and could not be refreshed during this update."
      ]
    },
    {
      "agent": "GPT-6 Luna, medium (01a0cb56-edc8-76a1-a843-99f97d94af90)",
      "task": "Independent eight-row audit of retained trial files and cluster state",
      "outcome": "Completed read-only audit; no answer/judge reruns and no duplicate cluster jobs submitted.",
      "useful_findings": [
        "Confirmed EXP-073/056/074/077 sample SD, strict all-evidence rate@20, evidence-unit recall@20, and MRR@20 from three raw trials each.",
        "Confirmed upper-bound N/A semantics. Its initial W11 EXP-083 ranking-retention conclusion was superseded by inspection of the candidate-pool artifacts recorded below."
      ]
    },
    {
      "agent": "GPT-6 Luna, medium (01a0cb56-dfa7-79c1-bf35-5193db431072)",
      "task": "Recover architecture retrieval metrics and assess alternate-k replay",
      "outcome": "Architecture raw-metric reconciliation complete; identified exact frozen-index/query replay path for additional k cutoffs and verified the original W11 EXP-083 natural candidate pool.",
      "useful_findings": [
        "Raw trial outputs retain only retrieval.k=20 metrics and pack-limited ranked hits for EXP-056/073/074/077; alternate score cutoffs require retrieval replay. Exact saved queries and frozen indexes are available; EXP-056 trial 0 has an ID-mapping mismatch requiring fidelity review.",
        "EXP-083's original W11 candidate-pool artifacts retain ordered all_candidates lists and support natural pre-injection @5/@10/@20 metrics; hashes and scores are recorded under EXP-083.",
        "Registry prose for EXP-056 reports 46% recall, but raw three-trial strict all-evidence retrieval averages 30.7%; discrepancy preserved for audit."
      ]
    }
  ],
  "completed_work": [
    {
      "experiment_id": null,
      "name": "abs300 selected-source / evidence-depth sweep",
      "status": "Completed; all submitted SLURM jobs verified COMPLETED with exit code 0",
      "last_observed_utc": "2026-09-22T23:09:41Z",
      "run_root": "~/benchmarks/results/wK-abs300-ksweep-20260922-223624",
      "source_index": "~/benchmarks/results/w8-20260918-041017/source-index-w8_abs300_readretrieve",
      "source_commit": "983b2688064b92be16c2d044c6de2a37b767fc16",
      "grid": {
        "selected_sources": [
          8,
          16,
          24
        ],
        "per_hop_final_k": [
          12,
          24,
          36
        ],
        "max_records": [
          8,
          16,
          24
        ],
        "max_chars": 12000,
        "trials_per_setting": 3,
        "stored_retrieval_metric_cutoff_k": 20
      },
      "context_job_ids": [
        2477788,
        2477796,
        2477807,
        2477812,
        2477821,
        2477832,
        2477839,
        2477844,
        2477850
      ],
      "dependent_job_id_range_observed": [
        2477790,
        2477854
      ],
      "attribution": {
        "wckey": "v1/model_development/macpaw/superagent",
        "comment": "billable=n"
      },
      "answer_accuracy_by_setting": [
        {
          "selected_sources": 8,
          "correct_by_trial": [
            26,
            26,
            23
          ],
          "accuracy_mean_percent": 50,
          "accuracy_sample_sd_percent": 3.5
        },
        {
          "selected_sources": 16,
          "correct_by_trial": [
            29,
            28,
            30
          ],
          "accuracy_mean_percent": 58,
          "accuracy_sample_sd_percent": 2
        },
        {
          "selected_sources": 24,
          "correct_by_trial": [
            29,
            32,
            27
          ],
          "accuracy_mean_percent": 58.7,
          "accuracy_sample_sd_percent": 5
        }
      ],
      "retrieval_by_setting": [
        {
          "selected_sources": 8,
          "strict_all_evidence_rate_percent": 62.7,
          "mean_evidence_unit_recall_percent": 84.8,
          "mrr": 0.885
        },
        {
          "selected_sources": 16,
          "strict_all_evidence_rate_percent": 70,
          "mean_evidence_unit_recall_percent": 87.2,
          "mrr": 0.884
        },
        {
          "selected_sources": 24,
          "strict_all_evidence_rate_percent": 74,
          "mean_evidence_unit_recall_percent": 88.7,
          "mrr": 0.884
        }
      ],
      "note": "Supporting study; not an alternate-cutoff recall@k curve or the fixed EXP-077 row. selected_sources, per_hop_final_k, and max_records all change together; each setting has three judged trials."
    },
    {
      "experiment_id": "EXP-083",
      "name": "Natural pre-injection candidate-pool replay cross-check",
      "study_id": "wTR-exp083-natural-rank-20260922-2326Z",
      "status": "Completed; exact ordered candidate IDs and scores match the original W11 artifacts for Adam, Bei, and Victoria.",
      "last_observed_utc": "2026-09-22T23:37:00Z",
      "run_root": "~/benchmarks/results/wTR-exp083-natural-rank-20260922-2326Z",
      "job_array": "2478637_[0-2]",
      "attribution": {
        "wckey": "v1/model_development/macpaw/superagent",
        "comment": "billable=n"
      },
      "candidate_parity": {
        "Adam": "exact ordered IDs and scores match",
        "Bei": "exact ordered IDs and scores match",
        "Victoria": "exact ordered IDs and scores match"
      },
      "note": "No answers or judges were run. EXP-083 recall/MRR in the table is scored on the natural pre-injection candidate pool; the gold-injected scored condition itself remains N/A for retrieval."
    },
    {
      "experiment_id": [
        "EXP-056",
        "EXP-073",
        "EXP-077",
        "EXP-074",
        "EXP-083",
        "EXP-085",
        "EXP-092",
        "EXP-098"
      ],
      "name": "Evidence-only reader-context token recount",
      "status": "Completed; CPU compute job 2478944 exited 0 and recounted every main-table evidence payload with the pinned Qwen tokenizer.",
      "last_observed_utc": "2026-09-22T23:48:00Z",
      "job_id": 2478944,
      "failed_prior_job_id": 2478616,
      "failed_prior_job_error": "Tokenizer received a non-string payload in the initial attempt; no counts from that job were used.",
      "output_path": "~/benchmarks/results/mnemos-evidence-token-counts-2478944.json",
      "output_sha256": "dbb2c08df0de3498808fe5a41a89359f2fc54540d23213226ffa76d5e51a64c7",
      "tokenizer": "Qwen/Qwen3.8-27B snapshot 1d4bf0f2ff6012fd82039f2fa52739d0dd7c60c0",
      "tokenizer_sha256": {
        "tokenizer.json": "0997f410c57a1f4e53b09e4be8f4a172d90edd956436fb0847030937229b9f3",
        "tokenizer_config.json": "b11349aafa7cdc6a320767cf7ceb29ed82f7eda5d65e8e0819e76f0ce947bf27"
      },
      "allocation": {
        "cpus": 4,
        "memory_gb": 16,
        "gpu": false,
        "attribution": {
          "wckey": "v1/model_development/macpaw/superagent",
          "comment": "billable=n"
        }
      },
      "method_requirement": "Tokenize evidence text only; exclude prompt framing, metadata and answer.usage.input_tokens. No GPU, answer generation, or judging."
    },
    {
      "experiment_id": [
        "EXP-056",
        "EXP-073",
        "EXP-074",
        "EXP-077"
      ],
      "name": "Widened retrieval-pool cutoff diagnostic",
      "study_id": "multik-retrieval-only-20260922-233540",
      "status": "Completed on GPU compute job 2478867; exit 0, 30m16s on tus1-p13-g15.",
      "run_root": "~/benchmarks/results/multik-retrieval-only-20260922-233540",
      "output_files": [
        "arms/EXP-056/results.json",
        "arms/EXP-073/results.json",
        "arms/EXP-074/results.json",
        "arms/EXP-077/results.json"
      ],
      "attribution": {
        "wckey": "v1/model_development/macpaw/superagent",
        "comment": "billable=n"
      },
      "method": "No answer generation or judging. Baseline packing replay used dense_k=40/candidate_cap=64 and matched saved packed-hit identities at EXP-056 0/150, EXP-073 150/150, EXP-074 0/150, EXP-077 3/150 query-trials. The depth sweep widened to dense_k=100/candidate_cap=128, then scored passage-ranked hits. This changed retrieval condition is not an exact extension of the original pack for EXP-056/074/077; EXP-073 baseline identity parity does not make its widened-depth results the original evaluated evidence pack. No retrieval latency was instrumented.",
      "metrics_by_experiment": {
        "EXP-056": {
          "baseline_packed_hit_identity_matches": {
            "trial-0": [
              0,
              50
            ],
            "trial-1": [
              0,
              50
            ],
            "trial-2": [
              0,
              50
            ]
          },
          "widened_depth_by_k": {
            "5": {
              "strict_all_evidence_rate_percent": 34,
              "mean_evidence_unit_recall_percent": 63.67,
              "mrr": 0.8507
            },
            "10": {
              "strict_all_evidence_rate_percent": 46,
              "mean_evidence_unit_recall_percent": 71.33,
              "mrr": 0.8507
            },
            "20": {
              "strict_all_evidence_rate_percent": 60,
              "mean_evidence_unit_recall_percent": 80.17,
              "mrr": 0.8535
            },
            "50": {
              "strict_all_evidence_rate_percent": 74,
              "mean_evidence_unit_recall_percent": 87.33,
              "mrr": 0.8544
            },
            "100": {
              "strict_all_evidence_rate_percent": 80,
              "mean_evidence_unit_recall_percent": 90.33,
              "mrr": 0.8544
            }
          }
        },
        "EXP-073": {
          "baseline_packed_hit_identity_matches": {
            "trial-0": [
              50,
              50
            ],
            "trial-1": [
              50,
              50
            ],
            "trial-2": [
              50,
              50
            ]
          },
          "widened_depth_by_k": {
            "5": {
              "strict_all_evidence_rate_percent": 36,
              "mean_evidence_unit_recall_percent": 67,
              "mrr": 0.8547
            },
            "10": {
              "strict_all_evidence_rate_percent": 42,
              "mean_evidence_unit_recall_percent": 71,
              "mrr": 0.8547
            },
            "20": {
              "strict_all_evidence_rate_percent": 56,
              "mean_evidence_unit_recall_percent": 79.5,
              "mrr": 0.8575
            },
            "50": {
              "strict_all_evidence_rate_percent": 74,
              "mean_evidence_unit_recall_percent": 88,
              "mrr": 0.8575
            },
            "100": {
              "strict_all_evidence_rate_percent": 82,
              "mean_evidence_unit_recall_percent": 92,
              "mrr": 0.8578
            }
          }
        },
        "EXP-074": {
          "baseline_packed_hit_identity_matches": {
            "trial-0": [
              0,
              50
            ],
            "trial-1": [
              0,
              50
            ],
            "trial-2": [
              0,
              50
            ]
          },
          "widened_depth_by_k": {
            "5": {
              "strict_all_evidence_rate_percent": 36,
              "mean_evidence_unit_recall_percent": 65,
              "mrr": 0.8707
            },
            "10": {
              "strict_all_evidence_rate_percent": 46,
              "mean_evidence_unit_recall_percent": 72,
              "mrr": 0.8707
            },
            "20": {
              "strict_all_evidence_rate_percent": 58,
              "mean_evidence_unit_recall_percent": 80.83,
              "mrr": 0.8756
            },
            "50": {
              "strict_all_evidence_rate_percent": 72,
              "mean_evidence_unit_recall_percent": 86.67,
              "mrr": 0.8756
            },
            "100": {
              "strict_all_evidence_rate_percent": 84,
              "mean_evidence_unit_recall_percent": 93,
              "mrr": 0.8756
            }
          }
        },
        "EXP-077": {
          "baseline_packed_hit_identity_matches": {
            "trial-0": [
              1,
              50
            ],
            "trial-1": [
              1,
              50
            ],
            "trial-2": [
              1,
              50
            ]
          },
          "widened_depth_by_k": {
            "5": {
              "strict_all_evidence_rate_percent": 34,
              "mean_evidence_unit_recall_percent": 64.67,
              "mrr": 0.8607
            },
            "10": {
              "strict_all_evidence_rate_percent": 50,
              "mean_evidence_unit_recall_percent": 74,
              "mrr": 0.8607
            },
            "20": {
              "strict_all_evidence_rate_percent": 64,
              "mean_evidence_unit_recall_percent": 83.83,
              "mrr": 0.8652
            },
            "50": {
              "strict_all_evidence_rate_percent": 78,
              "mean_evidence_unit_recall_percent": 90,
              "mrr": 0.8652
            },
            "100": {
              "strict_all_evidence_rate_percent": 88,
              "mean_evidence_unit_recall_percent": 95.67,
              "mrr": 0.8652
            }
          }
        }
      },
      "limitations": "Full output files preserve per-query rankings and trial-level summaries. Widened-depth cutoffs are supplementary only because expanding source candidates changes the evaluated evidence condition; they are retained in this audit and are not used for the main table's final-packed-context recall/MRR."
    }
  ],
  "active_work": [
    {
      "experiment_id": [
        "EXP-056",
        "EXP-073",
        "EXP-074",
        "EXP-077",
        "EXP-083",
        "EXP-085",
        "EXP-092",
        "EXP-098"
      ],
      "name": "Separate reader thinking-OFF comparison",
      "study_id": "wT-thinkoff-20260922-231157",
      "status": "In progress at 2026-09-22T23:49Z; answer arrays are running or pending behind dependencies, so this separate thinking-OFF comparison remains incomplete.",
      "last_observed_utc": "2026-09-22T23:49:47Z",
      "active_queue_entries_at_observation": 34,
      "current_answer_arrays": {
        "EXP-056": [
          2478881
        ],
        "EXP-074": [
          2478884,
          2478887
        ],
        "EXP-077": [
          2478890,
          2478893,
          2478898
        ],
        "EXP-083": [
          2478901
        ],
        "EXP-085": [
          2478904
        ],
        "EXP-092": [
          2478907,
          2478910
        ],
        "EXP-098": [
          2478913,
          2478916
        ]
      },
      "judge_trials_present_by_exp": {
        "EXP-056": [
          1,
          2
        ],
        "EXP-073": [
          0,
          1,
          2
        ],
        "EXP-074": [
          2
        ],
        "EXP-077": [],
        "EXP-083": [
          0,
          1
        ],
        "EXP-085": [],
        "EXP-092": [
          2
        ],
        "EXP-098": [
          2
        ]
      },
      "run_root": "~/benchmarks/results/wT-thinkoff-20260922-231157",
      "condition": "Reader thinking OFF; uses each original experiment's saved context/merged/results.json as input. This is separate from the main table's thinking-ON results and must not replace them.",
      "answer_jobs_by_exp": {
        "EXP-056": [
          2478366,
          2478369,
          2478372
        ],
        "EXP-073": [
          2478375,
          2478378,
          2478386
        ],
        "EXP-074": [
          2478390,
          2478394,
          2478397
        ],
        "EXP-077": [
          2478400,
          2478403,
          2478406
        ],
        "EXP-083": [
          2478409,
          2478412,
          2478415
        ],
        "EXP-085": [
          2478418
        ],
        "EXP-092": [
          2478421,
          2478424,
          2478427
        ],
        "EXP-098": [
          2478430,
          2478433,
          2478436
        ]
      },
      "attribution": {
        "wckey": "v1/model_development/macpaw/superagent",
        "comment": "billable=n"
      },
      "scope_note": "Thinking-OFF repeats were explicitly requested for EXP-085/092/083; the submitted batch also contains the other five table arms. EXP-085 has one trial only. Keep all outputs isolated; none replace the thinking-ON main results.",
      "context_note": "No fresh retrieval/context construction or tokenizer measurement is run here; these jobs consume the original saved evidence packs.",
      "submission_manifest": "~/benchmarks/results/wT-thinkoff-20260922-231157/submissions.tsv",
      "logs": "~/benchmarks/logs/to-*-to_exp*-<jobid>.out and .err"
    }
  ],
  "accuracy_convention": "Report headline accuracy is judge.answer_correct over the 50 answerable cases per trial; sample standard deviation is used across the three trial percentages. All-60-case accuracy is retained separately when available."
}
