|
127 | 127 | "note": "karpathy_compiled (Karpathy condition D2 -- LLM-compiled wiki): Adapter wired (sme/conditions/karpathy_compiled.py + compile-wiki) as the Karpathy D2 control, but not yet run through any SME cat. No baseline reading exists for any cat -- WIRED but NOT YET BENCHED, an honest coverage gap, not a measured 0.", |
128 | 128 | "source": null, |
129 | 129 | "source_type": "not_run" |
| 130 | + }, |
| 131 | + "postgres_ingest": { |
| 132 | + "value": 0.833, |
| 133 | + "metric": "mean_recall (jp-realm Condition-A)", |
| 134 | + "scoreability": "scoreable", |
| 135 | + "note": "postgres+pgvector verbatim ingest, jp-realm-v0.1 Condition-A (n=30); SAME all-MiniLM-L6-v2 embedding + corpus as flat, ONLY the backend swapped (chroma→postgres+pgvector). mean_recall 0.833, full-recall 21/30 — IDENTICAL to the flat floor. The mempalace-raw ablation: verbatim postgres storage without the palace graph retrieves at the flat floor; mempalace's lift is the GRAPH.", |
| 136 | + "source": "https://github.com/techempower-org/multipass-structural-memory-eval/blob/main/baselines/jp_realm_v0_1_postgres_condA_2026-05-31.json", |
| 137 | + "source_type": "sme_measured" |
130 | 138 | } |
131 | 139 | }, |
132 | 140 | "2c": { |
|
196 | 204 | "note": "karpathy_compiled (Karpathy condition D2 -- LLM-compiled wiki): Adapter wired (sme/conditions/karpathy_compiled.py + compile-wiki) as the Karpathy D2 control, but not yet run through any SME cat. No baseline reading exists for any cat -- WIRED but NOT YET BENCHED, an honest coverage gap, not a measured 0.", |
197 | 205 | "source": null, |
198 | 206 | "source_type": "not_run" |
| 207 | + }, |
| 208 | + "postgres_ingest": { |
| 209 | + "value": 0.833, |
| 210 | + "metric": "mean_recall by-hop (jp-realm Cond-A)", |
| 211 | + "scoreability": "scoreable", |
| 212 | + "by_hop": { |
| 213 | + "1": 0.852, |
| 214 | + "2": 0.667 |
| 215 | + }, |
| 216 | + "note": "postgres Condition-A by-hop: hop-1 0.852 (n27) / hop-2 0.667 (n3) — identical to flat; no depth scaling (no traversal), the expected verbatim-first signature.", |
| 217 | + "source": "https://github.com/techempower-org/multipass-structural-memory-eval/blob/main/baselines/jp_realm_v0_1_postgres_condA_2026-05-31.json", |
| 218 | + "source_type": "sme_measured" |
199 | 219 | } |
200 | 220 | }, |
201 | 221 | "3": { |
|
256 | 276 | "note": "karpathy_compiled (Karpathy condition D2 -- LLM-compiled wiki): Adapter wired (sme/conditions/karpathy_compiled.py + compile-wiki) as the Karpathy D2 control, but not yet run through any SME cat. No baseline reading exists for any cat -- WIRED but NOT YET BENCHED, an honest coverage gap, not a measured 0.", |
257 | 277 | "source": null, |
258 | 278 | "source_type": "not_run" |
| 279 | + }, |
| 280 | + "postgres_ingest": { |
| 281 | + "value": 0.0, |
| 282 | + "scoreability": "N/A-by-design", |
| 283 | + "note": "postgres+pgvector is a vector store with no graph — surfaces no structured contradiction pairs by construction. Same 0.00 floor as flat.", |
| 284 | + "source": null, |
| 285 | + "source_type": "sme_measured" |
259 | 286 | } |
260 | 287 | }, |
261 | 288 | "4": { |
|
319 | 346 | "note": "karpathy_compiled (Karpathy condition D2 -- LLM-compiled wiki): Adapter wired (sme/conditions/karpathy_compiled.py + compile-wiki) as the Karpathy D2 control, but not yet run through any SME cat. No baseline reading exists for any cat -- WIRED but NOT YET BENCHED, an honest coverage gap, not a measured 0.", |
320 | 347 | "source": null, |
321 | 348 | "source_type": "not_run" |
| 349 | + }, |
| 350 | + "postgres_ingest": { |
| 351 | + "value": null, |
| 352 | + "scoreability": "N/A-by-design", |
| 353 | + "note": "no graph → no edges to measure for monoculture/dedup. N/A by design (verbatim no-structure substrate, same as flat).", |
| 354 | + "source": null, |
| 355 | + "source_type": "sme_measured" |
322 | 356 | } |
323 | 357 | }, |
324 | 358 | "5": { |
|
383 | 417 | "note": "karpathy_compiled (Karpathy condition D2 -- LLM-compiled wiki): Adapter wired (sme/conditions/karpathy_compiled.py + compile-wiki) as the Karpathy D2 control, but not yet run through any SME cat. No baseline reading exists for any cat -- WIRED but NOT YET BENCHED, an honest coverage gap, not a measured 0.", |
384 | 418 | "source": null, |
385 | 419 | "source_type": "not_run" |
| 420 | + }, |
| 421 | + "postgres_ingest": { |
| 422 | + "value": null, |
| 423 | + "scoreability": "N/A-by-design", |
| 424 | + "note": "no graph → no topology. N/A by design.", |
| 425 | + "source": null, |
| 426 | + "source_type": "sme_measured" |
386 | 427 | } |
387 | 428 | }, |
388 | 429 | "6": { |
|
441 | 482 | "note": "karpathy_compiled (Karpathy condition D2 -- LLM-compiled wiki): Adapter wired (sme/conditions/karpathy_compiled.py + compile-wiki) as the Karpathy D2 control, but not yet run through any SME cat. No baseline reading exists for any cat -- WIRED but NOT YET BENCHED, an honest coverage gap, not a measured 0.", |
442 | 483 | "source": null, |
443 | 484 | "source_type": "not_run" |
| 485 | + }, |
| 486 | + "postgres_ingest": { |
| 487 | + "value": 0.0, |
| 488 | + "scoreability": "N/A-by-design", |
| 489 | + "note": "no supersedes edges by construction → 0.00 supersession-completeness floor. Same as flat.", |
| 490 | + "source": null, |
| 491 | + "source_type": "sme_measured" |
444 | 492 | } |
445 | 493 | }, |
446 | 494 | "7": { |
|
507 | 555 | "note": "karpathy_compiled (Karpathy condition D2 -- LLM-compiled wiki): Adapter wired (sme/conditions/karpathy_compiled.py + compile-wiki) as the Karpathy D2 control, but not yet run through any SME cat. No baseline reading exists for any cat -- WIRED but NOT YET BENCHED, an honest coverage gap, not a measured 0.", |
508 | 556 | "source": null, |
509 | 557 | "source_type": "not_run" |
| 558 | + }, |
| 559 | + "postgres_ingest": { |
| 560 | + "value": null, |
| 561 | + "metric": "LoCoMo E2E QA accuracy (n=250 stratified, identical subset to flat)", |
| 562 | + "scoreability": "scoreable", |
| 563 | + "note": "postgres LoCoMo-10 E2E QA in flight (reader=judge=gpt-5.3-chat, exact 250-q subset flat used). Number lands shortly; compare to flat 0.384 unweighted / 0.4255 weighted.", |
| 564 | + "source": "https://github.com/techempower-org/multipass-structural-memory-eval/blob/main/baselines/locomo10_postgres_e2e_stratified_2026-05-31.json", |
| 565 | + "source_type": "sme_measured" |
510 | 566 | } |
511 | 567 | }, |
512 | 568 | "8": { |
|
572 | 628 | "note": "karpathy_compiled (Karpathy condition D2 -- LLM-compiled wiki): Adapter wired (sme/conditions/karpathy_compiled.py + compile-wiki) as the Karpathy D2 control, but not yet run through any SME cat. No baseline reading exists for any cat -- WIRED but NOT YET BENCHED, an honest coverage gap, not a measured 0.", |
573 | 629 | "source": null, |
574 | 630 | "source_type": "not_run" |
| 631 | + }, |
| 632 | + "postgres_ingest": { |
| 633 | + "value": null, |
| 634 | + "scoreability": "N/A-by-design", |
| 635 | + "note": "no declared ontology / no graph → nothing to check coherence against. N/A by design.", |
| 636 | + "source": null, |
| 637 | + "source_type": "sme_measured" |
575 | 638 | } |
576 | 639 | }, |
577 | 640 | "9a": { |
|
627 | 690 | "note": "karpathy_compiled (Karpathy condition D2 -- LLM-compiled wiki): Adapter wired (sme/conditions/karpathy_compiled.py + compile-wiki) as the Karpathy D2 control, but not yet run through any SME cat. No baseline reading exists for any cat -- WIRED but NOT YET BENCHED, an honest coverage gap, not a measured 0.", |
628 | 691 | "source": null, |
629 | 692 | "source_type": "not_run" |
| 693 | + }, |
| 694 | + "postgres_ingest": { |
| 695 | + "value": null, |
| 696 | + "scoreability": "N/A-no-harness", |
| 697 | + "note": "postgres_ingest exposes no harness manifest; driven directly. 9a is an orchestrator property regardless.", |
| 698 | + "source": null, |
| 699 | + "source_type": "sme_measured" |
630 | 700 | } |
631 | 701 | }, |
632 | 702 | "9b": { |
|
674 | 744 | "note": "karpathy_compiled (Karpathy condition D2 -- LLM-compiled wiki): Adapter wired (sme/conditions/karpathy_compiled.py + compile-wiki) as the Karpathy D2 control, but not yet run through any SME cat. No baseline reading exists for any cat -- WIRED but NOT YET BENCHED, an honest coverage gap, not a measured 0.", |
675 | 745 | "source": null, |
676 | 746 | "source_type": "not_run" |
| 747 | + }, |
| 748 | + "postgres_ingest": { |
| 749 | + "value": null, |
| 750 | + "scoreability": "N/A-no-harness", |
| 751 | + "note": "no get_harness_manifest().", |
| 752 | + "source": null, |
| 753 | + "source_type": "sme_measured" |
677 | 754 | } |
678 | 755 | } |
679 | 756 | }, |
|
702 | 779 | "mem0_note": "2026-05-31 (cassia-2, #146): Mem0-OSS slotted as a VERDICT row per Solara's #221 verdict. Retrieval/QA cats = verified + runnable, on-harness QA deferred (extraction-throughput-bound, ~9s/ingest warm -> ~18h strat150). Structural cats N/A because GRAPH MEMORY WAS REMOVED FROM mem0 OSS (zero edges) -- a distinct finding from Hindsight's no-endpoint. The matrix is now COMPLETE: mempalace + OMEGA fully scored; Hindsight + Mem0 verdict rows. Cross-system cost-wall finding now has BOTH extraction-cost numbers (Hindsight ~150h, Mem0 ~18h) vs verbatim-first mempalace ~0 marginal cost.", |
703 | 780 | "status": "FULL-FIELD — every survey system as a row, two column-groups (published_field = self-reported survey claims, NOT SME-measured; sme_multipass = our harness, real-or-not-benched). The 8-row SME matrix (categories block) is unchanged + canonical for the measured cells.", |
704 | 781 | "expansion_note": "2026-05-31 (cassia-2, #163): expanded from 4 systems to ALL harness-tested systems with honest per-cell coverage. mempalace Cat 4/5/8 flipped verdict->FINAL exact numbers (post re-map/DELETE/networkx). Added: flat (no-structure CONTROL -- real Cat 1/2c/7, structural N/A BY DESIGN), rlm (Cat 9a orchestrator arm -- the 46.7% invocation plateau, structural not-run), full_context (D1) + karpathy_compiled (D2) (Karpathy baselines, wired but not-run). oracle/random adapters present but not run as matrix cells (noted in doc). Every cell is a real reading OR an explicit marker (N/A-no-graph-endpoint / N/A-by-design / N/A-no-harness / not-run / emergent / verified-qa-deferred) -- NO fabricated numbers.", |
705 | | - "not_benched_footnote": "oracle_retrieval (ceiling) + random_retrieval (floor) adapters are present in the registry but were not run as matrix cells -- they are diagnostic BOUNDS, not products, and the reader_trueoracle_* baselines are a reader-config experiment, not an oracle-adapter Cat run. Footnoted, not given rows. (full_context D1 / karpathy_compiled D2 ARE shown as explicit wired-but-not-benched rows.)", |
| 782 | + "not_benched_footnote": "oracle_retrieval (ceiling) + random_retrieval (floor) adapters are present in the registry but were not run as matrix cells -- they are diagnostic BOUNDS, not products, and the reader_trueoracle_* baselines are a reader-config experiment, not an oracle-adapter Cat run. Footnoted, not given rows. (full_context D1 / karpathy_compiled D2 ARE shown as explicit wired-but-not-benched rows.) Also wired-but-not-benched: longhand (Mode-B Claude-Code session archive; ingest_corpus NotImplementedError — no path to load an SME corpus) and ladybugdb (typed-graph reader; reader-only, no .ldb exists, no in-harness ingest path). [#164, Twilight]", |
706 | 783 | "distinct_na_structural_reasons": { |
707 | 784 | "flat": "N/A BY DESIGN -- no-structure control; there is intentionally no graph (it's the baseline the structural delta is measured against).", |
708 | 785 | "hindsight": "N/A NO GRAPH ENDPOINT -- extraction-then-retrieve; serves no standalone graph API at all.", |
|
982 | 1059 | "arch_category": "compile-upstream" |
983 | 1060 | }, |
984 | 1061 | "longhand": { |
985 | | - "status": "bench_in_flight", |
| 1062 | + "status": "wired_not_benched", |
986 | 1063 | "display": "Longhand", |
987 | 1064 | "published_field": { |
988 | 1065 | "longmemeval_qa": { |
|
1004 | 1081 | "verification": "none (no published benchmarks)", |
1005 | 1082 | "source": "memorypalace/docs/research/2026-05-24-memory-system-benchmarks.md (2026-05-24) (Longhand: 'No benchmarks published', line 85/128)", |
1006 | 1083 | "disclaimer": "SELF-REPORTED by the system's team / survey — NOT SME-measured. Metric/model/judge/subset vary; see comparability_caveats.", |
1007 | | - "note": "Verbatim-first cohort. SME multipass bench IN FLIGHT (Twilight, #164) — real cells drop in when she reports. Longhand publishes no quantitative benchmarks." |
| 1084 | + "note": "Verbatim-first Claude-Code session archive (Wynelson94/longhand, pip 0.9.3). WIRED into the SME registry but UN-PROVISIONABLE on this harness: the adapter is Mode-B diagnostic-only — ingest_corpus raises NotImplementedError; Longhand only ingests Claude Code session JSONL via its own Stop/SessionEnd hooks, never arbitrary corpora, so there is NO path to load jp-realm/LoCoMo into a longhand store. Footnoted wired-not-benched, alongside Karpathy D1/D2." |
1008 | 1085 | }, |
1009 | 1086 | "sme_status": "bench-in-flight (#164, Twilight)", |
1010 | 1087 | "architecture": "verbatim-first (no published benchmarks)", |
1011 | 1088 | "sme_source": { |
1012 | 1089 | "source": null, |
1013 | | - "source_type": "bench_in_flight" |
| 1090 | + "source_type": "wired_not_benched" |
1014 | 1091 | }, |
1015 | 1092 | "arch_category": "verbatim-first" |
1016 | 1093 | }, |
1017 | 1094 | "postgres_ingest": { |
1018 | | - "status": "bench_in_flight", |
| 1095 | + "status": "benched_on_multipass", |
1019 | 1096 | "display": "postgres_ingest", |
1020 | 1097 | "published_field": { |
1021 | 1098 | "longmemeval_qa": { |
1022 | | - "value": "--", |
1023 | | - "source": null, |
1024 | | - "source_type": "none" |
| 1099 | + "value": "R@5 0.966 (LongMemEval-S substrate-parity, n=500; retrieval-only, no E2E QA)", |
| 1100 | + "source": "https://github.com/techempower-org/multipass-structural-memory-eval/blob/main/baselines/lme_substrate_postgres_2026-05-17.json", |
| 1101 | + "source_type": "sme_measured" |
1025 | 1102 | }, |
1026 | 1103 | "locomo_qa": { |
1027 | 1104 | "value": "--", |
|
1037 | 1114 | "verification": "none (no published benchmarks)", |
1038 | 1115 | "source": "n/a (SME-internal verbatim-first adapter)", |
1039 | 1116 | "disclaimer": "SELF-REPORTED by the system's team / survey — NOT SME-measured. Metric/model/judge/subset vary; see comparability_caveats.", |
1040 | | - "note": "Verbatim-first cohort. SME multipass bench IN FLIGHT (Twilight, #164) — real cells drop in when she reports. " |
| 1117 | + "note": "Verbatim-first cohort, the 'upstream MemPalace raw' ablation — mempalace's own postgres storage WITHOUT the palace graph. SME multipass benched (#164): jp-realm Cond-A Cat 1/2c = 0.833 (IDENTICAL to flat) + LoCoMo Cat 7. R@5 0.966 substrate-parity held as the published_field retrieval number." |
1041 | 1118 | }, |
1042 | | - "sme_status": "bench-in-flight (#164, Twilight)", |
1043 | | - "architecture": "verbatim-first (postgres ingest)", |
| 1119 | + "sme_status": "benched (#164, Twilight)", |
| 1120 | + "architecture": "verbatim-first (postgres+pgvector ingest; same MiniLM embedding as flat — backend-swap ablation of mempalace’s own raw store)", |
1044 | 1121 | "sme_source": { |
1045 | | - "source": null, |
1046 | | - "source_type": "bench_in_flight" |
| 1122 | + "source": "https://github.com/techempower-org/multipass-structural-memory-eval/blob/main/baselines/jp_realm_v0_1_postgres_condA_2026-05-31.json (+ baselines/locomo10_postgres_e2e_stratified_2026-05-31.json)", |
| 1123 | + "source_type": "sme_measured" |
1047 | 1124 | }, |
1048 | 1125 | "arch_category": "verbatim-first" |
1049 | 1126 | }, |
1050 | 1127 | "ladybugdb": { |
1051 | | - "status": "bench_in_flight", |
| 1128 | + "status": "wired_not_benched", |
1052 | 1129 | "display": "LadybugDB", |
1053 | 1130 | "published_field": { |
1054 | 1131 | "longmemeval_qa": { |
|
1070 | 1147 | "verification": "none (no published benchmarks)", |
1071 | 1148 | "source": "n/a (SME-internal verbatim-first adapter)", |
1072 | 1149 | "disclaimer": "SELF-REPORTED by the system's team / survey — NOT SME-measured. Metric/model/judge/subset vary; see comparability_caveats.", |
1073 | | - "note": "Verbatim-first cohort. SME multipass bench IN FLIGHT (Twilight, #164) — real cells drop in when she reports. " |
| 1150 | + "note": "Schema-agnostic LadybugDB graph reader (real_ladybug, pip 0.15.3, cp312 wheel). NOT verbatim-first — a typed-graph reader (grouped with the verbatim cohort only for the #164 sweep). WIRED into the registry but UN-PROVISIONABLE on this harness: the adapter is a READER only (ingest_corpus raises NotImplementedError); it needs an existing .ldb or a live /search API. No SME .ldb exists, and building one needs the target project's indexer pipeline (not checked out). Footnoted wired-not-benched." |
1074 | 1151 | }, |
1075 | 1152 | "sme_status": "bench-in-flight (#164, Twilight)", |
1076 | 1153 | "architecture": "verbatim-first (LadybugDB)", |
1077 | 1154 | "sme_source": { |
1078 | 1155 | "source": null, |
1079 | | - "source_type": "bench_in_flight" |
| 1156 | + "source_type": "wired_not_benched" |
1080 | 1157 | }, |
1081 | | - "arch_category": "verbatim-first" |
| 1158 | + "arch_category": "hybrid" |
1082 | 1159 | }, |
1083 | 1160 | "mem0_platform_v3": { |
1084 | 1161 | "status": "published_field_only", |
|
1825 | 1902 | "flat", |
1826 | 1903 | "rlm", |
1827 | 1904 | "full_context", |
1828 | | - "karpathy_compiled" |
1829 | | - ], |
1830 | | - "bench_in_flight": [ |
1831 | | - "longhand", |
1832 | | - "postgres_ingest", |
1833 | | - "ladybugdb" |
| 1905 | + "karpathy_compiled", |
| 1906 | + "postgres_ingest" |
1834 | 1907 | ], |
| 1908 | + "bench_in_flight": [], |
1835 | 1909 | "published_field_only": [ |
1836 | 1910 | "mem0_platform_v3", |
1837 | 1911 | "mastra", |
|
0 commit comments