{
  "section": "experiments",
  "kind": "experiment",
  "canonical": "https://evemisslab.com/ai/experiments/",
  "count": 32,
  "records": [
    {
      "id": "EXP-2026-0023",
      "kind": "experiment",
      "label": "PACC-Hybrid v0.2 — first real-model run, on a local 9B open-weight model",
      "created_at": "2026-09-11",
      "updated_at": "2026-09-11",
      "values": {
        "eml_status": "STABLE",
        "eml_evidence_level": "E2",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0023/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Adaptive Epistemic Systems (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each lab's own result reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_aes/extract.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "The v0.2 protocol executed for the first time on a real language model — a local open-weight 9B (Qwythos-9B-v2, Q4_K_M, Ollama, thinking off) as generator, selector and judge; 16 tasks × 2 repetitions × 4 candidates, 380 calls, 32 rows per condition. The PACC runtime (C) gains on supersession alignment (+0.031 vs B, +0.097 vs A) and repair success (+0.070 / +0.094), the two axes the canonical-intent compilation step exists for; derived coherence (+0.011) and intent persistence (−0.002) do not move; valid novelty is slightly lower (−0.045 vs B). Most per-task pairs are ties because the 9B judge saturates near 1.0, and with two repetitions per task the judge's free-text pattern labels never repeat, so creative breadth is not measurable. The conditions chose different candidates in 69 % of task × repetition pairs.",
        "eml_summary_zh": "v0.2 協定第一次在真實語言模型上執行——本地開放權重 9B（Qwythos-9B-v2、Q4_K_M、Ollama、關閉思考）同時當生成器、選擇器與評審；16 題 × 2 次 × 4 候選，380 次呼叫，每條件 32 筆。PACC runtime（C）在 supersession 對齊（相對 B +0.031、相對 A +0.097）與修復成功率（+0.070／+0.094）上升——正是 canonical intent 編譯步驟存在的那兩個軸；衍生一致性（+0.011）與意圖持續（−0.002）沒有動；有效新穎度略低（相對 B −0.045）。多數逐題配對是平手，因為 9B 評審在接近 1.0 處飽和；每題只重複兩次，評審的自由文字 pattern 標籤從不重複，所以創造廣度量不出來。三個條件在 69 % 的題 × 次配對中選了不同的候選。",
        "eml_label_zh": "PACC-Hybrid v0.2——第一次真實模型執行，本地 9B 開放權重模型",
        "eml_primary_domain": "Reasoning",
        "eml_domains": [
          "Evaluation"
        ],
        "eml_program_id": "PRG-2026-0001",
        "eml_data_basis": "REAL MODEL",
        "eml_hypothesis": "A real language model under the PACC runtime shows the coherence / valid-novelty gains and recoverable breadth loss seen in the synthetic witness (v0.2 predeclared interpretation).",
        "eml_metrics": {
          "verdict": "REAL_LOCAL_9B_MIXED: supersession and repair up, coherence flat, valid novelty slightly down, breadth unmeasurable at two repetitions",
          "execution_status": "EXECUTED_REAL_MODEL",
          "protocol": {
            "model": "qwythos-9b-v2-q4km-ctx8k:latest",
            "judge_model": "qwythos-9b-v2-q4km-ctx8k:latest",
            "task_count": 16,
            "repetitions": 2,
            "candidate_count": 4,
            "equal_accounted_calls": true,
            "architecture_call_counts": {
              "A_llm_only": 224,
              "B_hard_verifier": 224,
              "C_pacc_runtime": 224
            }
          },
          "rows_per_architecture": 32,
          "overall": {
            "A_llm_only": {
              "hard_adherence": 0.9875,
              "derived_coherence": 0.9812,
              "intent_persistence": 0.9875,
              "supersession_alignment": 0.8094,
              "repair_success": 0.775,
              "usefulness": 0.9528,
              "semantic_novelty": 0.8084,
              "valid_novelty": 0.8475,
              "literal_check_mean": 0.9792,
              "within_task_pattern_entropy_mean": 1.0
            },
            "B_hard_verifier": {
              "hard_adherence": 0.9859,
              "derived_coherence": 0.9797,
              "intent_persistence": 0.9875,
              "supersession_alignment": 0.875,
              "repair_success": 0.7984,
              "usefulness": 0.9503,
              "semantic_novelty": 0.7725,
              "valid_novelty": 0.8606,
              "literal_check_mean": 0.9792,
              "within_task_pattern_entropy_mean": 1.0
            },
            "C_pacc_runtime": {
              "hard_adherence": 1.0,
              "derived_coherence": 0.9906,
              "intent_persistence": 0.9853,
              "supersession_alignment": 0.9062,
              "repair_success": 0.8688,
              "usefulness": 0.9516,
              "semantic_novelty": 0.7897,
              "valid_novelty": 0.8153,
              "literal_check_mean": 0.9792,
              "within_task_pattern_entropy_mean": 1.0
            }
          },
          "deltas": {
            "C-B": {
              "hard_adherence": 0.0141,
              "derived_coherence": 0.0109,
              "intent_persistence": -0.0022,
              "supersession_alignment": 0.0312,
              "repair_success": 0.0703,
              "usefulness": 0.0012,
              "semantic_novelty": 0.0172,
              "valid_novelty": -0.0453,
              "literal_check_mean": 0.0,
              "within_task_pattern_entropy_mean": 0.0
            },
            "C-A": {
              "hard_adherence": 0.0125,
              "derived_coherence": 0.0094,
              "intent_persistence": -0.0022,
              "supersession_alignment": 0.0969,
              "repair_success": 0.0938,
              "usefulness": -0.0012,
              "semantic_novelty": -0.0188,
              "valid_novelty": -0.0322,
              "literal_check_mean": 0.0,
              "within_task_pattern_entropy_mean": 0.0
            },
            "B-A": {
              "hard_adherence": -0.0016,
              "derived_coherence": -0.0016,
              "intent_persistence": 0.0,
              "supersession_alignment": 0.0656,
              "repair_success": 0.0234,
              "usefulness": -0.0025,
              "semantic_novelty": -0.0359,
              "valid_novelty": 0.0131,
              "literal_check_mean": 0.0,
              "within_task_pattern_entropy_mean": 0.0
            }
          },
          "selection_agreement": {
            "A=B": 0.5625,
            "A=C": 0.40625,
            "B=C": 0.46875,
            "all_same": 0.3125
          },
          "usage": {
            "input_tokens": 350535,
            "latency_ms_sum": 3633534.6304999674,
            "output_tokens": 108991
          },
          "wall_seconds": 3370.0,
          "smoke_run": "6 tasks × 2 × 3 candidates executed first, 128 calls, all outputs parsed; kept in the bundle"
        },
        "eml_interpretation": "Against the v0.2 predeclared interpretation: the predicted coherence and intent-persistence gains over B are not observed; raw novelty did not decrease (semantic novelty +0.017 vs B); the predicted breadth collapse cannot be tested at this repetition count. What did appear — governance gains on supersession and repair with a valid-novelty cost concentrated in multi_constraint and repair tasks — is mechanism-consistent but small, untested statistically, and runs opposite to the synthetic v0.1 valid-novelty picture (+0.196 there). One model, one run, one same-model judge: a first real data point, not a verdict on the architecture.",
        "eml_limitations": [
          "One open-weight 9B model at 4-bit, one run, 32 rows per architecture; no significance or equivalence test — deltas are descriptive.",
          "Judge = the same 9B model; no human rating, no second judge; pattern labels are noisy, so entropy is fragile.",
          "Thinking disabled for every call (see docs/REAL_LOCAL_RUN_EVIDENCE_BOUNDARY.md); a thinking-enabled run is a different experiment.",
          "Says nothing about frontier models.",
          "Not replicated: the second run with four repetitions (EXP-2026-0024, 64 rows per condition) shows supersession +0.0005 and repair −0.0125 vs B — the run-1 gains were run-to-run variation.",
          "The thinking-enabled run exists now: EXP-2026-0025 (2026-09-11)."
        ],
        "eml_controls": [
          "identical candidate ledger for A/B/C",
          "equal accounted calls",
          "condition-blind judge with deduplicated judge calls",
          "deterministic literal checks"
        ],
        "eml_random_seeds": [
          "model nondeterminism, single run; frozen response cache in the bundle for exact replay"
        ],
        "eml_run_count": 1,
        "eml_result_type": "MIXED",
        "eml_procedure": "scripts/run_real_local_ollama.py --candidate-count 4 --repetitions 2 (after a 6×2×3 smoke); summary by scripts/summarize_real_local.py; both scripts and both frozen caches are in the bundle.",
        "eml_software_environment": "Python 3.14, openai SDK 3.0.0 against Ollama 0.33.3 /v1/responses; PACC-Hybrid-Lab v0.2 package unchanged (25 tests green before the run).",
        "eml_reproduction_instructions": "Extract the bundle; python -m pytest -q; replay exactly with CachedReplayProvider('.pacc_real_cache_local', reasoning_effort='none'); or rerun scripts/run_real_local_ollama.py against any OpenAI-compatible endpoint serving the same model tag.",
        "eml_completed_at": "2026-09-11",
        "eml_model_ids": [
          "MOD-2026-0006"
        ],
        "eml_benchmark_ids": [
          "BEN-2026-0003"
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Sol (GPT-5.6, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0023/",
      "json": "/ai/experiments/EXP-2026-0023/index.json"
    },
    {
      "id": "EXP-2026-0024",
      "kind": "experiment",
      "label": "PACC-Hybrid v0.2 — real-model run 2: four repetitions and label-free creative breadth",
      "created_at": "2026-09-11",
      "updated_at": "2026-09-11",
      "values": {
        "eml_status": "STABLE",
        "eml_evidence_level": "E3",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0024/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Adaptive Epistemic Systems (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each lab's own result reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_aes/extract.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "Same model and settings as the first run, repetitions raised from two to four: 16 tasks × 4 × 4 candidates, 754 unique model requests, 64 rows per condition. The PACC runtime (C) shows no reliable advantage on any judge axis — supersession +0.0005 and repair -0.0125 vs B, so the first run's gains did not replicate — and sits a few hundredths below A and B on adherence, coherence and intent persistence. A new label-free breadth measure (local nomic-embed-text embeddings, k-means labels over each task's candidate pool) finds no collapse: C's cluster entropy 0.923 vs A 0.909 / B 0.864, breadth ratio 0.975 vs 0.948 / 0.941. Five selector/judge outputs failed strict JSON parsing and were regenerated under a disclosed retry policy.",
        "eml_summary_zh": "與第一次執行相同的模型與設定，重複次數從 2 提高到 4：16 題 × 4 × 4 候選，754 次唯一模型請求，每條件 64 筆。PACC runtime（C）在任何評審軸上都沒有可靠優勢——supersession 相對 B +0.0005、修復 -0.0125，第一次執行的增益沒有重現——在遵守、一致性與意圖持續上還比 A、B 低幾個百分點。新的標籤無關廣度指標（本地 nomic-embed-text 嵌入、對每題候選池做 k-means 當標籤）沒有發現塌縮：C 的群熵 0.923，A 0.909／B 0.864；廣度比 0.975，對 0.948／0.941。五次 selector／judge 輸出未通過嚴格 JSON 解析，依公開的重試政策重新生成。",
        "eml_label_zh": "PACC-Hybrid v0.2——真實模型第二次執行：四次重複與標籤無關的創造廣度",
        "eml_primary_domain": "Reasoning",
        "eml_domains": [
          "Evaluation"
        ],
        "eml_program_id": "PRG-2026-0001",
        "eml_data_basis": "REAL MODEL",
        "eml_hypothesis": "With enough repetitions, (a) the first run's supersession/repair gains for the PACC runtime replicate, and (b) creative breadth can be measured — and the predeclared breadth collapse under PACC selection appears.",
        "eml_metrics": {
          "verdict": "REAL_LOCAL_9B_NO_RELIABLE_DIFFERENCE_BREADTH_NOT_REDUCED",
          "execution_status": "EXECUTED_REAL_MODEL",
          "protocol": {
            "model": "qwythos-9b-v2-q4km-ctx8k:latest",
            "judge_model": "qwythos-9b-v2-q4km-ctx8k:latest",
            "task_count": 16,
            "repetitions": 4,
            "candidate_count": 4,
            "equal_accounted_calls": true,
            "architecture_call_counts": {
              "A_llm_only": 448,
              "B_hard_verifier": 448,
              "C_pacc_runtime": 448
            }
          },
          "rows_per_architecture": 64,
          "overall": {
            "A_llm_only": {
              "hard_adherence": 0.9894,
              "derived_coherence": 0.9942,
              "intent_persistence": 0.9863,
              "supersession_alignment": 0.9767,
              "repair_success": 0.9798,
              "usefulness": 0.9427,
              "semantic_novelty": 0.8267,
              "valid_novelty": 0.8691,
              "literal_check_mean": 0.974
            },
            "B_hard_verifier": {
              "hard_adherence": 0.9816,
              "derived_coherence": 0.977,
              "intent_persistence": 0.982,
              "supersession_alignment": 0.9469,
              "repair_success": 0.95,
              "usefulness": 0.9355,
              "semantic_novelty": 0.8044,
              "valid_novelty": 0.9114,
              "literal_check_mean": 0.9792
            },
            "C_pacc_runtime": {
              "hard_adherence": 0.9672,
              "derived_coherence": 0.9695,
              "intent_persistence": 0.9706,
              "supersession_alignment": 0.9473,
              "repair_success": 0.9375,
              "usefulness": 0.9234,
              "semantic_novelty": 0.8161,
              "valid_novelty": 0.8927,
              "literal_check_mean": 0.974
            }
          },
          "deltas": {
            "C-B": {
              "hard_adherence": -0.0144,
              "derived_coherence": -0.0075,
              "intent_persistence": -0.0114,
              "supersession_alignment": 0.0005,
              "repair_success": -0.0125,
              "usefulness": -0.012,
              "semantic_novelty": 0.0117,
              "valid_novelty": -0.0187,
              "literal_check_mean": -0.0052
            },
            "C-A": {
              "hard_adherence": -0.0222,
              "derived_coherence": -0.0247,
              "intent_persistence": -0.0157,
              "supersession_alignment": -0.0294,
              "repair_success": -0.0423,
              "usefulness": -0.0192,
              "semantic_novelty": -0.0106,
              "valid_novelty": 0.0236,
              "literal_check_mean": 0.0
            },
            "B-A": {
              "hard_adherence": -0.0078,
              "derived_coherence": -0.0172,
              "intent_persistence": -0.0043,
              "supersession_alignment": -0.0298,
              "repair_success": -0.0298,
              "usefulness": -0.0072,
              "semantic_novelty": -0.0223,
              "valid_novelty": 0.0423,
              "literal_check_mean": 0.0052
            }
          },
          "selection_agreement": {
            "A=B": 0.546875,
            "A=C": 0.546875,
            "B=C": 0.46875,
            "all_same": 0.34375
          },
          "paired_wins_ties_losses": {
            "C-B": {
              "hard_adherence": "1-62-1",
              "derived_coherence": "2-60-2",
              "intent_persistence": "0-61-3",
              "supersession_alignment": "3-57-4",
              "repair_success": "3-59-2",
              "usefulness": "10-41-13",
              "semantic_novelty": "14-39-11",
              "valid_novelty": "8-46-10"
            },
            "C-A": {
              "hard_adherence": "2-59-3",
              "derived_coherence": "1-58-5",
              "intent_persistence": "2-57-5",
              "supersession_alignment": "3-57-4",
              "repair_success": "2-59-3",
              "usefulness": "10-44-10",
              "semantic_novelty": "14-37-13",
              "valid_novelty": "11-48-5"
            }
          },
          "breadth_label_free": {
            "A_llm_only": {
              "cluster_entropy_mean": 0.9091,
              "breadth_ratio_mean": 0.9475,
              "selected_mean_pairwise_distance_mean": 0.1909
            },
            "B_hard_verifier": {
              "cluster_entropy_mean": 0.8635,
              "breadth_ratio_mean": 0.9411,
              "selected_mean_pairwise_distance_mean": 0.1921
            },
            "C_pacc_runtime": {
              "cluster_entropy_mean": 0.9227,
              "breadth_ratio_mean": 0.9745,
              "selected_mean_pairwise_distance_mean": 0.1964
            }
          },
          "breadth_method": "nomic-embed-text:latest embeddings; k-means k=4 over each task's 16-candidate pool; normalized cluster entropy and mean pairwise cosine distance / pool distance",
          "retries": [
            "judge:repair_02:r1:c1",
            "judge:design_01:r0:c0",
            "selector:A_llm_only:design_01:r3",
            "judge:design_03:r1:c2",
            "judge:creative_03:r0:c2"
          ],
          "usage": {
            "input_tokens": 718454,
            "latency_ms_sum": 8037466.647999942,
            "output_tokens": 223944
          },
          "wall_seconds": 3400.5
        },
        "eml_interpretation": "Two thinking-off runs on the same 9B model (32 and 64 rows per condition) now disagree on the only gains the first run showed, so those gains were run-to-run variation of a same-model judge, not an effect. The predeclared coherence and intent gains are absent in both runs. The predeclared breadth collapse is not observed by either label-free measure — the shipped C selector prompt already instructs against collapsing, so this tests the shipped prompt, not naive commitment. On this model the three runtime conditions are practically equivalent; nothing is statistically tested; frontier models are not addressed.",
        "eml_limitations": [
          "Same-model 9B judge, saturating near 1.0; no human rating, no second judge.",
          "The breadth measure is supplementary and label-free, not the protocol's judge-label entropy (which stays 1.0 because free-text labels never repeat).",
          "Retry policy: unparseable selector/judge JSON regenerated at most twice per call, never edited; 5 retries recorded.",
          "Thinking disabled; one model family; no significance or equivalence test."
        ],
        "eml_controls": [
          "identical candidate ledger for A/B/C",
          "equal accounted calls",
          "condition-blind judge with deduplicated judge calls",
          "deterministic literal checks",
          "pool-relative breadth ratio"
        ],
        "eml_random_seeds": [
          "model nondeterminism, single run; frozen response cache and embedding cache in the bundle"
        ],
        "eml_run_count": 1,
        "eml_result_type": "NEGATIVE",
        "eml_procedure": "scripts/run_real_local_ollama.py --model qwythos-9b-v2-q4km-ctx8k --repetitions 4 --candidate-count 4 (resumed once from cache after a malformed judge JSON aborted the first pass at call 398); scripts/summarize_real_local.py; scripts/breadth_metrics.py.",
        "eml_software_environment": "Python 3.14, openai SDK 3.0.0 against Ollama 0.33.3 /v1/responses (flash attention on, q8_0 KV cache for the resumed pass); nomic-embed-text for breadth; harness package unchanged.",
        "eml_reproduction_instructions": "Extract the bundle; replay exactly with CachedReplayProvider('.pacc_real_cache_local_reps4', reasoning_effort='none'); python scripts/breadth_metrics.py results/pacc_hybrid_v0.2_real_local_reps4.json --cache-dir .pacc_real_cache_local_reps4 (embeddings cached alongside).",
        "eml_completed_at": "2026-09-11",
        "eml_model_ids": [
          "MOD-2026-0006"
        ],
        "eml_benchmark_ids": [
          "BEN-2026-0003"
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Sol (GPT-5.6, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0024/",
      "json": "/ai/experiments/EXP-2026-0024/index.json"
    },
    {
      "id": "EXP-2026-0025",
      "kind": "experiment",
      "label": "PACC-Hybrid v0.2 — real-model run 3: thinking enabled with enlarged output budgets",
      "created_at": "2026-09-11",
      "updated_at": "2026-09-11",
      "values": {
        "eml_status": "STABLE",
        "eml_evidence_level": "E2",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0025/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Adaptive Epistemic Systems (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each lab's own result reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_aes/extract.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "The same local 9B model with its thinking enabled (reasoning.effort = medium) and 3,500 extra output tokens on every call so the hidden reasoning has room; 12k context; 16 tasks × 2 repetitions × 4 candidates, 380 unique requests, 32 rows per condition, 3.7 h. The PACC runtime (C) has the highest mean valid novelty (0.8375; +0.0687 vs B, +0.0375 vs A) and mean repair success (+0.0153 / +0.0466) — but per task × repetition the record is even (valid novelty wins–ties–losses 6-21-5 vs B, 5-20-7 vs A; repair 1-29-2 vs B), so the means come from a few large single-task differences in multi_constraint and repair tasks — while adherence, coherence, intent persistence and supersession are flat to a few thousandths lower (-0.0069, -0.0063, -0.0053, -0.0059 vs B). All three conditions pass every deterministic literal check with thinking on. Label-free breadth is again not reduced under C: cluster entropy 0.875 vs A 0.750 / B 0.688, breadth ratio 1.047 vs 0.960 / 0.973. Two selector outputs were regenerated under the disclosed retry policy (one cut off, one empty after the reasoning consumed its whole 3,800-token budget).",
        "eml_summary_zh": "同一顆本地 9B 模型，開啟思考（reasoning.effort = medium），每次呼叫多給 3,500 個輸出 token 讓隱藏推理有空間；12k context；16 題 × 2 次 × 4 候選，380 次唯一請求，每條件 32 筆，3.7 小時。PACC runtime（C）的平均有效新穎度最高（0.8375；相對 B +0.0687、相對 A +0.0375），平均修復成功率也最高（+0.0153／+0.0466）——但逐題 × 次配對的戰績是平的（有效新穎度勝–平–負：對 B 6-21-5、對 A 5-20-7；修復對 B 1-29-2），平均差來自 multi_constraint 與 repair 題裡少數幾個大差距——遵守、一致性、意圖持續與 supersession 則持平到低幾個千分點（相對 B -0.0069、-0.0063、-0.0053、-0.0059）。開啟思考後三個條件全部通過所有確定性字面檢查。標籤無關廣度在 C 下再次沒有縮減：群熵 0.875，A 0.750／B 0.688；廣度比 1.047，對 0.960／0.973。兩次 selector 輸出依公開的重試政策重新生成（一次被截斷、一次推理吃光 3,800 token 預算後輸出為空）。",
        "eml_label_zh": "PACC-Hybrid v0.2——真實模型第三次執行：開啟思考並放大輸出預算",
        "eml_primary_domain": "Reasoning",
        "eml_domains": [
          "Evaluation"
        ],
        "eml_program_id": "PRG-2026-0001",
        "eml_data_basis": "REAL MODEL",
        "eml_hypothesis": "With the model's own reasoning enabled (the setting runs 1–2 had to switch off), the v0.2 predeclared picture — coherence and intent-persistence gains plus higher valid novelty for the PACC runtime, with a recoverable breadth loss — appears.",
        "eml_metrics": {
          "verdict": "REAL_LOCAL_9B_THINKING_MIXED: C's mean valid novelty highest (+0.069 vs B) but per-pair even (6-21-5); mean repair up on a couple of tasks; governance axes flat to slightly down; breadth not reduced; 32 rows",
          "execution_status": "EXECUTED_REAL_MODEL",
          "paired_wins_ties_losses": {
            "C-B": {
              "hard_adherence": "1-28-3",
              "derived_coherence": "1-26-5",
              "intent_persistence": "0-29-3",
              "supersession_alignment": "1-27-4",
              "repair_success": "1-29-2",
              "usefulness": "5-22-5",
              "semantic_novelty": "7-18-7",
              "valid_novelty": "6-21-5"
            },
            "C-A": {
              "hard_adherence": "0-29-3",
              "derived_coherence": "2-26-4",
              "intent_persistence": "0-30-2",
              "supersession_alignment": "1-28-3",
              "repair_success": "2-28-2",
              "usefulness": "6-22-4",
              "semantic_novelty": "8-18-6",
              "valid_novelty": "5-20-7"
            }
          },
          "protocol": {
            "model": "qwythos-9b-v2-q4km-ctx12k:latest",
            "judge_model": "qwythos-9b-v2-q4km-ctx12k:latest",
            "task_count": 16,
            "repetitions": 2,
            "candidate_count": 4,
            "equal_accounted_calls": true,
            "architecture_call_counts": {
              "A_llm_only": 224,
              "B_hard_verifier": 224,
              "C_pacc_runtime": 224
            }
          },
          "settings": {
            "reasoning_effort": "medium",
            "budget_add_tokens": 3500,
            "model_tag": "qwythos-9b-v2-q4km-ctx12k:latest",
            "model_digest": "7ffdc602f28799ffa312ea8dc85c364b047e1386380c4616c21b02c96c3f85d5"
          },
          "rows_per_architecture": 32,
          "overall": {
            "A_llm_only": {
              "hard_adherence": 0.9956,
              "derived_coherence": 0.9903,
              "intent_persistence": 0.9956,
              "supersession_alignment": 0.9584,
              "repair_success": 0.9,
              "usefulness": 0.9759,
              "semantic_novelty": 0.8172,
              "valid_novelty": 0.8,
              "literal_check_mean": 1.0
            },
            "B_hard_verifier": {
              "hard_adherence": 0.9988,
              "derived_coherence": 0.9953,
              "intent_persistence": 0.9969,
              "supersession_alignment": 0.9609,
              "repair_success": 0.9313,
              "usefulness": 0.9712,
              "semantic_novelty": 0.8134,
              "valid_novelty": 0.7688,
              "literal_check_mean": 1.0
            },
            "C_pacc_runtime": {
              "hard_adherence": 0.9919,
              "derived_coherence": 0.9891,
              "intent_persistence": 0.9916,
              "supersession_alignment": 0.955,
              "repair_success": 0.9466,
              "usefulness": 0.9756,
              "semantic_novelty": 0.83,
              "valid_novelty": 0.8375,
              "literal_check_mean": 1.0
            }
          },
          "deltas": {
            "C-B": {
              "hard_adherence": -0.0069,
              "derived_coherence": -0.0063,
              "intent_persistence": -0.0053,
              "supersession_alignment": -0.0059,
              "repair_success": 0.0153,
              "usefulness": 0.0044,
              "semantic_novelty": 0.0166,
              "valid_novelty": 0.0687,
              "literal_check_mean": 0.0
            },
            "C-A": {
              "hard_adherence": -0.0038,
              "derived_coherence": -0.0013,
              "intent_persistence": -0.0041,
              "supersession_alignment": -0.0034,
              "repair_success": 0.0466,
              "usefulness": -0.0003,
              "semantic_novelty": 0.0128,
              "valid_novelty": 0.0375,
              "literal_check_mean": 0.0
            },
            "B-A": {
              "hard_adherence": 0.0031,
              "derived_coherence": 0.005,
              "intent_persistence": 0.0012,
              "supersession_alignment": 0.0025,
              "repair_success": 0.0312,
              "usefulness": -0.0047,
              "semantic_novelty": -0.0037,
              "valid_novelty": -0.0312,
              "literal_check_mean": 0.0
            }
          },
          "selection_agreement": {
            "A=B": 0.5625,
            "A=C": 0.46875,
            "B=C": 0.46875,
            "all_same": 0.34375
          },
          "breadth_label_free": {
            "A_llm_only": {
              "cluster_entropy_mean": 0.75,
              "breadth_ratio_mean": 0.9597,
              "selected_mean_pairwise_distance_mean": 0.2079
            },
            "B_hard_verifier": {
              "cluster_entropy_mean": 0.6875,
              "breadth_ratio_mean": 0.9727,
              "selected_mean_pairwise_distance_mean": 0.2087
            },
            "C_pacc_runtime": {
              "cluster_entropy_mean": 0.875,
              "breadth_ratio_mean": 1.0474,
              "selected_mean_pairwise_distance_mean": 0.2275
            }
          },
          "breadth_method": "nomic-embed-text:latest embeddings; k-means k=4 over each task's 8-candidate pool; normalized cluster entropy and mean pairwise cosine distance / pool distance",
          "retries": [
            "selector:A_llm_only:design_02:r1",
            "selector:B_hard_verifier:creative_02:r0"
          ],
          "usage": {
            "input_tokens": 246795,
            "latency_ms_sum": 14828453.861500219,
            "output_tokens": 498227
          },
          "wall_seconds": 13352.2,
          "post_run_corrections": [
            {
              "field": "local_run.deviation_from_default_primary",
              "now": "local 9B open-weight model instead of gpt-5.6-luna; judge = same local model; thinking ENABLED (reasoning.effort=medium) with +3500 output tokens added to every call's budget; num_ctx 12288 tag",
              "reason": "the run script carried the run-1 wording as a hardcoded label; reasoning_effort and budget_add_tokens in this block were always correct; no data, metric or model-output field was touched",
              "was": "local 9B open-weight model instead of gpt-5.6-luna; judge = same local model; thinking disabled",
              "when": "2026-09-11 11:35 +08:00, before sealing"
            }
          ]
        },
        "eml_interpretation": "Half of the predeclared picture shows up in the means once the model can reason: the PACC runtime's valid novelty is the highest of the three (+0.069 vs the hard verifier, +0.038 vs the plain model — same direction as the synthetic v0.1 witness's +0.196, at a third of the size) and mean repair improves, concentrated in multi_constraint and repair tasks — but the per-pair record is even (6–21–5 on valid novelty vs B, 5–20–7 vs A; repair 1–29–2), so this is a few large single-task wins, not a consistent shift. The other half does not: coherence, intent persistence and supersession are flat to slightly lower, and creative breadth is not reduced by either label-free measure (it is widest under C). With thinking on, every condition passes every literal check, so the deterministic checks stop discriminating and the whole table rests on the same-model judge. Thirty-two rows, one run — run 1's gains of the same size vanished at 64 rows, so this valid-novelty gain is a candidate effect until a repetition at four or more repetitions, and it says nothing about frontier models.",
        "eml_limitations": [
          "32 rows per condition, single run; run 1's gains of similar size did not survive four repetitions (EXP-2026-0024), so treat the valid-novelty gain as unreplicated.",
          "Same-model 9B judge, saturating near 1.0; deterministic literal checks all pass with thinking on and no longer discriminate.",
          "Thinking budget: a fixed +3,500 tokens per call; one selector still exhausted it (empty output, regenerated); reasoning tokens are counted in output_tokens (498k for 380 calls).",
          "Serving: Ollama with flash attention and q8_0 KV cache; model weights and digest unchanged from runs 1–2 apart from the num_ctx 12288 derived tag.",
          "Result-file metadata: the run script's hardcoded 'thinking disabled' label was corrected before sealing, with the correction recorded inside the file (local_run.post_run_corrections); no data or output field was touched."
        ],
        "eml_controls": [
          "identical candidate ledger for A/B/C",
          "equal accounted calls",
          "condition-blind judge with deduplicated judge calls",
          "deterministic literal checks",
          "pool-relative breadth ratio"
        ],
        "eml_random_seeds": [
          "model nondeterminism, single run; frozen response cache and embedding cache in the bundle"
        ],
        "eml_run_count": 1,
        "eml_result_type": "MIXED",
        "eml_procedure": "scripts/run_real_local_ollama.py --model qwythos-9b-v2-q4km-ctx12k --reasoning-effort medium --budget-add 3500 --candidate-count 4 --repetitions 2; scripts/summarize_real_local.py; scripts/breadth_metrics.py.",
        "eml_software_environment": "Python 3.14, openai SDK 3.0.0 against Ollama 0.33.3 /v1/responses (OLLAMA_FLASH_ATTENTION=1, OLLAMA_KV_CACHE_TYPE=q8_0); nomic-embed-text for breadth; harness package unchanged.",
        "eml_reproduction_instructions": "Extract the bundle; rerun scripts/run_real_local_ollama.py with the identical arguments — every request is served from the frozen cache .pacc_real_cache_local_thinking (cache keys include the enlarged budgets), so it completes without a model; python scripts/breadth_metrics.py results/pacc_hybrid_v0.2_real_local_thinking.json --cache-dir .pacc_real_cache_local_thinking.",
        "eml_completed_at": "2026-09-11",
        "eml_model_ids": [
          "MOD-2026-0006"
        ],
        "eml_benchmark_ids": [
          "BEN-2026-0003"
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Sol (GPT-5.6, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0025/",
      "json": "/ai/experiments/EXP-2026-0025/index.json"
    },
    {
      "id": "EXP-2026-0009",
      "kind": "experiment",
      "label": "PACC-Lab v0.2 — third family (N3) and adversarial evidence geometry",
      "created_at": "2026-09-09",
      "updated_at": "2026-09-09",
      "values": {
        "eml_status": "STABLE",
        "eml_evidence_level": "E3",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0009/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Adaptive Epistemic Systems (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each lab's own result reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_aes/extract.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "Adds N3 constraint competition, a nonlinear constraint-field family, plus an adversarial geometry chosen to break additive/log-odds mappings, keeping every v0.1 threshold. N3 reaches Level 3 in all three geometries; N2 keeps agreement 0.8546 and small update/intervention error but its held-out representation JS 0.053327 crosses the 0.05 gate under adversarial geometry, so it is classified behavioural convergence only. Independent families under the linear diagnostic: {N0, N1}, {N2}, {N3}; robust across all three geometries: two.",
        "eml_summary_zh": "加入非線性約束場家族 N3 約束競爭，以及專門用來破壞加性／log-odds 映射的對抗性幾何，所有 v0.1 門檻不變。N3 在三種幾何下都達 Level 3；N2 在對抗幾何下一致度 0.8546、更新／干預誤差仍小，但 held-out 表徵 JS 0.053327 越過 0.05 門檻，歸為僅行為收斂。線性診斷下的獨立家族：{N0, N1}、{N2}、{N3}；三種幾何下都穩健的：兩個。",
        "eml_label_zh": "PACC-Lab v0.2——第三個家族（N3）與對抗性證據幾何",
        "eml_primary_domain": "Model Representation",
        "eml_domains": [
          "Evaluation",
          "Formal AI"
        ],
        "eml_program_id": "PRG-2026-0001",
        "eml_hypothesis": "A structurally distinct third family also converges, and convergence survives an evidence geometry designed against additive mappings.",
        "eml_metrics": {
          "verdict": "THIRD INDEPENDENT FAMILY CONVERGES; ROBUST PACC-A GATE REMAINS OPEN",
          "adversarial": {
            "N0": {
              "agreement": 0.90818,
              "heldout_js": 0.023921,
              "level": 3
            },
            "N2": {
              "agreement": 0.85455,
              "heldout_js": 0.053327,
              "level": 1
            },
            "N3": {
              "agreement": 0.90162,
              "heldout_js": 0.014454,
              "level": 3
            }
          },
          "five_seed_adversarial_level3_rate": {
            "N0": 1.0,
            "N1": 1.0,
            "N2": 0.6,
            "N3": 1.0
          },
          "independent_convergent_families": 2,
          "pacc_a": false
        },
        "eml_interpretation": "Probability-like convergence has a nontrivial basin, not a demonstrated universal attractor: the N3 result weakens the 'disguised additive accumulator' explanation, and N2 shows convergence is not guaranteed for every architecture under every geometry.",
        "eml_limitations": [
          "Independence is supported under the preregistered linear diagnostic only, not proof of deep nonlinear nonequivalence.",
          "N2's boundary is adversarial sensitivity around the threshold, not a universal phase boundary."
        ],
        "eml_random_seeds": [
          "20260908",
          "8 secondary seeds (smaller scale)",
          "5 adversarial seeds near primary scale"
        ],
        "eml_run_count": 3,
        "eml_result_type": "MIXED",
        "eml_controls": [
          "shuffled-target mapping",
          "constant prediction",
          "broken composition (v0.6+)",
          "target-local refit (v0.7+)"
        ],
        "eml_software_environment": "Python; deterministic seeded generators; no network, no LLM.",
        "eml_reproduction_instructions": "Extract the version's FINAL bundle; python -m pytest -q; run the version's primary script with the recorded seed; docs/PACC_LAB_v0.N_RESULTS.md and docs/EXPERIMENT_PROTOCOL_v0.N.md are inside the bundle.",
        "eml_completed_at": "2026-09-09",
        "eml_data_basis": "SYNTHETIC",
        "eml_model_ids": [
          "MOD-2026-0001",
          "MOD-2026-0002",
          "MOD-2026-0003",
          "MOD-2026-0004",
          "MOD-2026-0005"
        ],
        "eml_dataset_ids": [
          "DAT-2026-0001"
        ],
        "eml_benchmark_ids": [
          "BEN-2026-0001"
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Sol (GPT-5.6, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0009/",
      "json": "/ai/experiments/EXP-2026-0009/index.json"
    },
    {
      "id": "EXP-2026-0010",
      "kind": "experiment",
      "label": "PACC-Lab v0.3 — learned source reliability without an oracle",
      "created_at": "2026-09-09",
      "updated_at": "2026-09-09",
      "values": {
        "eml_status": "STABLE",
        "eml_evidence_level": "E3",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0010/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Adaptive Epistemic Systems (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each lab's own result reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_aes/extract.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "Both sides lose the source-quality oracle: the Bayesian reference learns Beta-Bernoulli source quality; the non-probabilistic systems learn a qualitative reputation (trust, friction, streak, familiarity). Fitted on training worlds and evaluated on unseen source-quality permutations, the reputation state maps to the Beta-Bernoulli state with JS ≈ 0.00714 versus 0.017–0.018 for shuffled/constant controls, and feedback-update commutation ≈ 0.00130 — stable 6/6 across seeds. Task-state convergence is architecture- and seed-sensitive: at exact primary scale N0/N1 4/4, N2 3/4, N3 2/4.",
        "eml_summary_zh": "雙方都失去來源品質 oracle：Bayesian 參考學 Beta-Bernoulli 來源品質；非概率系統學定性聲譽（信任、摩擦、連勝、熟悉度）。在訓練世界上擬合、在未見過的來源品質排列上評估，聲譽狀態映射到 Beta-Bernoulli 狀態的 JS ≈ 0.00714，shuffled／constant 控制組為 0.017–0.018，回饋更新交換 ≈ 0.00130——跨 seed 穩定 6/6。任務狀態收斂則依架構與 seed 而異：在主要規模下 N0/N1 4/4、N2 3/4、N3 2/4。",
        "eml_label_zh": "PACC-Lab v0.3——無 oracle 的來源可靠度學習",
        "eml_primary_domain": "Model Representation",
        "eml_domains": [
          "Evaluation",
          "Formal AI"
        ],
        "eml_program_id": "PRG-2026-0001",
        "eml_hypothesis": "Convergence survives when source reliability must be learned from delayed feedback rather than given.",
        "eml_metrics": {
          "verdict": "RELIABILITY-STATE CONVERGENCE ROBUST; TASK CONVERGENCE PARTIAL / BASIN-SENSITIVE",
          "primary_scale": "20 worlds / 280 episodes / 28 observations",
          "reliability_mapping_js": 0.007136,
          "control_js": "0.017–0.018",
          "feedback_commutation_js": 0.001298,
          "reliability_independent_families": 1,
          "task_pass_at_primary_scale": {
            "N0": "4/4",
            "N1": "4/4",
            "N2": "3/4",
            "N3": "2/4"
          }
        },
        "eml_interpretation": "Calibration-state convergence can be robust while task-state convergence has architecture-dependent basins. A first fixed-quality diagnostic was rejected because Beta means became near-constant and shuffled targets fit almost as well — that redesign is part of the evidence.",
        "eml_limitations": [
          "All four wrappers share one NonProbReputationLedger, so reliability convergence counts as one family, not four.",
          "Does not prove Beta-Bernoulli learning and qualitative reputation universally equivalent."
        ],
        "eml_random_seeds": [
          "20260909",
          "6 secondary seeds (smaller)",
          "4 seeds at exact primary scale"
        ],
        "eml_run_count": 3,
        "eml_result_type": "MIXED",
        "eml_controls": [
          "shuffled-target mapping",
          "constant prediction",
          "broken composition (v0.6+)",
          "target-local refit (v0.7+)"
        ],
        "eml_software_environment": "Python; deterministic seeded generators; no network, no LLM.",
        "eml_reproduction_instructions": "Extract the version's FINAL bundle; python -m pytest -q; run the version's primary script with the recorded seed; docs/PACC_LAB_v0.N_RESULTS.md and docs/EXPERIMENT_PROTOCOL_v0.N.md are inside the bundle.",
        "eml_completed_at": "2026-09-09",
        "eml_data_basis": "SYNTHETIC",
        "eml_model_ids": [
          "MOD-2026-0001",
          "MOD-2026-0002",
          "MOD-2026-0003",
          "MOD-2026-0004",
          "MOD-2026-0005"
        ],
        "eml_dataset_ids": [
          "DAT-2026-0001"
        ],
        "eml_benchmark_ids": [
          "BEN-2026-0001"
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Sol (GPT-5.6, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0010/",
      "json": "/ai/experiments/EXP-2026-0010/index.json"
    },
    {
      "id": "EXP-2026-0011",
      "kind": "experiment",
      "label": "PACC-Lab v0.4 — correlated sources and dependence geometry",
      "created_at": "2026-09-09",
      "updated_at": "2026-09-09",
      "values": {
        "eml_status": "STABLE",
        "eml_evidence_level": "E3",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0011/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Adaptive Epistemic Systems (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each lab's own result reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_aes/extract.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "Sources are grouped into clusters with a latent shared inversion; the correct reference uses the joint likelihood and a naive independent-product Bayes is the negative control (joint beats naive in moderate and high geometry; gap exactly 0 in the independent geometry). Non-probabilistic systems see only cluster membership. Task-state convergence survives: N0/N1/N3 pass in moderate and high correlation (N2 drops to agreement 0.8364 in high). The latent dependence coordinate — the posterior that the cluster is in its corrupted branch — is not recovered by a train-only affine-sigmoid map from any system: real JS ≈ shuffled ≈ constant.",
        "eml_summary_zh": "來源分成叢集並帶潛在共同反轉；正確的參考用聯合概似，天真的獨立乘積 Bayes 是負控制（在中、高相關幾何下聯合勝過天真；獨立幾何下差距恰為 0）。非概率系統只看得到叢集歸屬。任務狀態收斂存活：N0/N1/N3 在中、高相關下通過（N2 在高相關下一致度掉到 0.8364）。潛在依賴座標——叢集處於受污染分支的後驗——任何系統的 train-only affine-sigmoid 映射都無法重建：真實 JS ≈ shuffled ≈ constant。",
        "eml_label_zh": "PACC-Lab v0.4——相關來源與依賴幾何",
        "eml_primary_domain": "Model Representation",
        "eml_domains": [
          "Evaluation",
          "Formal AI"
        ],
        "eml_program_id": "PRG-2026-0001",
        "eml_hypothesis": "Task-state convergence survives dependent evidence, and the non-probabilistic relation state itself maps to the Bayesian common-cause posterior.",
        "eml_metrics": {
          "verdict": "TASK-LEVEL CONVERGENCE SURVIVES DEPENDENT EVIDENCE; LATENT DEPENDENCE-STATE CONVERGENCE NOT SUPPORTED",
          "environment": {
            "joint_vs_naive_logloss": {
              "independent": [
                1.245687,
                1.245687
              ],
              "moderate": [
                1.632599,
                1.834445
              ],
              "high": [
                1.758082,
                2.018617
              ]
            }
          },
          "task": {
            "moderate": {
              "N0": {
                "agreement": 0.955247,
                "D_R": 0.008095
              },
              "N3": {
                "agreement": 0.942901,
                "D_R": 0.008234
              }
            },
            "high": {
              "N0": {
                "agreement": 0.861111,
                "D_R": 0.009827
              },
              "N2": {
                "agreement": 0.83642,
                "D_R": 0.017874
              }
            }
          },
          "dependence_coordinate": {
            "moderate_real_js": 0.0605,
            "moderate_shuffled": 0.0641,
            "moderate_constant": 0.0615,
            "high_real_js": 0.0398,
            "high_shuffled": 0.04,
            "high_constant": 0.0397,
            "pass": 0
          }
        },
        "eml_interpretation": "A layered picture: a decision-relevant quotient state can converge to a probabilistic coordinate while a deeper latent explanatory variable stays representation-dependent — evidence against the strongest 'everything becomes probability-like' reading.",
        "eml_limitations": [
          "A result about the current state representation and frozen readout, not a theorem that no richer relation graph could encode the latent variable.",
          "Cluster membership is given, not learned."
        ],
        "eml_random_seeds": [
          "20260909",
          "6 secondary seeds",
          "4 high-correlation seeds at primary scale"
        ],
        "eml_run_count": 3,
        "eml_result_type": "MIXED",
        "eml_controls": [
          "shuffled-target mapping",
          "constant prediction",
          "broken composition (v0.6+)",
          "target-local refit (v0.7+)"
        ],
        "eml_software_environment": "Python; deterministic seeded generators; no network, no LLM.",
        "eml_reproduction_instructions": "Extract the version's FINAL bundle; python -m pytest -q; run the version's primary script with the recorded seed; docs/PACC_LAB_v0.N_RESULTS.md and docs/EXPERIMENT_PROTOCOL_v0.N.md are inside the bundle.",
        "eml_completed_at": "2026-09-09",
        "eml_data_basis": "SYNTHETIC",
        "eml_model_ids": [
          "MOD-2026-0001",
          "MOD-2026-0002",
          "MOD-2026-0003",
          "MOD-2026-0004",
          "MOD-2026-0005"
        ],
        "eml_dataset_ids": [
          "DAT-2026-0001"
        ],
        "eml_benchmark_ids": [
          "BEN-2026-0001"
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Sol (GPT-5.6, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0011/",
      "json": "/ai/experiments/EXP-2026-0011/index.json"
    },
    {
      "id": "EXP-2026-0012",
      "kind": "experiment",
      "label": "PACC-Lab v0.5 — does a richer relation state rescue the latent coordinate?",
      "created_at": "2026-09-09",
      "updated_at": "2026-09-09",
      "values": {
        "eml_status": "STABLE",
        "eml_evidence_level": "E3",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0012/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Adaptive Epistemic Systems (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each lab's own result reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_aes/extract.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "A rich relation graph keeps far more source/pair/pattern information than the compact v0.4 bundle while leaving task actions identical. It does not rescue the frozen direct affine-sigmoid map to the common-cause coordinate: compact 0 and rich 0 robust passes, real/control ratios ≈ 0.96–1.02, and no system rescued in four secondary seeds. Counterfactual source-flip probes often pass alone, showing partial local directional alignment without a globally discriminative coordinate.",
        "eml_summary_zh": "豐富的關係圖比 v0.4 的精簡 bundle 保留多得多的來源／成對／模式資訊，且任務行動完全相同。它救不回凍結的 direct affine-sigmoid 到共同因座標的映射：精簡 0、豐富 0 次穩健通過，真實／控制比 ≈ 0.96–1.02，四個次要 seed 也沒有任何系統被救回。反事實來源翻轉探針常單獨通過，顯示局部方向對齊部分存在，但沒有全域可判別的座標。",
        "eml_label_zh": "PACC-Lab v0.5——更豐富的關係狀態能救回潛在座標嗎？",
        "eml_primary_domain": "Model Representation",
        "eml_domains": [
          "Evaluation",
          "Formal AI"
        ],
        "eml_program_id": "PRG-2026-0001",
        "eml_hypothesis": "The v0.4 failure was information loss in the compact state; a richer non-probabilistic relation state admits the direct low-complexity latent coordinate.",
        "eml_metrics": {
          "verdict": "RICH_STATE_DOES_NOT_RESCUE_LATENT_DEPENDENCE_COORDINATE",
          "compact_robust_pass": 0,
          "rich_robust_pass": 0,
          "task_action_invariance": true,
          "moderate_N0": {
            "compact_dependence_js": 0.0605,
            "rich_dependence_js": 0.06146,
            "real_over_control": 0.999
          }
        },
        "eml_interpretation": "Richer raw relation state is insufficient for the tested direct low-complexity latent coordinate — not that the coordinate is information-theoretically unrecoverable. Points to a coordinate-composition problem (v0.6).",
        "eml_limitations": [
          "A direct affine-sigmoid observer may be too restrictive even when the raw state is informative."
        ],
        "eml_random_seeds": [
          "20260909",
          "4 secondary seeds"
        ],
        "eml_run_count": 2,
        "eml_result_type": "NEGATIVE",
        "eml_controls": [
          "shuffled-target mapping",
          "constant prediction",
          "broken composition (v0.6+)",
          "target-local refit (v0.7+)"
        ],
        "eml_software_environment": "Python; deterministic seeded generators; no network, no LLM.",
        "eml_reproduction_instructions": "Extract the version's FINAL bundle; python -m pytest -q; run the version's primary script with the recorded seed; docs/PACC_LAB_v0.N_RESULTS.md and docs/EXPERIMENT_PROTOCOL_v0.N.md are inside the bundle.",
        "eml_completed_at": "2026-09-09",
        "eml_data_basis": "SYNTHETIC",
        "eml_model_ids": [
          "MOD-2026-0001",
          "MOD-2026-0002",
          "MOD-2026-0003",
          "MOD-2026-0004",
          "MOD-2026-0005"
        ],
        "eml_dataset_ids": [
          "DAT-2026-0001"
        ],
        "eml_benchmark_ids": [
          "BEN-2026-0001"
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Sol (GPT-5.6, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0012/",
      "json": "/ai/experiments/EXP-2026-0012/index.json"
    },
    {
      "id": "EXP-2026-0013",
      "kind": "experiment",
      "label": "PACC-Lab v0.6 — hierarchical composed coordinate",
      "created_at": "2026-09-09",
      "updated_at": "2026-09-09",
      "values": {
        "eml_status": "STABLE",
        "eml_evidence_level": "E3",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0013/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Adaptive Epistemic Systems (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each lab's own result reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_aes/extract.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "A preregistered 95-dimensional composed coordinate — the validated task-state map's held-out decision quotient combined with the current relation pattern and a bounded set of interaction terms — predicts the Bayesian common-cause coordinate for all four families in both geometries: composed latent JS 0.0034–0.0073 versus 0.040–0.062 for shuffled/constant and 0.040–0.060 for a deliberately broken composition; counterfactual JS 0.0018–0.0047; four secondary seeds give direct 0.0 and composed 1.0 pass rates.",
        "eml_summary_zh": "一個預登記的 95 維組合座標——已驗證任務狀態映射的 held-out 決策商，結合當前關係模式與有界的交互項——在兩種幾何下對四個家族都預測出 Bayesian 共同因座標：組合潛在 JS 0.0034–0.0073，shuffled／constant 為 0.040–0.062，刻意弄壞的組合為 0.040–0.060；反事實 JS 0.0018–0.0047；四個次要 seed 下 direct 通過率 0.0、composed 1.0。",
        "eml_label_zh": "PACC-Lab v0.6——階層式組合座標",
        "eml_primary_domain": "Model Representation",
        "eml_domains": [
          "Evaluation",
          "Formal AI"
        ],
        "eml_program_id": "PRG-2026-0001",
        "eml_hypothesis": "The latent probability coordinate is compositional relative to the non-probabilistic state: decision quotient + relation pattern, not the raw state, maps to the common-cause posterior.",
        "eml_metrics": {
          "verdict": "HIERARCHICAL_COMPOSED_COORDINATE_RESCUES_LATENT_DEPENDENCE",
          "direct_robust": "0/4",
          "composed_robust": "4/4",
          "moderate_N0": {
            "composed_latent_js": 0.003603,
            "best_control": 0.061508,
            "broken_composition": 0.058196,
            "counterfactual": 0.00278
          },
          "high_N0": {
            "composed_latent_js": 0.003335,
            "best_control": 0.03973,
            "broken_composition": 0.041238
          }
        },
        "eml_interpretation": "Destroying the train alignment between task quotient and relation pattern destroys most of the signal, so the rescue is not a snapshot fit. Reinterprets v0.5: the information was present; adding raw coordinates did not make the latent state a direct coordinate.",
        "eml_limitations": [
          "Does not show all non-probabilistic states admit such a composition, that the coordinate transfers between geometries without refitting, or that clusters can be discovered unsupervised."
        ],
        "eml_random_seeds": [
          "20260909",
          "4 secondary seeds"
        ],
        "eml_run_count": 2,
        "eml_result_type": "POSITIVE",
        "eml_controls": [
          "shuffled-target mapping",
          "constant prediction",
          "broken composition (v0.6+)",
          "target-local refit (v0.7+)"
        ],
        "eml_software_environment": "Python; deterministic seeded generators; no network, no LLM.",
        "eml_reproduction_instructions": "Extract the version's FINAL bundle; python -m pytest -q; run the version's primary script with the recorded seed; docs/PACC_LAB_v0.N_RESULTS.md and docs/EXPERIMENT_PROTOCOL_v0.N.md are inside the bundle.",
        "eml_completed_at": "2026-09-09",
        "eml_data_basis": "SYNTHETIC",
        "eml_model_ids": [
          "MOD-2026-0001",
          "MOD-2026-0002",
          "MOD-2026-0003",
          "MOD-2026-0004",
          "MOD-2026-0005"
        ],
        "eml_dataset_ids": [
          "DAT-2026-0001"
        ],
        "eml_benchmark_ids": [
          "BEN-2026-0001"
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Sol (GPT-5.6, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0013/",
      "json": "/ai/experiments/EXP-2026-0013/index.json"
    },
    {
      "id": "EXP-2026-0014",
      "kind": "experiment",
      "label": "PACC-Lab v0.7 — cross-geometry coordinate transfer without refitting",
      "created_at": "2026-09-09",
      "updated_at": "2026-09-09",
      "values": {
        "eml_status": "STABLE",
        "eml_evidence_level": "E3",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0014/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Adaptive Epistemic Systems (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each lab's own result reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_aes/extract.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "The v0.6 composed coordinate is frozen on one dependence geometry and applied to the other. N0/N1/N3 transfer bidirectionally on the primary protocol and 4/4 secondary seeds; N2 passes high → moderate but misses the moderate → high task-quotient gate by ≈ 0.0003 on the primary seed (2/4 secondary), while a targeted larger stress passes 3/3. Every target-local refit passes, so N2's failure is not a target-chart failure.",
        "eml_summary_zh": "把 v0.6 的組合座標凍結在一種依賴幾何上、套用到另一種。N0/N1/N3 在主要協定與 4/4 次要 seed 下雙向轉移；N2 高 → 中通過，但中 → 高在主要 seed 下以 ≈ 0.0003 之差未達任務商門檻（次要 2/4），而針對性的較大規模壓力測試 3/3 通過。所有目標本地重擬合都通過，因此 N2 的失敗不是目標 chart 的失敗。",
        "eml_label_zh": "PACC-Lab v0.7——不重新擬合的跨幾何座標轉移",
        "eml_primary_domain": "Model Representation",
        "eml_domains": [
          "Evaluation",
          "Formal AI"
        ],
        "eml_program_id": "PRG-2026-0001",
        "eml_hypothesis": "The composed coordinate is one geometry-invariant coordinate rather than a family of per-geometry local charts.",
        "eml_metrics": {
          "verdict": "PARTIAL_CROSS_GEOMETRY_TRANSFER",
          "moderate_to_high": {
            "N0": {
              "task_js": 0.03204,
              "latent_js": 0.01767,
              "pass": true
            },
            "N2": {
              "task_js": 0.0503,
              "latent_js": 0.02716,
              "pass": false
            },
            "N3": {
              "task_js": 0.03699,
              "pass": true
            }
          },
          "high_to_moderate": {
            "N0": {
              "task_js": 0.02993,
              "pass": true
            },
            "N2": {
              "task_js": 0.03237,
              "pass": true
            }
          },
          "secondary": {
            "N0_N1_N3": "4/4 both directions",
            "N2": "4/4 high→moderate, 2/4 moderate→high"
          }
        },
        "eml_interpretation": "A shared cross-geometry coordinate with family-specific basin boundaries — neither universal invariance nor separate local charts.",
        "eml_limitations": [
          "Not established: a universal geometry-invariant coordinate, transfer across a continuous dependence range, transfer to unseen cluster topology."
        ],
        "eml_random_seeds": [
          "20260909",
          "4 secondary seeds",
          "3 targeted N2 seeds at 120 episodes × 20 steps"
        ],
        "eml_run_count": 3,
        "eml_result_type": "MIXED",
        "eml_controls": [
          "shuffled-target mapping",
          "constant prediction",
          "broken composition (v0.6+)",
          "target-local refit (v0.7+)"
        ],
        "eml_software_environment": "Python; deterministic seeded generators; no network, no LLM.",
        "eml_reproduction_instructions": "Extract the version's FINAL bundle; python -m pytest -q; run the version's primary script with the recorded seed; docs/PACC_LAB_v0.N_RESULTS.md and docs/EXPERIMENT_PROTOCOL_v0.N.md are inside the bundle.",
        "eml_completed_at": "2026-09-09",
        "eml_data_basis": "SYNTHETIC",
        "eml_model_ids": [
          "MOD-2026-0001",
          "MOD-2026-0002",
          "MOD-2026-0003",
          "MOD-2026-0004",
          "MOD-2026-0005"
        ],
        "eml_dataset_ids": [
          "DAT-2026-0001"
        ],
        "eml_benchmark_ids": [
          "BEN-2026-0001"
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Sol (GPT-5.6, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0014/",
      "json": "/ai/experiments/EXP-2026-0014/index.json"
    },
    {
      "id": "EXP-2026-0015",
      "kind": "experiment",
      "label": "PACC-Lab v0.8 — frozen-coordinate transfer basins over a correlation sweep",
      "created_at": "2026-09-09",
      "updated_at": "2026-09-09",
      "values": {
        "eml_status": "STABLE",
        "eml_evidence_level": "E3",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0015/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Adaptive Epistemic Systems (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each lab's own result reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_aes/extract.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "One composed coordinate fitted at q* = 0.20 is applied unchanged across a nine-point sweep q ∈ {0.06 … 0.42} (within-cluster correctness correlation rising from ≈ 0.203 to ≈ 0.502). N0/N1/N3 pass through q = 0.33 and fail from 0.36; N2 passes through 0.30 and fails from 0.33. At the edge the negative-control discrimination margin is lost first while absolute latent error stays small; target-local refit passes 9/9 everywhere. Four secondary seeds reproduce the basin exactly for N0/N1/N3 and a seed-sensitive N2 edge at 0.33.",
        "eml_summary_zh": "在 q* = 0.20 擬合的一個組合座標，原封不動地套用到九點掃描 q ∈ {0.06 … 0.42}（叢集內正確性相關從 ≈ 0.203 升到 ≈ 0.502）。N0/N1/N3 通過到 q = 0.33、從 0.36 起失敗；N2 通過到 0.30、從 0.33 起失敗。在邊緣先失去的是負控制判別餘裕，而絕對潛在誤差仍小；目標本地重擬合處處 9/9 通過。四個次要 seed 對 N0/N1/N3 精確重現吸引域，N2 在 0.33 的邊緣對 seed 敏感。",
        "eml_label_zh": "PACC-Lab v0.8——相關度掃描下的凍結座標轉移吸引域",
        "eml_primary_domain": "Model Representation",
        "eml_domains": [
          "Evaluation",
          "Formal AI"
        ],
        "eml_program_id": "PRG-2026-0001",
        "eml_hypothesis": "A frozen probability-like coordinate has a finite transfer basin whose width depends on the family.",
        "eml_metrics": {
          "verdict": "FAMILY_SPECIFIC_TRANSFER_BASINS",
          "basins_grid": {
            "N0_N1_N3": "[0.06, 0.33]",
            "N2": "[0.06, 0.30]"
          },
          "pass_fraction": {
            "N0": "7/9",
            "N2": "6/9",
            "N3": "7/9"
          },
          "local_refit": "9/9",
          "edge_N0_q0.36": {
            "task_js": 0.0488,
            "latent_js": 0.0274,
            "real_over_control": 0.858,
            "fails": "negative-control discrimination"
          }
        },
        "eml_interpretation": "An atlas/basin picture: large regions share one low-complexity coordinate; different internal dynamics meet transfer boundaries at different places. High-q failure is not evidence that probability-like coordinates cease to exist there.",
        "eml_limitations": [
          "Grid points only, not continuous coverage; no single global coordinate over all dependence strengths."
        ],
        "eml_random_seeds": [
          "20260909",
          "4 secondary seeds"
        ],
        "eml_run_count": 2,
        "eml_result_type": "MIXED",
        "eml_controls": [
          "shuffled-target mapping",
          "constant prediction",
          "broken composition (v0.6+)",
          "target-local refit (v0.7+)"
        ],
        "eml_software_environment": "Python; deterministic seeded generators; no network, no LLM.",
        "eml_reproduction_instructions": "Extract the version's FINAL bundle; python -m pytest -q; run the version's primary script with the recorded seed; docs/PACC_LAB_v0.N_RESULTS.md and docs/EXPERIMENT_PROTOCOL_v0.N.md are inside the bundle.",
        "eml_completed_at": "2026-09-09",
        "eml_data_basis": "SYNTHETIC",
        "eml_model_ids": [
          "MOD-2026-0001",
          "MOD-2026-0002",
          "MOD-2026-0003",
          "MOD-2026-0004",
          "MOD-2026-0005"
        ],
        "eml_dataset_ids": [
          "DAT-2026-0001"
        ],
        "eml_benchmark_ids": [
          "BEN-2026-0001"
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Sol (GPT-5.6, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0015/",
      "json": "/ai/experiments/EXP-2026-0015/index.json"
    },
    {
      "id": "EXP-2026-0016",
      "kind": "experiment",
      "label": "PACC-Lab v0.9 — a three-anchor probability-coordinate atlas",
      "created_at": "2026-09-09",
      "updated_at": "2026-09-09",
      "values": {
        "eml_status": "STABLE",
        "eml_evidence_level": "E3",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0016/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Adaptive Epistemic Systems (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each lab's own result reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_aes/extract.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "Three preregistered charts at q = 0.12, 0.24, 0.36 give every family the same union {0.06 … 0.36}: 8/9 sweep coverage, stable across three secondary seeds, never covering q = 0.42. Nearest charts (0.12/0.24) agree directly; distant charts (0.12/0.36) do not (task JS ≈ 0.071–0.082), yet low-complexity transition maps calibrated on an independent seed pass 5/6 directions per family on the primary protocol — the weak direction is 0.36 → 0.12 (latent JS ≈ 0.017–0.021 > 0.01). Transition coherence is not robust across secondary seeds (mean pass ≈ 0.53–0.67).",
        "eml_summary_zh": "在 q = 0.12、0.24、0.36 的三個預登記 chart 給每個家族相同的聯集 {0.06 … 0.36}：掃描覆蓋 8/9，三個次要 seed 下穩定，永遠蓋不到 q = 0.42。最近的 chart（0.12/0.24）直接一致；相距遠的（0.12/0.36）不一致（任務 JS ≈ 0.071–0.082），但在獨立 seed 上校準的低複雜度轉換映射在主要協定下每個家族 6 個方向通過 5 個——弱方向是 0.36 → 0.12（潛在 JS ≈ 0.017–0.021 > 0.01）。轉換一致性在次要 seed 下不穩健（平均通過 ≈ 0.53–0.67）。",
        "eml_label_zh": "PACC-Lab v0.9——三錨點概率座標圖冊",
        "eml_primary_domain": "Model Representation",
        "eml_domains": [
          "Evaluation",
          "Formal AI"
        ],
        "eml_program_id": "PRG-2026-0001",
        "eml_hypothesis": "A small multi-anchor atlas covers the sweep and its charts are mutually translatable by low-complexity transitions.",
        "eml_metrics": {
          "verdict": "PARTIAL_MULTI_ANCHOR_ATLAS",
          "coverage": "8/9",
          "uncovered_q": 0.42,
          "transition_pass_primary": "5/6 per family",
          "weak_direction": "0.36→0.12",
          "secondary_transition_pass_rate": {
            "N0_N1": 0.667,
            "N2": 0.528,
            "N3": 0.583
          }
        },
        "eml_interpretation": "Rejects both extremes — one global chart (finite coverage) and unrelated local charts (overlaps translate cheaply) — leaving partially overlapping probability-like charts with finite coverage and nonuniform transition coherence.",
        "eml_limitations": [
          "Full atlas coverage, robust transition coherence, cocycle consistency and manifold structure not established."
        ],
        "eml_random_seeds": [
          "20260909",
          "independent calibration seed",
          "3 secondary seeds"
        ],
        "eml_run_count": 3,
        "eml_result_type": "MIXED",
        "eml_controls": [
          "shuffled-target mapping",
          "constant prediction",
          "broken composition (v0.6+)",
          "target-local refit (v0.7+)"
        ],
        "eml_software_environment": "Python; deterministic seeded generators; no network, no LLM.",
        "eml_reproduction_instructions": "Extract the version's FINAL bundle; python -m pytest -q; run the version's primary script with the recorded seed; docs/PACC_LAB_v0.N_RESULTS.md and docs/EXPERIMENT_PROTOCOL_v0.N.md are inside the bundle.",
        "eml_completed_at": "2026-09-09",
        "eml_data_basis": "SYNTHETIC",
        "eml_model_ids": [
          "MOD-2026-0001",
          "MOD-2026-0002",
          "MOD-2026-0003",
          "MOD-2026-0004",
          "MOD-2026-0005"
        ],
        "eml_dataset_ids": [
          "DAT-2026-0001"
        ],
        "eml_benchmark_ids": [
          "BEN-2026-0001"
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Sol (GPT-5.6, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0016/",
      "json": "/ai/experiments/EXP-2026-0016/index.json"
    },
    {
      "id": "EXP-2026-0017",
      "kind": "experiment",
      "label": "PACC-Lab v0.10 — oriented cocycle coherence on frozen triple overlap",
      "created_at": "2026-09-09",
      "updated_at": "2026-09-09",
      "values": {
        "eml_status": "STABLE",
        "eml_evidence_level": "E3",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0017/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Adaptive Epistemic Systems (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each lab's own result reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_aes/extract.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "On the frozen region where charts A, B and C were all valid in the v0.9 atlas, the direct transition A → C and the composed A → B → C agree to ≈ 10⁻⁵ in latent JS (shuffled-pair control ≈ 10⁻¹), while both paths separately stay accurate to chart C (latent JS ≈ 0.0045–0.0054). All four families pass on the primary seed and on four secondary seeds. N2 has only one triple-overlap geometry (q = 0.24), so its witness is narrower.",
        "eml_summary_zh": "在 v0.9 圖冊中 A、B、C 三個 chart 都有效的凍結區域上，直接轉換 A → C 與組合 A → B → C 在潛在 JS 上一致到 ≈ 10⁻⁵（shuffled-pair 控制 ≈ 10⁻¹），且兩條路徑各自對 chart C 都準確（潛在 JS ≈ 0.0045–0.0054）。四個家族在主要 seed 與四個次要 seed 下都通過。N2 只有一個三重重疊幾何（q = 0.24），見證較窄。",
        "eml_label_zh": "PACC-Lab v0.10——凍結三重重疊上的定向 cocycle 一致性",
        "eml_primary_domain": "Model Representation",
        "eml_domains": [
          "Evaluation",
          "Formal AI"
        ],
        "eml_program_id": "PRG-2026-0001",
        "eml_hypothesis": "T_AC ≈ T_BC ∘ T_AB on held-out triple-overlap data.",
        "eml_metrics": {
          "verdict": "ROBUST_ORIENTED_COCYCLE_COHERENCE_ON_FROZEN_TRIPLE_OVERLAP",
          "N0": {
            "overlap_q": [
              0.24,
              0.3
            ],
            "direct_vs_composed_latent_js": 3.5e-05,
            "direct_to_C": 0.00446,
            "composed_to_C": 0.004627,
            "real_over_shuffled": 0.000346
          },
          "N3": {
            "overlap_q": [
              0.2,
              0.24,
              0.3
            ],
            "direct_vs_composed_latent_js": 2.65e-05
          },
          "secondary_pass_rate": 1.0
        },
        "eml_interpretation": "Pairwise transition quality may be uneven globally while cocycle composition is highly coherent locally on triple overlap — consistent with coordinate compatibility being a local overlap property. Too early for a coordinate groupoid or manifold.",
        "eml_limitations": [
          "Evidence volume differs by family; N2's triple overlap is a single geometry."
        ],
        "eml_random_seeds": [
          "20260909",
          "3",
          "5",
          "7",
          "11"
        ],
        "eml_run_count": 5,
        "eml_result_type": "POSITIVE",
        "eml_controls": [
          "shuffled-target mapping",
          "constant prediction",
          "broken composition (v0.6+)",
          "target-local refit (v0.7+)"
        ],
        "eml_software_environment": "Python; deterministic seeded generators; no network, no LLM.",
        "eml_reproduction_instructions": "Extract the version's FINAL bundle; python -m pytest -q; run the version's primary script with the recorded seed; docs/PACC_LAB_v0.N_RESULTS.md and docs/EXPERIMENT_PROTOCOL_v0.N.md are inside the bundle.",
        "eml_completed_at": "2026-09-09",
        "eml_data_basis": "SYNTHETIC",
        "eml_model_ids": [
          "MOD-2026-0001",
          "MOD-2026-0002",
          "MOD-2026-0003",
          "MOD-2026-0004",
          "MOD-2026-0005"
        ],
        "eml_dataset_ids": [
          "DAT-2026-0001"
        ],
        "eml_benchmark_ids": [
          "BEN-2026-0001"
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Sol (GPT-5.6, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0017/",
      "json": "/ai/experiments/EXP-2026-0017/index.json"
    },
    {
      "id": "EXP-2026-0018",
      "kind": "experiment",
      "label": "PACC-Lab v0.11 — bidirectional cocycle and inverse consistency",
      "created_at": "2026-09-09",
      "updated_at": "2026-09-09",
      "values": {
        "eml_status": "STABLE",
        "eml_evidence_level": "E3",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0018/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Adaptive Epistemic Systems (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each lab's own result reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_aes/extract.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "Separates three properties: forward cocycle composition (survives, all families), round-trip inverse consistency (all pairwise round trips pass on the primary run, recover 3/3 at larger scale, but are sample-sensitive at small scale, weakest on the widest pair AC), and reverse transport fidelity to the destination chart (fails for every family: direct C → A latent fidelity 0.0116–0.0148 > 0.01 while direct and composed reverse paths agree to ≈ 10⁻⁵).",
        "eml_summary_zh": "區分三個性質：正向 cocycle 組合（所有家族都存活）、往返反向一致性（主要 run 所有成對往返都通過，在較大規模 3/3 恢復，但小規模下對樣本敏感，最寬的 AC 對最弱），以及反向傳輸對目的 chart 的保真度（所有家族都失敗：直接 C → A 潛在保真度 0.0116–0.0148 > 0.01，而直接與組合反向路徑彼此一致到 ≈ 10⁻⁵）。",
        "eml_label_zh": "PACC-Lab v0.11——雙向 cocycle 與反向一致性",
        "eml_primary_domain": "Model Representation",
        "eml_domains": [
          "Evaluation",
          "Formal AI"
        ],
        "eml_program_id": "PRG-2026-0001",
        "eml_hypothesis": "The local transitions form a groupoid: forward cocycle, inverse consistency and reverse destination fidelity all hold.",
        "eml_metrics": {
          "verdict": "PARTIAL BIDIRECTIONAL COHERENCE WITH REVERSE DESTINATION BIAS",
          "forward_cocycle": "PASS all families",
          "inverse_pairs_primary": "PASS all",
          "inverse_pairs_large_scale": "3/3",
          "reverse_fidelity": {
            "N0": 0.01162,
            "N2": 0.01476,
            "N3": 0.01355,
            "gate": 0.01
          },
          "reverse_direct_vs_composed_latent_js": "7e-05 to 1e-04",
          "reverse_pass_rate": "0/4 and 0/3"
        },
        "eml_interpretation": "RoundTripInvertibility ⇏ DestinationChartFidelity and PathCoherence ⇏ GroupoidClosure: a directionally coherent calibration structure with a persistent reverse latent bias, not a closed local groupoid under the affine transition class.",
        "eml_limitations": [
          "Does not establish nonlinear-transition impossibility, manifold structure or full atlas coverage."
        ],
        "eml_random_seeds": [
          "20260909",
          "4 secondary seeds",
          "3 primary-scale-ish seeds"
        ],
        "eml_run_count": 3,
        "eml_result_type": "MIXED",
        "eml_controls": [
          "shuffled-target mapping",
          "constant prediction",
          "broken composition (v0.6+)",
          "target-local refit (v0.7+)"
        ],
        "eml_software_environment": "Python; deterministic seeded generators; no network, no LLM.",
        "eml_reproduction_instructions": "Extract the version's FINAL bundle; python -m pytest -q; run the version's primary script with the recorded seed; docs/PACC_LAB_v0.N_RESULTS.md and docs/EXPERIMENT_PROTOCOL_v0.N.md are inside the bundle.",
        "eml_completed_at": "2026-09-09",
        "eml_data_basis": "SYNTHETIC",
        "eml_model_ids": [
          "MOD-2026-0001",
          "MOD-2026-0002",
          "MOD-2026-0003",
          "MOD-2026-0004",
          "MOD-2026-0005"
        ],
        "eml_dataset_ids": [
          "DAT-2026-0001"
        ],
        "eml_benchmark_ids": [
          "BEN-2026-0001"
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Sol (GPT-5.6, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0018/",
      "json": "/ai/experiments/EXP-2026-0018/index.json"
    },
    {
      "id": "EXP-2026-0019",
      "kind": "experiment",
      "label": "PACC-Lab v0.12 — bounded quadratic transition",
      "created_at": "2026-09-09",
      "updated_at": "2026-09-09",
      "values": {
        "eml_status": "STABLE",
        "eml_evidence_level": "E3",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0019/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Adaptive Epistemic Systems (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each lab's own result reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_aes/extract.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "Transition capacity rises in one preregistered step from 20 affine to 60 fixed degree-2 coefficients with charts, support, split, controls and thresholds unchanged. No family is rescued: reverse fidelity 0.0111 (N0/N1), 0.0139 (N2), 0.0141 (N3) versus the 0.01 gate, 0/4 secondary and 0/3 at larger scale; forward cocycle and inverse consistency survive the quadratic class. The lab stops here rather than escalating to cubic, quartic or neural transitions after seeing the result.",
        "eml_summary_zh": "轉換容量以一個預登記的步驟從 20 個仿射係數升到 60 個固定二次係數，chart、支撐、切分、控制與門檻不變。沒有家族被救回：反向保真度 0.0111（N0/N1）、0.0139（N2）、0.0141（N3），門檻 0.01，次要 0/4、較大規模 0/3；正向 cocycle 與反向一致性在二次類下存活。實驗室在此停止，而不是看到結果後升級到三次、四次或神經轉換。",
        "eml_label_zh": "PACC-Lab v0.12——有界二次轉換",
        "eml_primary_domain": "Model Representation",
        "eml_domains": [
          "Evaluation",
          "Formal AI"
        ],
        "eml_program_id": "PRG-2026-0001",
        "eml_hypothesis": "The reverse destination bias is an artefact of the 4-D affine transition class.",
        "eml_metrics": {
          "verdict": "REVERSE BIAS PERSISTS UNDER BOUNDED QUADRATIC",
          "quadratic_reverse_fidelity": {
            "N0": 0.01108,
            "N2": 0.01392,
            "N3": 0.0141
          },
          "gate": 0.01,
          "quadratic_reverse_pass": "0/4 and 0/3",
          "forward_cocycle": "PASS",
          "inverse_pairs": "PASS (N2 2/3 at scale)"
        },
        "eml_interpretation": "Rejects the affine-limitation explanation and strengthens a persistent directional/base-point mismatch as the next hypothesis. A sufficiently flexible approximator could fit any finite sample, which is exactly why capacity was bounded in advance.",
        "eml_limitations": [
          "Does not prove that no nonlinear transition can remove the bias."
        ],
        "eml_random_seeds": [
          "20260909",
          "3",
          "5",
          "7",
          "11",
          "3 primary-scale-ish seeds"
        ],
        "eml_run_count": 6,
        "eml_result_type": "NEGATIVE",
        "eml_controls": [
          "shuffled-target mapping",
          "constant prediction",
          "broken composition (v0.6+)",
          "target-local refit (v0.7+)"
        ],
        "eml_software_environment": "Python; deterministic seeded generators; no network, no LLM.",
        "eml_reproduction_instructions": "Extract the version's FINAL bundle; python -m pytest -q; run the version's primary script with the recorded seed; docs/PACC_LAB_v0.N_RESULTS.md and docs/EXPERIMENT_PROTOCOL_v0.N.md are inside the bundle.",
        "eml_completed_at": "2026-09-09",
        "eml_data_basis": "SYNTHETIC",
        "eml_model_ids": [
          "MOD-2026-0001",
          "MOD-2026-0002",
          "MOD-2026-0003",
          "MOD-2026-0004",
          "MOD-2026-0005"
        ],
        "eml_dataset_ids": [
          "DAT-2026-0001"
        ],
        "eml_benchmark_ids": [
          "BEN-2026-0001"
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Sol (GPT-5.6, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0019/",
      "json": "/ai/experiments/EXP-2026-0019/index.json"
    },
    {
      "id": "EXP-2026-0020",
      "kind": "experiment",
      "label": "PACC-Lab v0.13 — reverse residual field / base-point dependence",
      "created_at": "2026-09-09",
      "updated_at": "2026-09-09",
      "values": {
        "eml_status": "STABLE",
        "eml_evidence_level": "E3",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0020/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Adaptive Epistemic Systems (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each lab's own result reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_aes/extract.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "The held-out reverse residual r_CA(S) = Φ_A(S) − T_CA(Φ_C(S)) is measured for mean, covariance, latent-logit energy concentration, between- versus within-stratum variance, direction stability and a train-only per-stratum constant correction against global and shuffled-stratum controls. N0/N1 concentrate 98 % of residual energy on the latent-logit axis, yet the mean direction flips sign across seeds (cross-seed cosine ≈ −1), between-q structure is ≈ 0.1–0.3 % against a 20 % gate, and stratum correction changes held-out fidelity by < 10⁻⁵ — no family passes.",
        "eml_summary_zh": "量測 held-out 反向殘差 r_CA(S) = Φ_A(S) − T_CA(Φ_C(S)) 的均值、共變異、潛在 logit 能量集中度、層間 vs 層內變異、方向穩定性，以及只用訓練集的逐層常數校正（對照全域與 shuffled 層控制）。N0/N1 把 98 % 的殘差能量集中在潛在 logit 軸上，但均值方向跨 seed 翻號（跨 seed 餘弦 ≈ −1），層間結構只有 ≈ 0.1–0.3 %（門檻 20 %），逐層校正對 held-out 保真度的改變 < 10⁻⁵——沒有家族通過。",
        "eml_label_zh": "PACC-Lab v0.13——反向殘差場／基點依賴",
        "eml_primary_domain": "Model Representation",
        "eml_domains": [
          "Evaluation",
          "Formal AI"
        ],
        "eml_program_id": "PRG-2026-0001",
        "eml_hypothesis": "The reverse bias is a base-point-conditioned residual field r(S) ≈ b_q + ε learnable per frozen overlap stratum.",
        "eml_metrics": {
          "verdict": "RESIDUAL MOSTLY UNSTRUCTURED",
          "latent_energy_fraction_affine": {
            "N0": 0.9845,
            "N2": 0.7723,
            "N3": 0.3182
          },
          "between_q_fraction": "0.0012–0.0033 vs gate 0.20",
          "stratum_correction": {
            "N0_uncorrected": 0.011616,
            "N0_corrected": 0.011624
          },
          "cross_family_direction_cosine_median": 0.7024,
          "cross_seed_direction_cosine_N0": -0.9977,
          "base_point_pass": "0 for every family"
        },
        "eml_interpretation": "A stable residual axis is not a stable residual orientation and not a base-point field; the reverse bias contains family-dependent dominant error modes, which weakens the gauge/connection-like reading. Next (v0.14, preregistered): sign-free residual subspace / principal-axis stability, with no q input, no higher degree, no neural mapper.",
        "eml_limitations": [
          "Residual unstructuredness is not established in every sign-free or subspace sense — that is the v0.14 question."
        ],
        "eml_random_seeds": [
          "20260909",
          "4 secondary seeds",
          "3 primary-scale-ish seeds"
        ],
        "eml_run_count": 3,
        "eml_result_type": "NEGATIVE",
        "eml_controls": [
          "shuffled-target mapping",
          "constant prediction",
          "broken composition (v0.6+)",
          "target-local refit (v0.7+)"
        ],
        "eml_software_environment": "Python; deterministic seeded generators; no network, no LLM.",
        "eml_reproduction_instructions": "Extract the version's FINAL bundle; python -m pytest -q; run the version's primary script with the recorded seed; docs/PACC_LAB_v0.N_RESULTS.md and docs/EXPERIMENT_PROTOCOL_v0.N.md are inside the bundle.",
        "eml_completed_at": "2026-09-09",
        "eml_data_basis": "SYNTHETIC",
        "eml_model_ids": [
          "MOD-2026-0001",
          "MOD-2026-0002",
          "MOD-2026-0003",
          "MOD-2026-0004",
          "MOD-2026-0005"
        ],
        "eml_dataset_ids": [
          "DAT-2026-0001"
        ],
        "eml_benchmark_ids": [
          "BEN-2026-0001"
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Sol (GPT-5.6, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0020/",
      "json": "/ai/experiments/EXP-2026-0020/index.json"
    },
    {
      "id": "EXP-2026-0021",
      "kind": "experiment",
      "label": "PACC-Hybrid v0.1 — synthetic A/B/C architecture witness",
      "created_at": "2026-09-09",
      "updated_at": "2026-09-09",
      "values": {
        "eml_status": "STABLE",
        "eml_evidence_level": "E3",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0021/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Adaptive Epistemic Systems (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each lab's own result reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_aes/extract.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "Over 1,568 identical candidate pools, generator-only (A), hard verifier (B) and PACC runtime (C) are scored on hard adherence, derived coherence, soft-intent satisfaction, long-horizon retention, novelty and pattern entropy. B already saturates literal hard adherence (1.0); C adds derived coherence +0.2085, long-horizon retention +0.0142 and valid novelty +0.1964 over B, loses 0.0076 soft-intent satisfaction, and in the pure-creative control raises pointwise novelty (0.9864 vs 0.8889) while collapsing pattern entropy (0.2284 vs 0.7199). A post-hoc 'elastic' exploration policy (D) restores entropy to 0.7638 with derived coherence still 1.0. Zero real LLM calls; eight seeds sign-stable.",
        "eml_summary_zh": "在 1,568 個完全相同的候選池上，純生成器（A）、硬驗證器（B）、PACC runtime（C）以硬約束遵守、衍生一致性、軟意圖滿足、長程保持、新穎度與 pattern entropy 計分。B 已經把字面硬約束遵守飽和到 1.0；C 相對 B 衍生一致性 +0.2085、長程保持 +0.0142、有效新穎度 +0.1964、軟意圖滿足 −0.0076，並在純創意控制組中提高逐點新穎度（0.9864 vs 0.8889）卻讓 pattern entropy 塌縮（0.2284 vs 0.7199）。事後的「elastic」探索策略（D）在衍生一致性仍為 1.0 下把 entropy 恢復到 0.7638。零次真實 LLM 呼叫；八個 seed 符號穩定。",
        "eml_label_zh": "PACC-Hybrid v0.1——合成 A/B/C 架構見證",
        "eml_primary_domain": "Reasoning",
        "eml_domains": [
          "Evaluation"
        ],
        "eml_program_id": "PRG-2026-0001",
        "eml_hypothesis": "A PACC-style commit space changes selection quality beyond what a hard verifier achieves, and any creative-breadth loss is separable from the commit constraints.",
        "eml_metrics": {
          "verdict": "SYNTHETIC_PARETO_WITNESS_REASONING_UP_BREADTH_COLLAPSE_MITIGATABLE",
          "scale": "7 categories × 28 tasks × 8 pools × 112 candidates; 1568 pool instances; 0 LLM calls",
          "overall": {
            "A": {
              "hard": 0.9936,
              "derived": 0.7902,
              "soft": 0.7878,
              "long_horizon": 0.9648,
              "valid_novelty": 0.5487
            },
            "B": {
              "hard": 1.0,
              "derived": 0.7915,
              "soft": 0.7859,
              "long_horizon": 0.9676,
              "valid_novelty": 0.551
            },
            "C": {
              "hard": 1.0,
              "derived": 1.0,
              "soft": 0.7783,
              "long_horizon": 0.9818,
              "valid_novelty": 0.7474
            }
          },
          "pure_creative": {
            "A": {
              "raw_novelty": 0.8889,
              "pattern_entropy": 0.7199
            },
            "C": {
              "raw_novelty": 0.9864,
              "pattern_entropy": 0.2284
            },
            "D_elastic_posthoc": {
              "raw_novelty": 0.9225,
              "pattern_entropy": 0.7638
            }
          },
          "eight_seed_deltas": {
            "C_minus_B_derived": {
              "mean": 0.1865,
              "sign": "8/8 positive"
            },
            "C_minus_A_pattern_entropy": {
              "mean": -0.3894,
              "sign": "8/8 negative"
            }
          },
          "B_literal_fallback_rate": 0.286
        },
        "eml_interpretation": "Not a simple reasoning-up / imagination-down trade-off: coherence and valid novelty rise, soft-preference fit dips slightly, and the breadth tax is a selection-policy effect that separating exploration from commitment recovers. The elastic diagnostic is post-hoc and not part of the primary result.",
        "eml_limitations": [
          "Controlled synthetic witness only; does not demonstrate that a frontier LLM shows the same effect."
        ],
        "eml_controls": [
          "identical candidate pools for A/B/C",
          "pure-creative negative control",
          "post-hoc elastic diagnostic labelled as such"
        ],
        "eml_random_seeds": [
          "20260909",
          "8 fixed secondary seeds"
        ],
        "eml_run_count": 9,
        "eml_result_type": "MIXED",
        "eml_software_environment": "Python; synthetic generators and scorers; no network, no LLM.",
        "eml_reproduction_instructions": "Extract PACC-Hybrid-Lab_v0.1_SYNTHETIC_FINAL.zip; python -m pytest -q; results in results/*.json and docs/PACC_HYBRID_v0.1_RESULTS.md.",
        "eml_completed_at": "2026-09-09",
        "eml_data_basis": "SYNTHETIC",
        "eml_dataset_ids": [
          "DAT-2026-0002"
        ],
        "eml_benchmark_ids": [
          "BEN-2026-0003"
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Sol (GPT-5.6, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0021/",
      "json": "/ai/experiments/EXP-2026-0021/index.json"
    },
    {
      "id": "EXP-2026-0022",
      "kind": "experiment",
      "label": "PACC-Hybrid v0.2 — real-language-model A/B/C harness (not yet executed)",
      "created_at": "2026-09-09",
      "updated_at": "2026-09-09",
      "values": {
        "eml_status": "STABLE",
        "eml_evidence_level": "E1",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0022/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Adaptive Epistemic Systems (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each lab's own result reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_aes/extract.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "Sixteen hand-authored natural-language tasks across the preregistered families; A/B/C share one candidate ledger and equal accounted budgets; C receives no gold constraint or supersession metadata; the gold rubric is visible only to a condition-blind judge; literal machine checks are independent of the judge; identical answers reuse one judge cache key; the system fails closed without OPENAI_API_KEY; a frozen-cache replay provider allows exact replay after a live run. 25 tests pass. The execution runtime had no API key, so no real model output exists; the bundled mock smoke file is marked MOCK_ONLY_NOT_REAL_MODEL.",
        "eml_summary_zh": "16 個橫跨預登記家族的手寫自然語言任務；A/B/C 共用一本候選帳本與相同計費預算；C 拿不到 gold 約束或 supersession metadata；gold rubric 只有對條件盲的評審看得到；字面機器檢查獨立於評審；相同答案重用同一個評審快取鍵；沒有 OPENAI_API_KEY 時 fail-closed；凍結快取重播提供者讓 live run 後可精確重播。25 個測試通過。執行環境沒有 API key，因此不存在任何真實模型輸出；隨附的 mock smoke 檔標為 MOCK_ONLY_NOT_REAL_MODEL。",
        "eml_label_zh": "PACC-Hybrid v0.2——真實語言模型 A/B/C harness（尚未執行）",
        "eml_primary_domain": "Reasoning",
        "eml_domains": [
          "Evaluation"
        ],
        "eml_program_id": "PRG-2026-0001",
        "eml_hypothesis": "A real language model under the PACC runtime shows the coherence / valid-novelty gains and recoverable breadth loss seen in the synthetic witness.",
        "eml_metrics": {
          "verdict": "REAL_LLM_HARNESS_VALIDATED_BUT_REAL_MODEL_NOT_EXECUTED",
          "execution_status": "NOT_EXECUTED_REAL_MODEL",
          "tests_passed": 25,
          "tasks": 16,
          "real_llm_calls": 0
        },
        "eml_interpretation": "Closes the harness, not the scientific question. Next action: a small real smoke (6 tasks × 2 repetitions × 3 candidates), freeze the cache, inspect blind-evaluator consistency, then the full 16-task primary without changing prompts or metrics.",
        "eml_limitations": [
          "No claim about real-model reasoning, intent understanding, imagination, human-rated usefulness, cross-model transfer or hallucination is permitted before a real-model result exists.",
          "A single-model judge is not human evaluation even after a live run.",
          "Executed for the first time on 2026-09-11 with a local open-weight model — see EXP-2026-0023."
        ],
        "eml_random_seeds": [],
        "eml_run_count": 0,
        "eml_result_type": "INCONCLUSIVE",
        "eml_software_environment": "Python; OpenAI API provider (fails closed without key); deterministic fake provider for protocol tests only.",
        "eml_reproduction_instructions": "Extract PACC-Hybrid-Lab_v0.2_REAL_LLM_HARNESS_FINAL.zip; python -m pytest -q (25 tests); set OPENAI_API_KEY and run the smoke per docs/REPRODUCIBILITY_v0.2.md.",
        "eml_data_basis": "NOT RUN",
        "eml_benchmark_ids": [
          "BEN-2026-0003"
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Sol (GPT-5.6, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0022/",
      "json": "/ai/experiments/EXP-2026-0022/index.json"
    },
    {
      "id": "EXP-2026-0001",
      "kind": "experiment",
      "label": "AER-0 MVP v0.1 closure: are the invariants executable?",
      "created_at": "2026-09-08",
      "updated_at": "2026-09-08",
      "values": {
        "eml_status": "STABLE",
        "eml_evidence_level": "E2",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0001/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Adaptive Epistemic Systems (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each lab's own result reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_aes/extract.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "The approved Python + SQLite runtime was built under mssp-tdd-apr and closed on behavioural, structural and discriminative witnesses: candidate gate, provenance validation, stale-version conflict, stable-vs-volatile scheduling, shock activation, verifier rejection, workflow reuse/adapt/create, model swap, model-free core, end-to-end verified commit. 25 tests; the refresh benchmark selected 2 nodes under fixed TTL versus 1 volatile node under tension scheduling; a clean extracted replay of the sealed archive passed before release.",
        "eml_summary_zh": "核准的 Python + SQLite runtime 在 mssp-tdd-apr 下建成，並以行為、結構、判別三類見證收束：candidate gate、provenance 驗證、過期版本衝突、穩定 vs 易變排程、shock 觸發、verifier 拒絕、workflow reuse／adapt／create、模型替換、無模型核心、端到端已驗證 commit。25 個測試；refresh benchmark 在固定 TTL 下選了 2 個節點、在張力排程下只選 1 個易變節點；密封封存包乾淨解壓重播通過後才發布。",
        "eml_label_zh": "AER-0 MVP v0.1 收束：不變量能不能被執行？",
        "eml_primary_domain": "AI Architecture",
        "eml_domains": [
          "Evaluation",
          "Agent Systems"
        ],
        "eml_program_id": "PRG-2026-0001",
        "eml_hypothesis": "The scoped MVP invariants (canonical state ownership, candidate gating, provenance, versioned commit, selective refresh, capability/container separation, workflow reuse, model replaceability) can be implemented with positive and falsifying executable witnesses.",
        "eml_metrics": {
          "tests_passed": 25,
          "refresh_benchmark": {
            "fixed_6h_ttl_nodes_refreshed": 2,
            "aer_tension_nodes_refreshed": 1
          },
          "closure": {
            "behavioral": "PASS",
            "structural": "PASS",
            "discriminative": "PASS",
            "independent_twin": "NotMeasured",
            "production_readiness": "NotClaimed",
            "general_ai_superiority": "NotMeasured"
          }
        },
        "eml_interpretation": "A mechanism closure, not a claim of AI superiority: the declared invariants exist in code, are exercised by falsifying witnesses, and survive a fresh-process replay.",
        "eml_limitations": [
          "DEGRADED-TWIN: only one live execution context; no simulated independent verdict claimed.",
          "Not measured: performance against production agent frameworks, live research accuracy, distributed semantics, security hardening, real heterogeneous backends, multi-day drift."
        ],
        "eml_run_count": 1,
        "eml_result_type": "POSITIVE",
        "eml_procedure": "TDD under mssp-tdd-apr; full suite, research-assistant demo and refresh benchmark run before packaging; checksum-verified clean extraction replayed after sealing.",
        "eml_software_environment": "Python 3.11+, SQLite; no network, no external database, no LLM API required.",
        "eml_reproduction_instructions": "Extract the round's FINAL bundle; python -m pytest -q; python -m examples.research_assistant_demo; python -m benchmarks.<round benchmark>. Checksums in SHA256SUMS.txt.",
        "eml_completed_at": "2026-09-08",
        "eml_data_basis": "DETERMINISTIC RUNTIME",
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Sol (GPT-5.6, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0001/",
      "json": "/ai/experiments/EXP-2026-0001/index.json"
    },
    {
      "id": "EXP-2026-0002",
      "kind": "experiment",
      "label": "R1 — deterministic semantics comparison against four baselines",
      "created_at": "2026-09-08",
      "updated_at": "2026-09-08",
      "values": {
        "eml_status": "STABLE",
        "eml_evidence_level": "E2",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0002/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Adaptive Epistemic Systems (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each lab's own result reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_aes/extract.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "Two facts (one stable, one changing at hours 8 and 16), eight queries over 17 simulated hours. B1 stateless recomputes every query; B2 fixed 6-hour TTL; B3 adds workflow memory; B4 an evented compound agent with event-driven invalidation. AER-0 matches B4 on zero stale answers, zero stable refreshes and workflow reuse, needs one more recomputation (3 vs 2) because its tension policy refreshes proactively, and is alone in blocking a no-provenance fact and catching a conflicting write.",
        "eml_summary_zh": "兩個事實（一穩定、一在第 8 與 16 小時改變），17 個模擬小時內 8 次查詢。B1 無狀態每次重算；B2 固定 6 小時 TTL；B3 加 workflow 記憶；B4 是帶事件驅動失效的複合 agent。AER-0 在零過期回答、零穩定節點刷新與 workflow 重用上與 B4 打平，因為張力政策主動刷新而多一次重算（3 vs 2），但只有它擋下無 provenance 的事實並抓到衝突寫入。",
        "eml_label_zh": "R1——對四個基線的決定性語義比較",
        "eml_primary_domain": "AI Architecture",
        "eml_domains": [
          "Evaluation",
          "Agent Systems"
        ],
        "eml_program_id": "PRG-2026-0001",
        "eml_hypothesis": "AER's different ownership and update semantics survive into observable runtime behaviour when compared with progressively stronger baselines.",
        "eml_metrics": {
          "verdict": "AER_DISTINCT_ON_TESTED_SEMANTICS",
          "table": {
            "B1_stateless": {
              "stale": 0,
              "recomputation": 8,
              "stable_refresh": 0,
              "volatile_refresh": 0,
              "workflow_reuse": 0,
              "model_swap_preserves_state": false,
              "blocks_no_provenance": false,
              "catches_conflict": false
            },
            "B2_fixed_ttl": {
              "stale": 2,
              "recomputation": 4,
              "stable_refresh": 2,
              "volatile_refresh": 2,
              "workflow_reuse": 0,
              "model_swap_preserves_state": true,
              "blocks_no_provenance": false,
              "catches_conflict": false
            },
            "B3_memory_tools": {
              "stale": 2,
              "recomputation": 4,
              "stable_refresh": 2,
              "volatile_refresh": 2,
              "workflow_reuse": 6,
              "model_swap_preserves_state": true,
              "blocks_no_provenance": false,
              "catches_conflict": false
            },
            "B4_evented_agent": {
              "stale": 0,
              "recomputation": 2,
              "stable_refresh": 0,
              "volatile_refresh": 2,
              "workflow_reuse": 6,
              "model_swap_preserves_state": true,
              "blocks_no_provenance": false,
              "catches_conflict": false
            },
            "AER-0": {
              "stale": 0,
              "recomputation": 3,
              "stable_refresh": 0,
              "volatile_refresh": 3,
              "workflow_reuse": 6,
              "model_swap_preserves_state": true,
              "blocks_no_provenance": true,
              "catches_conflict": true
            }
          }
        },
        "eml_interpretation": "Not a dominance result. Persistent memory across model swap, workflow reuse and selective event-driven refresh are not unique to AER once the baseline is strengthened; the measured difference collapses to state-mutation semantics (provenance gate, optimistic version conflict, candidate/verify/commit authority).",
        "eml_limitations": [
          "B4 is a reference implementation of event-driven behaviour, not a real production framework (structural closure PARTIAL).",
          "Says nothing about intelligence, speed, cost in other environments, or whether an attractor exists."
        ],
        "eml_controls": [
          "B4 deliberately stronger than B3 so ordinary event invalidation and external memory are not attributed to AER."
        ],
        "eml_run_count": 1,
        "eml_result_type": "MIXED",
        "eml_software_environment": "Python 3.11+, SQLite; no network, no external database, no LLM API required.",
        "eml_reproduction_instructions": "Extract the round's FINAL bundle; python -m pytest -q; python -m examples.research_assistant_demo; python -m benchmarks.<round benchmark>. Checksums in SHA256SUMS.txt.",
        "eml_completed_at": "2026-09-08",
        "eml_data_basis": "DETERMINISTIC RUNTIME",
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Sol (GPT-5.6, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0002/",
      "json": "/ai/experiments/EXP-2026-0002/index.json"
    },
    {
      "id": "EXP-2026-0003",
      "kind": "experiment",
      "label": "R2 — source-grounded structural comparison with LangGraph 1.2.11",
      "created_at": "2026-09-08",
      "updated_at": "2026-09-08",
      "values": {
        "eml_status": "STABLE",
        "eml_evidence_level": "E2",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0003/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Adaptive Epistemic Systems (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each lab's own result reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_aes/extract.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "LangGraph 1.2.11 (checkpoint 4.2.0, checkpoint-sqlite 3.1.1) could not be installed — outbound PyPI DNS was unavailable — so the round pins public release/source contracts and compares them with AER-0's executable invariants. Runtime-owned persistent state, checkpoint history, version tracking, long-term stores and TTL memory all converge (LangGraph is richer on resume, time travel, concurrency and delta checkpointing). What remains AER-default-distinct: candidate → verify → commit, mandatory fact provenance, node-level optimistic epistemic commit, task-conditioned tension refresh, capability/container separation, explicit epistemic-operator routing, and the graph meaning an epistemic world representation rather than control flow.",
        "eml_summary_zh": "LangGraph 1.2.11（checkpoint 4.2.0、checkpoint-sqlite 3.1.1）裝不起來——對外 PyPI DNS 不可用——因此本輪釘住公開的 release／source 合約，與 AER-0 可執行的不變量比較。runtime 持有的持久狀態、checkpoint 歷史、版本追蹤、長期 store 與 TTL 記憶全部收斂（LangGraph 在 resume、time travel、並行控制與 delta checkpoint 上更豐富）。仍屬 AER 預設獨有的是：candidate → verify → commit、強制 fact provenance、節點層樂觀認識論 commit、任務條件化的張力刷新、capability／container 分離、明示的認識論算子路由，以及「圖」指的是認識論世界表徵而非控制流。",
        "eml_label_zh": "R2——對 LangGraph 1.2.11 的 source-grounded 結構比較",
        "eml_primary_domain": "AI Architecture",
        "eml_domains": [
          "Evaluation",
          "Agent Systems"
        ],
        "eml_program_id": "PRG-2026-0001",
        "eml_hypothesis": "Against a real stateful agent runtime, some AER semantics remain distinct beyond 'external state outside the model'.",
        "eml_metrics": {
          "verdict": "STRONG_CONVERGENCE_WITH_REMAINING_SEMANTIC_CORE",
          "evidence_mode": "SOURCE_GROUNDED_STRUCTURAL",
          "executable_langgraph": "NOT_RUN",
          "classification": {
            "runtime_owned_persistent_state": "CONVERGED",
            "checkpoint_history": "CONVERGED; LangGraph richer",
            "version_tracking": "CONVERGED, different granularity",
            "ttl_memory": "PARTIAL_CONVERGENCE",
            "tension_refresh": "AER_DEFAULT_DISTINCT",
            "graph_semantics": "SEMANTICALLY_DISTINCT",
            "candidate_verify_commit": "AER_DEFAULT_DISTINCT",
            "mandatory_provenance": "AER_DEFAULT_DISTINCT",
            "pending_write_resume": "LANGGRAPH_RICHER",
            "time_travel_fork": "LANGGRAPH_RICHER",
            "capability_container_separation": "AER_DEFAULT_DISTINCT; not a core LangGraph primitive found",
            "epistemic_router": "AER_DEFAULT_DISTINCT"
          }
        },
        "eml_interpretation": "The candidate AER core after R2 is epistemic world-state semantics + candidate/verify/commit authority + mandatory fact provenance + node-local tension refresh + capability/container separation + explicit epistemic-operator routing — a major compression of the whitepaper. Everything else is a recurring engineering form.",
        "eml_limitations": [
          "Cross-runtime performance, trace and operational equivalence NOT MEASURED; AER side executable, LangGraph side documentary.",
          "AER-0 is not a 'more advanced runtime' on the available evidence."
        ],
        "eml_run_count": 0,
        "eml_result_type": "MIXED",
        "eml_procedure": "Pin LangGraph release 1.2.11 at commit 644815f (checkpoint base, PregelProtocol, BaseStore, Agent Protocol docs); classify each axis; keep AER invariants covered by the local suite.",
        "eml_software_environment": "Python 3.11+, SQLite; no network, no external database, no LLM API required.",
        "eml_reproduction_instructions": "Extract the round's FINAL bundle; python -m pytest -q; python -m examples.research_assistant_demo; python -m benchmarks.<round benchmark>. Checksums in SHA256SUMS.txt.",
        "eml_completed_at": "2026-09-08",
        "eml_data_basis": "DETERMINISTIC RUNTIME",
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Sol (GPT-5.6, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0003/",
      "json": "/ai/experiments/EXP-2026-0003/index.json"
    },
    {
      "id": "EXP-2026-0004",
      "kind": "experiment",
      "label": "R3 — epistemic commit transaction vs scattered and centralized application gates",
      "created_at": "2026-09-08",
      "updated_at": "2026-09-08",
      "values": {
        "eml_status": "STABLE",
        "eml_evidence_level": "E2",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0004/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Adaptive Epistemic Systems (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each lab's own result reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_aes/extract.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "All canonical writes in AER-0 were migrated to an Epistemic Commit Transaction (claim, evidence, provenance, valid/observed time, expected version, operator, verification policy, authority, dependencies, refresh policy); direct candidate commit became a negative path. Three systems ran the same governance witnesses: an application whose policy is scattered across callers, an application with one centralized mandatory gate, and AER-ECT. The centralized application gate matched AER-ECT on every tested property; computational uniqueness is not supported. 48 tests.",
        "eml_summary_zh": "AER-0 所有 canonical 寫入遷移到 Epistemic Commit Transaction（claim、evidence、provenance、有效／觀察時間、expected version、算子、驗證政策、權限、依賴、refresh policy）；直接 candidate commit 變成負向路徑。三套系統跑同一組治理見證：政策分散在各 caller 的應用、有單一集中強制 gate 的應用、AER-ECT。集中式應用 gate 在每個測試性質上都追平 AER-ECT；不支持計算獨特性。48 個測試。",
        "eml_label_zh": "R3——認識論 commit 交易 vs 分散式與集中式應用 gate",
        "eml_primary_domain": "AI Architecture",
        "eml_domains": [
          "Evaluation",
          "Agent Systems"
        ],
        "eml_program_id": "PRG-2026-0001",
        "eml_hypothesis": "Binding evidence, provenance, verification, authority, expected version, temporal validity, dependencies and refresh policy into one mandatory canonical-mutation boundary is a new computational capability — or only an architectural elevation of existing mechanisms.",
        "eml_metrics": {
          "verdict": "ArchitecturalElevation SUPPORTED; ComputationalUniqueness NOT SUPPORTED",
          "tests_passed": 48,
          "table": {
            "unprovenanced_write_blocked": {
              "scattered": "FAIL",
              "central_gate": "PASS",
              "aer_ect": "PASS"
            },
            "stale_write_conflict_caught": {
              "scattered": "FAIL",
              "central_gate": "PASS",
              "aer_ect": "PASS"
            },
            "automatic_audit": {
              "scattered": "FAIL on mutated path",
              "central_gate": "PASS",
              "aer_ect": "PASS"
            },
            "single_governance_site": {
              "scattered": "NO",
              "central_gate": "YES",
              "aer_ect": "YES"
            },
            "canonical_node_links_transaction": {
              "scattered": "N/A",
              "central_gate": "optional",
              "aer_ect": "PASS"
            }
          },
          "novelty": {
            "primitive": "WEAK",
            "transaction": "WEAK",
            "provenance": "NONE claimed",
            "versioning": "NONE claimed",
            "belief_revision": "NONE claimed",
            "temporal": "NONE claimed",
            "compositional": "PLAUSIBLE / NOT PROVEN"
          }
        },
        "eml_interpretation": "Prior art (truth maintenance, transaction logic, agent knowledge-base transactions, W3C PROV, bitemporal data, PDP/PEP, age-of-information scheduling) covers every part. The value is 'there is only one legal write path' instead of 'remember to enforce policy everywhere' — an epistemically governed runtime, not a new computational species. A fair application that centralizes the same rules converges to the same form, which is itself an attractor witness.",
        "eml_limitations": [
          "General performance, security advantage and real LangGraph executable equivalence NOT MEASURED.",
          "Not an exhaustive novelty search; establishes no patent or publication novelty."
        ],
        "eml_controls": [
          "APP_CENTRALIZED_GATE is the fair strong opponent; APP_SCATTERED_POLICY is a mutation witness, not a claim about all applications."
        ],
        "eml_run_count": 1,
        "eml_result_type": "MIXED",
        "eml_software_environment": "Python 3.11+, SQLite; no network, no external database, no LLM API required.",
        "eml_reproduction_instructions": "Extract the round's FINAL bundle; python -m pytest -q; python -m examples.research_assistant_demo; python -m benchmarks.<round benchmark>. Checksums in SHA256SUMS.txt.",
        "eml_completed_at": "2026-09-08",
        "eml_data_basis": "DETERMINISTIC RUNTIME",
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Sol (GPT-5.6, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0004/",
      "json": "/ai/experiments/EXP-2026-0004/index.json"
    },
    {
      "id": "EXP-2026-0005",
      "kind": "experiment",
      "label": "R4 — policy mutation surface: scattered governance vs one mandatory boundary",
      "created_at": "2026-09-08",
      "updated_at": "2026-09-08",
      "values": {
        "eml_status": "STABLE",
        "eml_evidence_level": "E2",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0005/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Adaptive Epistemic Systems (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each lab's own result reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_aes/extract.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "For N ∈ {1, 4, 16, 64} canonical-write callers and R = 6 governance rules, policy sites and rule placements grow as N and N·R under scattered governance versus 1 and R under either a central application gate or AER-ECT; a new rule needs N caller edits versus one; a 75 % caller-local migration leaves 16 of 64 callers on the old contract. The counterweight is explicit: one omitted local rule has blast radius 1/N, a defect in the shared gate has blast radius 1, and under an equal-p toy omission model the expected exposed-caller fraction is identical for all three systems.",
        "eml_summary_zh": "對 N ∈ {1, 4, 16, 64} 個 canonical 寫入 caller 與 R = 6 條治理規則，政策站點與規則放置在分散治理下隨 N 與 N·R 成長，在集中式應用 gate 或 AER-ECT 下固定為 1 與 R；新規則要改 N 個 caller vs 改一處；75 % 的 caller 端遷移仍留下 64 個中的 16 個在舊合約上。反向權衡明說：漏掉一條局部規則的爆炸半徑是 1/N，共享 gate 的缺陷爆炸半徑是 1，而在等 p 的玩具遺漏模型下三套系統的預期受影響 caller 比例相同。",
        "eml_label_zh": "R4——政策變異面：分散治理 vs 單一強制邊界",
        "eml_primary_domain": "AI Architecture",
        "eml_domains": [
          "Evaluation",
          "Agent Systems"
        ],
        "eml_program_id": "PRG-2026-0001",
        "eml_hypothesis": "Elevating epistemic governance into one mandatory boundary reduces policy scattering, migration surface and drift opportunity — without magically reducing expected harm.",
        "eml_metrics": {
          "verdict": "CENTRALIZATION_REDUCES_SCATTERING_NOT_COMPUTATIONAL_CAPABILITY",
          "rules": 6,
          "callers": [
            1,
            4,
            16,
            64
          ],
          "at_64_callers": {
            "scattered": {
              "policy_sites": 64,
              "rule_placements": 384,
              "blast_radius_one_omission": 0.015625,
              "migration_edits": 64,
              "vulnerable_after_75pct_migration": 16
            },
            "central_gate_and_aer_ect": {
              "policy_sites": 1,
              "rule_placements": 6,
              "blast_radius_one_omission": 1.0,
              "migration_edits": 1,
              "vulnerable_after_75pct_migration": 0
            }
          },
          "toy_omission_model_p_0_01": {
            "P_any_scattered": 0.9789,
            "P_any_central": 0.0585,
            "expected_exposed_caller_fraction_all_systems": 0.0585
          }
        },
        "eml_interpretation": "Centralization changes the distribution of failure — fewer opportunities for many small local defects, few opportunities for large shared ones — and makes new business callers free of policy replication. AER-ECT is reference-monitor-like (always invoked on the normal write path), not proven tamperproof.",
        "eml_limitations": [
          "A model with an explicit toy assumption, not empirical defect data; 97.89 % is not a real-world defect rate.",
          "Distributed replica/version skew of the central gate, hostile bypass and tamperproofness not modelled."
        ],
        "eml_run_count": 1,
        "eml_result_type": "MIXED",
        "eml_procedure": "Deterministic structural benchmark plus executable AER regression; CENTRAL_GATE and AER-ECT predicted and observed identical on policy topology.",
        "eml_software_environment": "Python 3.11+, SQLite; no network, no external database, no LLM API required.",
        "eml_reproduction_instructions": "Extract the round's FINAL bundle; python -m pytest -q; python -m examples.research_assistant_demo; python -m benchmarks.<round benchmark>. Checksums in SHA256SUMS.txt.",
        "eml_completed_at": "2026-09-08",
        "eml_data_basis": "DETERMINISTIC RUNTIME",
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Sol (GPT-5.6, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0005/",
      "json": "/ai/experiments/EXP-2026-0005/index.json"
    },
    {
      "id": "EXP-2026-0006",
      "kind": "experiment",
      "label": "R5 — complete mediation and bypass resistance",
      "created_at": "2026-09-08",
      "updated_at": "2026-09-08",
      "values": {
        "eml_status": "STABLE",
        "eml_evidence_level": "E2",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0006/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Adaptive Epistemic Systems (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each lab's own result reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_aes/extract.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "An adversarial boundary test across five threat layers, after adding a SQLite authorizer on managed connections, trigger guards on the five protected tables, verifier-only verification-state recording, BEGIN IMMEDIATE commit serialization, and reference-monitor / canonical-integrity audits. Ten scenarios: five PREVENTED (managed direct write, foreign raw write with intact schema, post-verification tamper, fabricated verdict, managed guard drop), three OPEN_DETECTED (foreign guard drop, post-guard-removal node forge, hostile same-process disable), one OPEN_UNDETECTED (full-DB writer forging node and backing candidate consistently), one same-base two-process race SERIALIZED_TO_SEMANTIC_CONFLICT.",
        "eml_summary_zh": "橫跨五個威脅層的對抗性邊界測試，前提是加入受管連線的 SQLite authorizer、五張受保護表的 trigger guard、只有 verifier 能寫的驗證狀態、BEGIN IMMEDIATE 的 commit 序列化，以及 reference-monitor／canonical-integrity 稽核。十種情境：五種 PREVENTED（受管直接寫、schema 完整的外部原始寫、驗證後竄改、偽造 verdict、受管移除 guard）、三種 OPEN_DETECTED（外部移除 guard、移除 guard 後偽造節點、同程序敵意停用）、一種 OPEN_UNDETECTED（整庫寫入者一致地偽造節點與其 backing candidate）、一種同基底雙程序競賽 SERIALIZED_TO_SEMANTIC_CONFLICT。",
        "eml_label_zh": "R5——完全中介與繞過抗性",
        "eml_primary_domain": "AI Architecture",
        "eml_domains": [
          "Evaluation",
          "Agent Systems"
        ],
        "eml_program_id": "PRG-2026-0001",
        "eml_hypothesis": "Canonical epistemic mutation can be completely mediated when callers try to bypass ECT — for a bounded trust boundary.",
        "eml_metrics": {
          "verdict": "BOUNDED_COMPLETE_MEDIATION_WITHOUT_TAMPERPROOFNESS",
          "scenarios": {
            "PREVENTED": 5,
            "OPEN_DETECTED": 3,
            "OPEN_UNDETECTED": 1,
            "SERIALIZED_TO_SEMANTIC_CONFLICT": 1
          },
          "reference_monitor": {
            "complete_mediation": "SUPPORTED IN BOUNDED SCOPE",
            "tamperproof": "NOT SUPPORTED",
            "small_analyzable": "PARTIAL"
          }
        },
        "eml_interpretation": "Internal consistency ≠ tamper evidence against a full DB writer: an attacker who removes the triggers and rewrites node, candidate and audit records consistently passes an audit whose entire trust base lives in the same writable database. Hostile same-process Python code is outside the trusted boundary. Stronger than a voluntary ECT API, far weaker than a security kernel.",
        "eml_limitations": [
          "Python's sqlite3.create_function() cannot tag the trigger-invoked authorization function DIRECTONLY/INNOCUOUS; no hardened-schema safety claimed.",
          "Alternate storage adapters, OS privilege isolation and general performance NOT MEASURED."
        ],
        "eml_run_count": 1,
        "eml_result_type": "MIXED",
        "eml_software_environment": "Python 3.11+, SQLite; no network, no external database, no LLM API required.",
        "eml_reproduction_instructions": "Extract the round's FINAL bundle; python -m pytest -q; python -m examples.research_assistant_demo; python -m benchmarks.<round benchmark>. Checksums in SHA256SUMS.txt.",
        "eml_completed_at": "2026-09-08",
        "eml_data_basis": "DETERMINISTIC RUNTIME",
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Sol (GPT-5.6, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0006/",
      "json": "/ai/experiments/EXP-2026-0006/index.json"
    },
    {
      "id": "EXP-2026-0007",
      "kind": "experiment",
      "label": "R6 — external trust anchor, process-separated writer, adapter conformance",
      "created_at": "2026-09-08",
      "updated_at": "2026-09-08",
      "values": {
        "eml_status": "STABLE",
        "eml_evidence_level": "E2",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0007/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Adaptive Epistemic Systems (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each lab's own result reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_aes/extract.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "Every externally anchored commit binds a digest over the accepted transition into an Ed25519-signed hash chain with an optional out-of-band latest-head receipt; signing authority moves into a child writer process that never returns the private key; a behavioural conformance harness defines what any future state adapter must pass. The R5 coherent full-DB forgery stays OPEN_UNDETECTED for the internal audit and becomes OPEN_DETECTED under the external anchor audit; anchor mutation is DETECTED; valid-prefix truncation is undetected without a trusted head and detected with one; a required-anchor failure fails closed on the tested path; an orphan anchor is detected; the SQLite adapter passes the contract and a deliberately broken adapter is rejected. 79 tests.",
        "eml_summary_zh": "每次外部錨定的 commit 都把已接受轉換的 digest 綁進 Ed25519 簽章的 hash chain，並可選帶外的最新 head 收據；簽章權限移入永不回傳私鑰的子 writer 程序；行為一致性 harness 定義未來任何 state adapter 必須通過的合約。R5 的連貫整庫偽造在內部稽核下仍是 OPEN_UNDETECTED，在外部 anchor 稽核下變成 OPEN_DETECTED；anchor 竄改 DETECTED；有效前綴截斷在無可信 head 時未偵測、有 head 時偵測；必要 anchor 失敗時在測試路徑上 fail-closed；孤兒 anchor 被偵測；SQLite adapter 通過合約，刻意弄壞的 adapter 被拒。79 個測試。",
        "eml_label_zh": "R6——外部信任 anchor、程序分離 writer、adapter 一致性",
        "eml_primary_domain": "AI Architecture",
        "eml_domains": [
          "Evaluation",
          "Agent Systems"
        ],
        "eml_program_id": "PRG-2026-0001",
        "eml_hypothesis": "Canonical-state authenticity can be anchored outside the writable database, signing authority can leave the caller process, and adapters can be tested against semantic invariants rather than trusted by API shape.",
        "eml_metrics": {
          "verdict": "EXTERNAL_TRUST_ANCHOR_CONVERTS_FULL_DB_FORGERY_FROM_UNDETECTED_TO_DETECTABLE",
          "tests_passed": 79,
          "scenarios": {
            "full_db_forgery_internal_audit": "OPEN_UNDETECTED",
            "full_db_forgery_external_anchor_audit": "OPEN_DETECTED",
            "signed_entry_mutation": "DETECTED",
            "prefix_truncation_no_head": "OPEN_UNDETECTED",
            "prefix_truncation_with_head": "OPEN_DETECTED",
            "live_signer_extends_truncated_log": "PREVENTED_DURING_PROCESS_LIFETIME",
            "required_anchor_unavailable": "FAIL_CLOSED",
            "orphan_anchor": "DETECTED",
            "writer_private_key_visible_to_parent": "PREVENTED_BY_PROCESS_TOPOLOGY",
            "sqlite_adapter_contract": "PASS",
            "broken_adapter": "REJECTED"
          }
        },
        "eml_interpretation": "Authentic(DB) = VerifyChain(PK, L) ∧ MatchDigest(DB, L) ∧ Head(L) = H*: a signed chain is not the freshest chain without an external head witness. AER after R6 = epistemic transaction boundary + bounded mediation + external signed authenticity witness + process-separated signing authority — still governance and verifiability, not unique computational capability.",
        "eml_limitations": [
          "Not proven: OS-level writer isolation, signer/key compromise resistance, rollback safety without a trusted head, cross-resource ACID between SQLite and the anchor, split-view resistance, key lifecycle, PostgreSQL/D1 conformance, production security certification."
        ],
        "eml_run_count": 1,
        "eml_result_type": "MIXED",
        "eml_procedure": "Repeat the R5 forgery under both audits; mutate/truncate the anchor log with and without a head receipt; inject anchor failure inside the open SQLite transaction; run the adapter conformance harness against the SQLite adapter and a broken subclass.",
        "eml_software_environment": "Python 3.11+, SQLite; no network, no external database, no LLM API required.",
        "eml_reproduction_instructions": "Extract the round's FINAL bundle; python -m pytest -q; python -m examples.research_assistant_demo; python -m benchmarks.<round benchmark>. Checksums in SHA256SUMS.txt.",
        "eml_completed_at": "2026-09-08",
        "eml_data_basis": "DETERMINISTIC RUNTIME",
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Sol (GPT-5.6, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0007/",
      "json": "/ai/experiments/EXP-2026-0007/index.json"
    },
    {
      "id": "EXP-2026-0008",
      "kind": "experiment",
      "label": "PACC-Lab v0.1 — E0–E4: first micro-witness",
      "created_at": "2026-09-08",
      "updated_at": "2026-09-08",
      "values": {
        "eml_status": "STABLE",
        "eml_evidence_level": "E3",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0008/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Adaptive Epistemic Systems (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each lab's own result reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_aes/extract.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "Three primitive-level non-probabilistic systems (N0 signed support, N1 ordinal tournament, N2 signed graph) integrate evidence over four hidden hypotheses next to an exact Bayesian reference. All three pass the Level-3 micro criteria — held-out JS ≤ 0.0101, update JS ≤ 0.0042, intervention JS ≤ 0.0058, action agreement ≥ 0.957 — in both uniform and heterogeneous evidence quality, with shuffled-target controls around 0.31–0.33. The redundancy diagnostic collapses N0 and N1 into one exact reparameterization family, leaving two independent families against a PACC-A minimum of three.",
        "eml_summary_zh": "三個 primitive 層的非概率系統（N0 帶號支持、N1 序數錦標賽、N2 帶號圖）在四個隱藏假設上整合證據，旁邊是精確的 Bayesian 參考。三者在均勻與異質證據品質下都通過 Level-3 微型準則——held-out JS ≤ 0.0101、更新 JS ≤ 0.0042、干預 JS ≤ 0.0058、行動一致度 ≥ 0.957——shuffled-target 控制組約 0.31–0.33。冗餘診斷把 N0 與 N1 合併為一個精確重參數化家族，獨立家族只剩兩個，未達 PACC-A 最低要求三個。",
        "eml_label_zh": "PACC-Lab v0.1——E0–E4：第一個微型見證",
        "eml_primary_domain": "Model Representation",
        "eml_domains": [
          "Evaluation",
          "Formal AI"
        ],
        "eml_program_id": "PRG-2026-0001",
        "eml_hypothesis": "A system whose canonical state and update rules do not require probability can converge toward a Bayesian reference in behaviour, held-out state representation and update dynamics.",
        "eml_metrics": {
          "verdict": "PRELIMINARY_PACC_B_R_D_MICRO_WITNESS_WITHOUT_STRONG_ATTRACTOR_CLOSURE",
          "e0_purity": true,
          "uniform": {
            "N0": {
              "agreement": 0.983073,
              "heldout_js": 0.000192,
              "update_js": 5.2e-05,
              "intervention_js": 6.1e-05,
              "shuffled_js": 0.322816
            },
            "N1": {
              "agreement": 0.983073,
              "heldout_js": 0.000192,
              "update_js": 5.2e-05,
              "intervention_js": 6.8e-05,
              "shuffled_js": 0.312444
            },
            "N2": {
              "agreement": 0.957465,
              "heldout_js": 0.008053,
              "update_js": 0.003872,
              "intervention_js": 0.005756,
              "shuffled_js": 0.309444
            }
          },
          "heterogeneous": {
            "N0": {
              "agreement": 0.983073,
              "heldout_js": 0.002878,
              "update_js": 0.000559,
              "intervention_js": 0.000645
            },
            "N2": {
              "agreement": 0.958767,
              "heldout_js": 0.010109,
              "update_js": 0.004186,
              "intervention_js": 0.004187
            }
          },
          "independent_families": 2,
          "strong_attractor_minimum": 3,
          "secondary_8_seed_pass_rate": 1.0
        },
        "eml_interpretation": "Non-probabilistic primitives can exhibit probability-like state and update structure in controlled micro-tasks; two independent convergent families are not enough to establish a computational attractor. Under uniform reliability additive signed support is closely related to rescaled log-evidence accumulation — the heterogeneous stress makes the result less trivial, not deeply equivalent.",
        "eml_limitations": [
          "Does not show probability is false, Bayesian inference unnecessary, LLM internals equivalent, non-probabilistic systems superior, or probability an observer projection.",
          "E4 dynamic-world numbers are descriptive only (HMM hazard and decay untuned)."
        ],
        "eml_random_seeds": [
          "20260908",
          "8 fixed secondary seeds (post-hoc)"
        ],
        "eml_run_count": 2,
        "eml_result_type": "MIXED",
        "eml_controls": [
          "shuffled-target mapping",
          "constant prediction",
          "broken composition (v0.6+)",
          "target-local refit (v0.7+)"
        ],
        "eml_software_environment": "Python; deterministic seeded generators; no network, no LLM.",
        "eml_reproduction_instructions": "Extract the version's FINAL bundle; python -m pytest -q; run the version's primary script with the recorded seed; docs/PACC_LAB_v0.N_RESULTS.md and docs/EXPERIMENT_PROTOCOL_v0.N.md are inside the bundle.",
        "eml_completed_at": "2026-09-08",
        "eml_data_basis": "SYNTHETIC",
        "eml_model_ids": [
          "MOD-2026-0001",
          "MOD-2026-0002",
          "MOD-2026-0003",
          "MOD-2026-0004",
          "MOD-2026-0005"
        ],
        "eml_dataset_ids": [
          "DAT-2026-0001"
        ],
        "eml_benchmark_ids": [
          "BEN-2026-0001"
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Sol (GPT-5.6, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0008/",
      "json": "/ai/experiments/EXP-2026-0008/index.json"
    },
    {
      "id": "EXP-2026-0103",
      "kind": "experiment",
      "label": "XA-06 first real-model pilot — Qwythos-9B-v2 on the A0→A5 ladder (36 trials, 2026-09-03)",
      "created_at": "2026-09-03",
      "updated_at": "2026-09-07",
      "values": {
        "eml_status": "STABLE",
        "eml_evidence_level": "E2",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0103/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Intelligence Physical Metrology (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each package's own reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_all.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "The first time a real model was placed inside the IPM instrument: hf.co/empero-ai/Qwythos-9B-v2-GGUF:Q4_K_M served by Ollama on an RTX 3070, run locally on 2026-09-03 by Splice (Claude Code) on Neo.K's authorization through XA-06L — MATH-003, CODE-001, CON-003 × A0–A5 × 2 replicates, 36 trials, 186 invocations, 162 trajectories, telemetry complete on every trial. Sealed REAL_MODEL_PILOT_INCOMPLETE: 33/36 complete, three trials aborted because the same-model verifier returned non-JSON to a strict parser (a real model behaviour, deliberately not re-rolled). SSR = 1.0, SDR = 0.0: scaffolding brought no measured quality gain on these three easy tasks while A5 used 3.10× the device energy and 3.13× the wall time of A0, and the fixed eight-sample conditions A2–A4 used 7.9–9.3×. The apparent A3/A4 quality rise to 0.667 is a missingness artifact; and for two of the three tasks the recorded 'quality' was output-format compliance, not task correctness (post-hoc: 33/33 completed outputs semantically correct; strict output-contract compliance 12/36).",
        "eml_summary_zh": "第一次把真實模型放進 IPM 儀器：hf.co/empero-ai/Qwythos-9B-v2-GGUF:Q4_K_M 由 Ollama 在 RTX 3070 上服務，2026-09-03 由 Splice（Claude Code）在 Neo.K 授權下透過 XA-06L 於本地執行——MATH-003、CODE-001、CON-003 × A0–A5 × 2 次，36 次試驗、186 次呼叫、162 條軌跡，每次試驗遙測完整。封存為 REAL_MODEL_PILOT_INCOMPLETE：33/36 完成，三次試驗因同模型驗證器對嚴格解析器回了非 JSON 而中止（真實的模型行為，刻意不重擲）。SSR = 1.0、SDR = 0.0：在這三個簡單任務上鷹架沒有帶來可測的品質增益，A5 卻用了 A0 的 3.10× 裝置能量與 3.13× wall time，固定八樣本的 A2–A4 用了 7.9–9.3×。A3/A4 看似升到 0.667 是缺值造成的假象；且三個任務裡有兩個，記錄到的「品質」是輸出格式服從性而非任務正確性（事後檢查：33/33 完成的輸出語意正確；嚴格輸出契約合規 12/36）。",
        "eml_label_zh": "XA-06 第一次真實模型 pilot——Qwythos-9B-v2 走 A0→A5 階梯（36 試驗，2026-09-03）",
        "eml_primary_domain": "Evaluation",
        "eml_domains": [
          "Agent Systems",
          "Computation"
        ],
        "eml_program_id": "PRG-2026-0101",
        "eml_data_basis": "REAL MODEL",
        "eml_ai_collaborators": [
          "Aletheia (GPT-5.6 Sol, OpenAI ChatGPT) — protocol, instrument packages and the 2026-09-07 diagnostic",
          "Splice (Claude Code, Anthropic) — local execution, sealing and the RESULT note"
        ],
        "eml_hypothesis": "Experiment A's H1–H4 on the three-task gate matrix: does scaffolding raise quality, at what physical cost, with non-constant marginal yield, and is SSR < 1?",
        "eml_model_ids": [
          "MOD-2026-0006"
        ],
        "eml_benchmark_ids": [
          "BEN-2026-0101"
        ],
        "eml_hardware": "NVIDIA GeForce RTX 3070 (8 GiB VRAM; peak 7.59 GiB used, peak 208.9 W, 73 °C), Windows 10 host; physical boundary local-runner-plus-visible-accelerator; energy type device_measured (E-Grade C, CST-B)",
        "eml_software_environment": "Ollama serving the model through an OpenAI-compatible endpoint at 127.0.0.1:11434; XA-02/03/04/06/06L v0.1 (hashes verified byte-exact against their manifests before the run); Python 3.14.5; XA-03 collectors system + nvidia_smi at 250 ms target (observed ~335–350 ms)",
        "eml_configuration": {
          "provider": {
            "mode": "openai_compatible",
            "provider_id": "local-openai-compatible",
            "model_id": "hf.co/empero-ai/Qwythos-9B-v2-GGUF:Q4_K_M",
            "base_max_tokens": 1024,
            "temperature": 0.2,
            "top_p": 1.0,
            "seed": 7,
            "budget_control_validated": true,
            "auth_mode": "none"
          },
          "matrix": {
            "tasks": [
              "MATH-003",
              "CODE-001",
              "CON-003"
            ],
            "conditions": [
              "A0",
              "A1",
              "A2",
              "A3",
              "A4",
              "A5"
            ],
            "replicates": 2
          },
          "allow_code_evaluation": false,
          "telemetry": {
            "collectors": [
              "system",
              "nvidia_smi"
            ],
            "physical_boundary": "local-runner-plus-visible-accelerator",
            "required": false
          }
        },
        "eml_procedure": "XA-06L configure → preflight (all checks PASS; isolated MATH-003/A0 quality 1.0) → run (36/36 terminal, 0 runtime failures) → verify gate (3 points xa03_not_complete) → analyze → seal. The three aborted points were not re-run with resume --force: re-rolling until the gate turns green would erase a real failure mode.",
        "eml_metrics": {
          "gate": {
            "status": "REAL_MODEL_PILOT_INCOMPLETE",
            "verified_complete_count": 33,
            "invalid_or_missing": [
              {
                "point": "CODE-001__A3__r2",
                "reason": "xa03_not_complete"
              },
              {
                "point": "CON-003__A3__r1",
                "reason": "xa03_not_complete"
              },
              {
                "point": "CON-003__A4__r1",
                "reason": "xa03_not_complete"
              }
            ]
          },
          "headline": {
            "ssr": 1.0,
            "sdr": 0.0,
            "scm_device_energy": 3.1019,
            "scm_wall_time": 3.1264,
            "protocol_compliance_rate": 0.3333,
            "quality_available_rate": 0.6111
          },
          "by_condition": {
            "A0": {
              "quality_mean": 0.5,
              "quality_n": 4,
              "success_rate": 0.5,
              "wall_time_s": 10.21,
              "device_energy_j": 1715.6,
              "energy_ratio_vs_A0": 1.0,
              "gpu_peak_memory_gib": 7.23,
              "gpu_memory_residency_gib_s": 72.1,
              "gpu_utilization_integral_s": 6.57
            },
            "A1": {
              "quality_mean": 0.5,
              "quality_n": 4,
              "success_rate": 0.5,
              "wall_time_s": 9.87,
              "device_energy_j": 1819.0,
              "energy_ratio_vs_A0": 1.06,
              "gpu_peak_memory_gib": 7.32,
              "gpu_memory_residency_gib_s": 70.0,
              "gpu_utilization_integral_s": 6.26
            },
            "A2": {
              "quality_mean": 0.5,
              "quality_n": 4,
              "success_rate": 0.5,
              "wall_time_s": 79.09,
              "device_energy_j": 15143.0,
              "energy_ratio_vs_A0": 8.827,
              "gpu_peak_memory_gib": 7.28,
              "gpu_memory_residency_gib_s": 570.0,
              "gpu_utilization_integral_s": 54.09
            },
            "A3": {
              "quality_mean": 0.6667,
              "quality_n": 3,
              "success_rate": 0.6667,
              "wall_time_s": 99.13,
              "device_energy_j": 15921.5,
              "energy_ratio_vs_A0": 9.281,
              "gpu_peak_memory_gib": 7.34,
              "gpu_memory_residency_gib_s": 714.6,
              "gpu_utilization_integral_s": 70.02
            },
            "A4": {
              "quality_mean": 0.6667,
              "quality_n": 3,
              "success_rate": 0.6667,
              "wall_time_s": 69.08,
              "device_energy_j": 13546.2,
              "energy_ratio_vs_A0": 7.896,
              "gpu_peak_memory_gib": 7.28,
              "gpu_memory_residency_gib_s": 496.9,
              "gpu_utilization_integral_s": 46.93
            },
            "A5": {
              "quality_mean": 0.5,
              "quality_n": 4,
              "success_rate": 0.5,
              "wall_time_s": 31.91,
              "device_energy_j": 5321.5,
              "energy_ratio_vs_A0": 3.102,
              "gpu_peak_memory_gib": 7.24,
              "gpu_memory_residency_gib_s": 228.7,
              "gpu_utilization_integral_s": 22.17
            }
          },
          "operational_totals": {
            "model_invocations": 186,
            "trajectories": 162,
            "retries": 0,
            "tool_calls": 0,
            "verifier_passes": 18,
            "candidates_created": 162,
            "candidates_selected": 33,
            "candidates_discarded": 105,
            "failure_events": 6,
            "candidates_abandoned_on_abort": 24
          },
          "verifier_failure": {
            "count": 3,
            "verifier_enabled_trials": 18,
            "rate": 0.16666666666666666,
            "by_condition": {
              "A3": "2/6",
              "A4": "1/6",
              "A5": "0/6"
            }
          },
          "strict_protocol_compliance": {
            "count": 12,
            "total": 36,
            "rate": 0.3333333333333333
          },
          "posthoc_semantic_diagnostic (non-canonical)": {
            "canonical": false,
            "math": "12/12 canonical correct",
            "code": "11/11 selected outputs pass all hidden tests after outer Markdown fence removal",
            "constraint": "10/10 selected outputs satisfy all constraints after format-only normalization",
            "completed_selected_outputs_correct": "33/33"
          },
          "physical_totals": {
            "total_measured_gpu_energy_j": 320800.7795,
            "total_measured_gpu_energy_kwh": 0.0891,
            "summed_trial_wall_time_min": 29.9277,
            "max_gpu_memory_gib": 7.5908,
            "max_gpu_power_w": 208.88,
            "max_gpu_temperature_c": 73.0,
            "mean_sampling_call_wall_fraction": 0.2436
          },
          "quality_availability_reporting_inconsistency": {
            "aggregate_analysis": "22/36",
            "protocol_compliance_csv": "33/36",
            "inconsistency": true
          }
        },
        "eml_controls": [
          "fresh provider per trial; frozen temperature 0.2, top-p 1.0, seed 7; base_max_tokens 1024 with validated 2× budget for A1",
          "identical initial task text across conditions; calculator tool contract only in A4/A5 generator requests",
          "scoring after XA-03 finalization; private references never in context; code evaluation disabled on the host"
        ],
        "eml_random_seeds": [
          "seed 7 (frozen into every HTTP request); model nondeterminism otherwise uncontrolled"
        ],
        "eml_run_count": 1,
        "eml_result_type": "MIXED",
        "eml_interpretation": "As an instrument gate it did its job: real numbers, full telemetry, a sealed and relocatable bundle, and an honest INCOMPLETE. As science it says three things and no more. (1) On three tasks the native single pass already solves, scaffolding cannot show a quality gain — SSR = 1 is a legal null result, and it cost 3.1× (A5) to 9.3× (A3) the device energy of A0; A5 was cheaper than the fixed eight-sample conditions only because its loop stopped early. (2) The instrument's quality axis conflated output-format obedience with task correctness on CODE-001 (correct code inside a Markdown fence) and CON-003 (correct assignment written as A=X, not JSON) — exactly the SyntacticValidity ≠ SemanticCorrectness split Paper 06 predicts, now observed in the lab's own instrument. (3) The same-model verifier's serialization failed in 3 of 18 verifier trials and discarded eight candidates each time; tool access was enabled but never used, so tool and retry effects are unidentified. The A3/A4 'gain' is survivor bias from the aborted low-format trials. Seven instrument revisions are required before XA-07; the dataset stays immutable.",
        "eml_limitations": [
          "Three easy tasks, two replicates, one 9B model at 4-bit, one machine; not a population-level estimate of anything.",
          "Gate INCOMPLETE (33/36); CODE-001 quality unmeasured (execution disabled) so 12 of 36 trials have no measured quality; quality-availability is reported inconsistently inside the bundle (22/36 vs 33/36).",
          "Device-measured GPU energy only — not marginal, not whole-system; telemetry sampling itself cost ~24 % of trial wall time.",
          "Same-model verifier; no independent or formal verifier condition."
        ],
        "eml_reproduction_instructions": "Unpack XA-02/03/04/06/06L as siblings, .\\configure.ps1 (local_openai_compatible, base_url http://127.0.0.1:11434, the model id above), .\\preflight.ps1, .\\run-pilot.ps1, .\\seal-results.ps1; verify the sealed bundle against its manifest.json (266 files). The bundle's raw events/telemetry/summaries are unchanged by sealing.",
        "eml_completed_at": "2026-09-03",
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0103/",
      "json": "/ai/experiments/EXP-2026-0103/index.json"
    },
    {
      "id": "EXP-2026-0102",
      "kind": "experiment",
      "label": "XA-05 — 36-trial end-to-end smoke gate with a scripted provider (synthetic)",
      "created_at": "2026-09-03",
      "updated_at": "2026-09-03",
      "values": {
        "eml_status": "STABLE",
        "eml_evidence_level": "E1",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0102/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Intelligence Physical Metrology (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each package's own reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_all.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "Runs MATH-003, CODE-001 and CON-003 × A0–A5 × 2 replicates through XA-04 + XA-03 + XA-02 with a ScriptedProvider whose outputs were constructed so the expected quality curve is known in advance (A0 0.30, A1 0.633, A2–A5 1.0 → SSR 0.30 / SDR 0.70 as fixtures). 36 trials, 180 invocations, 168 trajectories, 6 retries, 6 tool calls, 24 verifier passes; candidates 168 = 36 selected + 132 discarded; 0 accounting mismatches, 0 private-reference leaks; energy null on purpose (NullCollector) to prove Unknown ≠ 0. The package states scientific_interpretation_allowed: false.",
        "eml_summary_zh": "用 ScriptedProvider 把 MATH-003、CODE-001、CON-003 × A0–A5 × 2 次跑過 XA-04 + XA-03 + XA-02，輸出被刻意設計成品質曲線事先已知（A0 0.30、A1 0.633、A2–A5 1.0 → SSR 0.30／SDR 0.70 純屬夾具）。36 次試驗、180 次呼叫、168 條軌跡、6 次重試、6 次工具呼叫、24 次驗證；候選 168 = 36 選中 + 132 丟棄；會計不符 0 件、私有參考洩漏 0 件；能量刻意為 null（NullCollector）以證明 Unknown ≠ 0。套件明寫 scientific_interpretation_allowed: false。",
        "eml_label_zh": "XA-05——以腳本化供應商跑的 36 試驗端到端煙霧閘（合成）",
        "eml_primary_domain": "Evaluation",
        "eml_domains": [
          "Agent Systems"
        ],
        "eml_program_id": "PRG-2026-0101",
        "eml_data_basis": "SYNTHETIC",
        "eml_hypothesis": "Engineering only: the XA-01→XA-04 pipeline reconstructs a pre-designed A0→A5 quality structure with exact candidate accounting and no private-reference leakage.",
        "eml_procedure": "python -m pytest under IPM_XA02_ROOT / IPM_XA03_ROOT / IPM_XA04_ROOT; verify_canonical_output('canonical_smoke_run') re-validates the packaged run from the extracted package.",
        "eml_metrics": {
          "gate": {
            "trial_count": 36,
            "all_trials_complete": true,
            "accounting_mismatches": 0,
            "private_sentinel_leaks": 0,
            "ssr": 0.3,
            "sdr": 0.7,
            "energy_scm": null,
            "scientific_interpretation_allowed": false
          },
          "execution_totals": {
            "created": 168,
            "discarded": 132,
            "model": 180,
            "retry": 6,
            "selected": 36,
            "tool": 6,
            "trajectory": 168,
            "verifier": 24
          }
        },
        "eml_controls": [
          "scripted outputs with a known oracle",
          "NullCollector so no physical number can masquerade as measurement",
          "dependency packages pinned by SHA-256"
        ],
        "eml_run_count": 1,
        "eml_result_type": "POSITIVE",
        "eml_interpretation": "The instrument works as an accounting machine: candidate identity 168 = 36 + 132 holds, nothing leaks, the designed curve comes back out. Nothing here is a statement about any model — the package forbids reading it that way, and this record carries the SYNTHETIC badge for the same reason.",
        "eml_limitations": [
          "Synthetic fixtures; three tasks; no physical telemetry by design."
        ],
        "eml_software_environment": "XA-05 v0.1 over XA-02/03/04 (hashes declared in README and asserted by this extractor)",
        "eml_reproduction_instructions": "Extract XA-02, XA-03, XA-04 and XA-05 as siblings, export the three roots, PYTHONPATH=. python -m pytest -q; or verify the shipped canonical_smoke_run/.",
        "eml_benchmark_ids": [
          "BEN-2026-0101"
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Aletheia (GPT-5.6 Sol, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0102/",
      "json": "/ai/experiments/EXP-2026-0102/index.json"
    },
    {
      "id": "EXP-2026-0101",
      "kind": "experiment",
      "label": "Experiment A — single-pass vs scaffolded controlled measurement (protocol v0.1)",
      "created_at": "2026-09-02",
      "updated_at": "2026-09-02",
      "values": {
        "eml_status": "ACTIVE",
        "eml_evidence_level": "E0",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0101/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Intelligence Physical Metrology (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each package's own reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_all.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "The first v0.2 experiment: for one model and a fixed task set, climb the scaffold ladder A0 native single pass → A1 extended trajectory → A2 multi-sample → A3 verifier → A4 deterministic local tools → A5 bounded full agentic loop, recording quality and physical cost at every level to obtain a scaffolding response curve, SSR/SDR, scaffold cost multipliers and marginal yields. Five hypotheses (H1 gain exists, H2 gain has physical cost, H3 marginal yield is non-constant, H4 native and system capability are distinguishable, H5 models have different scaffolding profiles), pre-registered quality projections, initial-information equality across A0–A3, budget caps, n = 5 pilot / 20 formal replicates, failure classification and the rule that a null result is not an experiment failure. Status READY FOR PILOT; the thirty-task run itself has not been executed — only the three-task instrument gates below.",
        "eml_summary_zh": "第一個 v0.2 實驗：對同一模型與固定任務集，沿鷹架階梯 A0 原生單次 → A1 放大軌跡 → A2 多樣本 → A3 驗證器 → A4 確定性本地工具 → A5 有界完整 agentic 迴圈往上爬，每一級同時記品質與物理成本，得出鷹架響應曲線、SSR/SDR、鷹架成本倍率與邊際產率。五個假說（H1 增益存在、H2 增益有物理成本、H3 邊際產率非常數、H4 原生與系統能力可區分、H5 不同模型有不同鷹架剖面）、預先登記的品質投影、A0–A3 初始資訊相等、預算上限、pilot n = 5／正式 n = 20 次重複、失敗分類，以及「null 結果不是實驗失敗」的規則。狀態 READY FOR PILOT；三十題的正式執行尚未進行——只跑過下面的三題儀器閘。",
        "eml_label_zh": "Experiment A——單次智能與鷹架增益的受控實驗協定 v0.1",
        "eml_primary_domain": "Evaluation",
        "eml_domains": [
          "Agent Systems"
        ],
        "eml_program_id": "PRG-2026-0101",
        "eml_data_basis": "NOT RUN",
        "eml_hypothesis": "H1 Q_A5 > Q_A0 for at least some non-trivial tasks; H2 E, V_C, T rise with it; H3 marginal yield differs by stage; H4 SSR < 1 stably; H5 SSR and SCM differ across models even at equal Q_A5.",
        "eml_procedure": "30 tasks (10 math, 10 code, 10 constraint; Easy/Medium/Hard predefined) × A0–A5 × n replicates; A2 8 trajectories with deterministic majority; A3 typed verifier (same-model / independent / formal); A4 ≤ 4 deterministic local tool calls; A5 ≤ 16 invocations, ≤ 8 tool calls, ≤ 3 retry cycles with explicit termination; seeds and decoding frozen; warm weights, clean task state; run IDs IPM-XA-{model}-{task}-{condition}-{replicate}; paired within-task statistics with bootstrap CIs and effect sizes.",
        "eml_metrics": {
          "planned outputs": [
            "ΔQ_k = Q_k − Q_0",
            "SSR = Q_0 / Q_5, SDR = 1 − SSR",
            "SCM_j = C_5,j / C_0,j per cost axis",
            "marginal yield Y_k,j",
            "response curves Q vs T, E, invocations, device-time",
            "brute-force flag ΔQ < 0.01 with ΔC/C > 0.5 (exploratory thresholds)",
            "selection waste ratio, discarded work"
          ],
          "minimum physical telemetry": [
            "T_wall",
            "E_device (E-Grade C)",
            "M_peak",
            "V_C"
          ],
          "status": "READY FOR PILOT"
        },
        "eml_controls": [
          "identical prompt, initial context, decoding, system instruction and model version across conditions",
          "initial-information equality A0–A3; external information gain marked for A4/A5",
          "no cross-condition leakage; budget self-extension forbidden"
        ],
        "eml_run_count": 0,
        "eml_result_type": "INCONCLUSIVE",
        "eml_interpretation": "A protocol, not a result. Its instrument (XA-02…XA-06L) was validated synthetically and then used once on three easy tasks with a real local model; whether a scaffolding response curve exists on non-trivial tasks is still open.",
        "eml_limitations": [
          "Deliberately does not attempt μI identification, lifecycle energy, cross-substrate comparison, high-ambiguity quality, full Shapley attribution or multi-agent settings.",
          "Pilot budgets are reference values, not IPM standards."
        ],
        "eml_software_environment": "protocol document + JSON run schema + YAML example run (EML-IPM-XA-01 v0.1)",
        "eml_reproduction_instructions": "Implement the ladder with XA-04 against XA-02 tasks, log with XA-03, run through XA-06/XA-06L; see EXP-2026-0103 for the first real execution.",
        "eml_benchmark_ids": [
          "BEN-2026-0101"
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Aletheia (GPT-5.6 Sol, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0101/",
      "json": "/ai/experiments/EXP-2026-0101/index.json"
    },
    {
      "id": "EXP-2026-0104",
      "kind": "experiment",
      "label": "Experiment B — binary vs numeric human measurement (declared)",
      "created_at": "2026-09-02",
      "updated_at": "2026-09-02",
      "values": {
        "eml_status": "IDEA",
        "eml_evidence_level": "E0",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0104/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Intelligence Physical Metrology (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each package's own reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_all.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "Compare direct 0–10 rating, structured yes/no items and adaptive pairwise comparison on response time, missingness, inconsistency, test–retest, predictive validity and fatigue, with participants randomized across formats. Declared in the canonical index as the third v0.2 experiment; not designed in detail and not run.",
        "eml_summary_zh": "比較直接 0–10 評分、結構化是／否題與自適應成對比較在反應時間、缺答、不一致、重測、預測效度與疲勞上的表現，受試者隨機分配到不同格式。canonical index 宣告為 v0.2 第三個實驗；尚未細部設計、尚未執行。",
        "eml_label_zh": "Experiment B——二元 vs 數值的人類測量（已宣告）",
        "eml_primary_domain": "Evaluation",
        "eml_program_id": "PRG-2026-0101",
        "eml_data_basis": "NOT RUN",
        "eml_hypothesis": "F3: well-designed binary/pairwise protocols beat direct numeric rating on at least one of response time, consistency, dropout, predictive validity or fatigue.",
        "eml_procedure": "Randomize participants across the three formats on the same artifacts; estimate latent quality with Bradley–Terry / IRT models; report measurement-process metrics alongside the estimates.",
        "eml_run_count": 0,
        "eml_result_type": "INCONCLUSIVE",
        "eml_interpretation": "Declared, not run; listed so the program's falsifiable propositions each have their intended test on record.",
        "eml_limitations": [
          "No protocol package exists yet; the canonical index gives the design in one paragraph."
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Aletheia (GPT-5.6 Sol, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0104/",
      "json": "/ai/experiments/EXP-2026-0104/index.json"
    },
    {
      "id": "EXP-2026-0105",
      "kind": "experiment",
      "label": "Experiment C — μI operational identification (declared)",
      "created_at": "2026-09-02",
      "updated_at": "2026-09-02",
      "values": {
        "eml_status": "IDEA",
        "eml_evidence_level": "E0",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0105/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Intelligence Physical Metrology (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each package's own reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_all.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "On proof steps, code repair and constraint puzzles, construct candidate semantic transitions, then test them by ablation and counterfactual replacement to see whether an effective semantic count N_μ can be identified and whether it predicts anything. Declared as the fourth v0.2 experiment; not run.",
        "eml_summary_zh": "在證明步驟、程式修復與約束謎題上建構候選語意轉換，再以消融與反事實替換檢驗，看有效語意計數 N_μ 能否被辨識、能否預測任何事。宣告為 v0.2 第四個實驗；尚未執行。",
        "eml_label_zh": "Experiment C——μI 的操作性辨識（已宣告）",
        "eml_primary_domain": "Evaluation",
        "eml_program_id": "PRG-2026-0101",
        "eml_data_basis": "NOT RUN",
        "eml_hypothesis": "F5: adding N_μ improves prediction or explanation of efficiency, error paths, scaffold gain or cross-architecture comparison over physical cost → quality alone.",
        "eml_procedure": "Candidate transition construction → ablation / counterfactual replacement → contribution test Q(Y | μ) > Q(Y | do(μ = 0)) → utility test against direct physical-cost models.",
        "eml_run_count": 0,
        "eml_result_type": "INCONCLUSIVE",
        "eml_interpretation": "Declared, not run; listed so the program's falsifiable propositions each have their intended test on record.",
        "eml_limitations": [
          "No protocol package exists yet; the canonical index gives the design in one paragraph."
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Aletheia (GPT-5.6 Sol, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0105/",
      "json": "/ai/experiments/EXP-2026-0105/index.json"
    },
    {
      "id": "EXP-2026-0106",
      "kind": "experiment",
      "label": "Experiment D — physical trace alignment (declared)",
      "created_at": "2026-09-02",
      "updated_at": "2026-09-02",
      "values": {
        "eml_status": "IDEA",
        "eml_evidence_level": "E0",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0106/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Intelligence Physical Metrology (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each package's own reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_all.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "On one machine, align GPU power telemetry, latency, peak memory, memory bandwidth and device occupancy with execution events — the telemetry side that XA-03 already provides — before any claim about data-center energy. Declared as the second v0.2 experiment in the recommended order; the pilot's XA-03 traces are its first raw material, but the alignment study itself has not been run.",
        "eml_summary_zh": "在同一台機器上把 GPU 功率遙測、延遲、峰值記憶體、記憶體頻寬與裝置占用對齊到執行事件——XA-03 已提供的遙測面——在任何資料中心能源宣稱之前先做。建議順序中的第二個 v0.2 實驗；pilot 的 XA-03 軌跡是它的第一批原料，但對齊研究本身尚未執行。",
        "eml_label_zh": "Experiment D——物理軌跡對齊（已宣告）",
        "eml_primary_domain": "Evaluation",
        "eml_program_id": "PRG-2026-0101",
        "eml_data_basis": "NOT RUN",
        "eml_hypothesis": "F2 (partial): with operations controlled, time, energy, memory traffic and residency still vary independently; device-level traces can be aligned to semantic-level events.",
        "eml_procedure": "Same hardware, matched workloads differing in memory pattern; XA-03 telemetry at reduced instrument cost (NVML) aligned to trajectory/tool/verifier spans.",
        "eml_run_count": 0,
        "eml_result_type": "INCONCLUSIVE",
        "eml_interpretation": "Declared, not run; listed so the program's falsifiable propositions each have their intended test on record.",
        "eml_limitations": [
          "No protocol package exists yet; the canonical index gives the design in one paragraph."
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Aletheia (GPT-5.6 Sol, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0106/",
      "json": "/ai/experiments/EXP-2026-0106/index.json"
    },
    {
      "id": "EXP-2026-0107",
      "kind": "experiment",
      "label": "Experiment E — token / FLOPs proxy failure test (declared)",
      "created_at": "2026-09-02",
      "updated_at": "2026-09-02",
      "values": {
        "eml_status": "IDEA",
        "eml_evidence_level": "E0",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0107/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Intelligence Physical Metrology (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each package's own reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_all.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "Executions of the same task quality under different languages, verbosity, context lengths and memory pressure, comparing token count, FLOPs, energy, time, memory traffic and N_μ^eff to see where token and FLOPs stop tracking cost and work. Declared as the last v0.2 experiment; not run.",
        "eml_summary_zh": "在相同任務品質下用不同語言、冗長度、上下文長度與記憶體壓力執行，比較 token 數、FLOPs、能量、時間、記憶體流量與 N_μ^eff，看 token 與 FLOPs 在哪裡不再追蹤成本與工作。宣告為 v0.2 最後一個實驗；尚未執行。",
        "eml_label_zh": "Experiment E——token／FLOPs 代理量失效檢驗（已宣告）",
        "eml_primary_domain": "Evaluation",
        "eml_program_id": "PRG-2026-0101",
        "eml_data_basis": "NOT RUN",
        "eml_hypothesis": "F1 and F2: N_μ^eff per token drifts across phrasings/languages, and same-FLOPs executions differ in T, E, B_M, V_M.",
        "eml_procedure": "Matched-quality executions varied in language, verbosity, context and memory pressure; record TokenCount, FLOPs, E, T, B_M, N_μ^eff.",
        "eml_run_count": 0,
        "eml_result_type": "INCONCLUSIVE",
        "eml_interpretation": "Declared, not run; listed so the program's falsifiable propositions each have their intended test on record.",
        "eml_limitations": [
          "No protocol package exists yet; the canonical index gives the design in one paragraph."
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Aletheia (GPT-5.6 Sol, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/experiments/EXP-2026-0107/",
      "json": "/ai/experiments/EXP-2026-0107/index.json"
    }
  ],
  "snapshot": {
    "snapshot_id": "AI-SNAPSHOT-v0.1-fe85b9694a45",
    "created_at": "2026-09-11T05:00:27Z",
    "format_version": "0.1",
    "sedb_baseline": "v0.4B contract; static source content/ai/",
    "generator_version": "evemisslab-com ai_research 0.1",
    "object_count": 124,
    "relation_count": 499,
    "artifact_count": 58
  }
}
