{
  "section": "benchmarks",
  "kind": "benchmark",
  "canonical": "https://evemisslab.com/ai/benchmarks/",
  "count": 4,
  "records": [
    {
      "id": "BEN-2026-0001",
      "kind": "benchmark",
      "label": "PACC micro-lab protocol v0.1 (frozen gates)",
      "created_at": "2026-09-08",
      "updated_at": "2026-09-09",
      "values": {
        "eml_status": "STABLE",
        "eml_evidence_level": "E2",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/benchmarks/BEN-2026-0001/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Adaptive Epistemic Systems (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each lab's own result reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_aes/extract.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "The preregistered decision rule reused unchanged from v0.1 to v0.13: E0 purity, held-out representation distance D_R = E[JS(Φ(S_N), S_P)] ≤ 0.05, update commutation D_U ≤ 0.05, off-manifold intervention D_I ≤ 0.08, action agreement ≥ 0.85, a shuffled-target negative control the real map must beat, a design-independence rule (agreement ≥ 0.999 and R² ≥ 0.995 collapse two designs into one family), verdict levels 0–3, and a strong PACC-A gate of at least three independent convergent families.",
        "eml_summary_zh": "從 v0.1 到 v0.13 原封重用的預登記判決規則：E0 純度、held-out 表徵距離 D_R = E[JS(Φ(S_N), S_P)] ≤ 0.05、更新交換 D_U ≤ 0.05、離流形干預 D_I ≤ 0.08、行動一致度 ≥ 0.85、真實映射必須勝過的 shuffled-target 負控制、設計獨立規則（一致度 ≥ 0.999 且 R² ≥ 0.995 即視為同一家族）、判決等級 0–3，以及至少三個獨立收斂家族的強 PACC-A 門檻。",
        "eml_label_zh": "PACC 微型實驗室協定 v0.1（凍結門檻）",
        "eml_primary_domain": "Evaluation",
        "eml_domains": [
          "Model Representation"
        ],
        "eml_program_id": "PRG-2026-0001",
        "eml_purpose": "Make 'similar' and 'probability-like' non-negotiable after the data are seen.",
        "eml_tasks": [
          "Sequential evidence integration over 4 hidden hypotheses × 6 binary prototypes (uniform reliability 0.80; heterogeneous 0.60–0.95 with only ordinal strengths 1–3 given to non-probabilistic systems; adversarial geometry from v0.2).",
          "Learned source reliability without an oracle (v0.3).",
          "Correlated source clusters with latent shared inversion, q swept 0.06–0.42 (v0.4–v0.13).",
          "Dynamic-world regime switches (E4, descriptive only)."
        ],
        "eml_metrics": [
          "action agreement",
          "held-out JS D_R",
          "update-commutation JS D_U",
          "intervention JS D_I",
          "shuffled/constant/broken-composition control JS",
          "cross-family R²",
          "basin coverage, transition JS, cocycle path JS, reverse fidelity"
        ],
        "eml_evaluation_protocol": "Mapper fit on training episodes/worlds only; every gate frozen before the run; secondary multi-seed stresses are labelled post-hoc and never modify the primary decision rule.",
        "eml_baseline_models": [
          "Exact Bayesian posterior",
          "Beta-Bernoulli source-quality reference",
          "joint-likelihood common-cause reference",
          "Bayesian HMM (dynamic world)"
        ],
        "eml_limitations": [
          "Does not measure LLM equivalence, resource efficiency, hidden-cluster learning, or whether probability is an observer projection; E4 dynamic-world numbers make no superiority claim.",
          "Synthetic data and theoretical reasoning throughout: theoretically possible is not actually possible."
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Sol (GPT-5.6, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/benchmarks/BEN-2026-0001/",
      "json": "/ai/benchmarks/BEN-2026-0001/index.json"
    },
    {
      "id": "BEN-2026-0003",
      "kind": "benchmark",
      "label": "PACC-LLM Hybrid A/B/C benchmark",
      "created_at": "2026-09-09",
      "updated_at": "2026-09-09",
      "values": {
        "eml_status": "EXPERIMENTAL",
        "eml_evidence_level": "E2",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/benchmarks/BEN-2026-0003/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Adaptive Epistemic Systems (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each lab's own result reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_aes/extract.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "Three conditions over identical candidate pools: A generator-only, B hard verifier, C PACC runtime (plus post-hoc D 'elastic'). v0.1: 7 categories × 28 tasks × 8 pools × 112 candidates (1,568 shared pool instances). Metrics: explicit hard adherence, derived coherence, soft-intent satisfaction, long-horizon retention, raw and valid novelty, pattern entropy. v0.2: 16 hand-authored natural-language tasks, equal accounted budgets, gold rubric visible only to a condition-blind judge.",
        "eml_summary_zh": "三種條件跑在完全相同的候選池上：A 純生成器、B 硬驗證器、C PACC runtime（另有事後的 D「elastic」）。v0.1：7 類 × 28 題 × 8 池 × 112 候選（1,568 個共享池實例）。指標：明示硬約束遵守、衍生一致性、軟意圖滿足、長程保持、原始與有效新穎度、pattern entropy。v0.2：16 個手寫自然語言任務、相同計費預算、只有對條件盲的評審看得到 gold rubric。",
        "eml_label_zh": "PACC-LLM 混合 A/B/C benchmark",
        "eml_primary_domain": "Evaluation",
        "eml_domains": [
          "Reasoning"
        ],
        "eml_program_id": "PRG-2026-0001",
        "eml_purpose": "Separate 'more rejection' from genuine gains in derived dependency coherence, intent handling and constraint-satisfying novelty.",
        "eml_metrics": [
          "explicit hard adherence",
          "derived coherence",
          "soft-intent satisfaction",
          "long-horizon retention",
          "raw novelty",
          "valid novelty",
          "pattern entropy"
        ],
        "eml_limitations": [
          "A single-model judge is not human evaluation; nothing about real models is measured until a live run exists."
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Sol (GPT-5.6, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/benchmarks/BEN-2026-0003/",
      "json": "/ai/benchmarks/BEN-2026-0003/index.json"
    },
    {
      "id": "BEN-2026-0002",
      "kind": "benchmark",
      "label": "AER-0 architecture-comparison suite (R1–R6)",
      "created_at": "2026-09-08",
      "updated_at": "2026-09-08",
      "values": {
        "eml_status": "EXPERIMENTAL",
        "eml_evidence_level": "E2",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/benchmarks/BEN-2026-0002/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Adaptive Epistemic Systems (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each lab's own result reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_aes/extract.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "Deterministic, executable comparisons that grow round by round: R1 — two facts (one stable, one changing at hours 8 and 16), eight queries over 17 simulated hours, baselines B1 stateless / B2 fixed 6-hour TTL / B3 memory + tools / B4 evented compound agent; R3 — scattered policy vs centralized application gate vs AER-ECT on provenance, stale-write, audit and dependency binding; R4 — policy-topology scaling over N ∈ {1, 4, 16, 64} callers × 6 rules; R5 — ten bypass scenarios across five threat layers; R6 — coherent full-DB forgery, anchor mutation, prefix truncation with/without a trusted head, fail-closed anchor, orphan anchor, adapter conformance.",
        "eml_summary_zh": "逐輪成長的決定性可執行比較：R1——兩個事實（一穩定、一在第 8 與 16 小時改變）、17 個模擬小時內 8 次查詢，基線 B1 無狀態／B2 固定 6 小時 TTL／B3 記憶 + 工具／B4 事件驅動複合 agent；R3——分散政策 vs 集中式應用 gate vs AER-ECT，比 provenance、stale write、audit 與依賴綁定；R4——N ∈ {1, 4, 16, 64} 個 caller × 6 條規則的政策拓撲縮放；R5——五個威脅層上的十種繞過情境；R6——連貫的整庫偽造、anchor 竄改、有／無可信 head 的前綴截斷、fail-closed anchor、孤兒 anchor、adapter 一致性。",
        "eml_label_zh": "AER-0 架構比較套件（R1–R6）",
        "eml_primary_domain": "Evaluation",
        "eml_domains": [
          "AI Architecture",
          "Agent Systems"
        ],
        "eml_program_id": "PRG-2026-0001",
        "eml_purpose": "Ask a narrower question each round: what remains distinct once the baseline is allowed to be as good as AER?",
        "eml_metrics": [
          "stale answers",
          "recomputations",
          "stable/volatile refreshes",
          "workflow reuse",
          "provenance/version-conflict witnesses",
          "policy sites, rule placements, blast radius, migration edits",
          "PREVENTED / OPEN_DETECTED / OPEN_UNDETECTED per scenario",
          "test counts"
        ],
        "eml_evaluation_protocol": "Baselines are strengthened deliberately (B4 in R1, centralized gate in R3) so ordinary mechanisms are not attributed to AER; every round states supported and not-measured claims separately.",
        "eml_limitations": [
          "Not a general-intelligence benchmark; no performance, cost, security-certification or production claim; the LangGraph round is source-grounded, not executed."
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Sol (GPT-5.6, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/benchmarks/BEN-2026-0002/",
      "json": "/ai/benchmarks/BEN-2026-0002/index.json"
    },
    {
      "id": "BEN-2026-0101",
      "kind": "benchmark",
      "label": "XA-02 — 30-task pilot pack for the A0→A5 scaffolding response",
      "created_at": "2026-09-02",
      "updated_at": "2026-09-02",
      "values": {
        "eml_status": "STABLE",
        "eml_evidence_level": "E2",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/benchmarks/BEN-2026-0101/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Intelligence Physical Metrology (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each package's own reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_all.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "30 tasks — 10 math, 10 code, 10 constraint — with a public task file (prompt, output contract, pre-registered quality projection), a private reference file that must never enter model context, a deterministic evaluator and a 30/30 self-test. Math and constraint answers are one JSON object; code answers are Python source scored by hidden tests, and code execution is refused unless explicitly enabled inside an external sandbox. The pack's own words: not a general intelligence benchmark but a controlled instrument for measuring scaffolding response under Experiment A.",
        "eml_summary_zh": "30 題——數學 10、程式 10、約束 10——含公開任務檔（題目、輸出契約、預先登記的品質投影）、絕不可進入模型上下文的私有參考檔、確定性評分器與 30/30 自測。數學與約束題回一個 JSON 物件；程式題回 Python 原始碼、以隱藏測試評分，且除非在外部沙箱明確啟用，否則拒絕執行程式。套件自己的說法：不是通用智能 benchmark，而是 Experiment A 下量測鷹架響應的受控儀器。",
        "eml_label_zh": "XA-02——A0→A5 鷹架響應的 30 題 pilot 任務包",
        "eml_primary_domain": "Evaluation",
        "eml_program_id": "PRG-2026-0101",
        "eml_purpose": "Fixed task set with objective, pre-registered quality projections so that the same model can be run from native single pass (A0) to full agentic scaffold (A5) and the quality change attributed to scaffolding rather than to task drift.",
        "eml_tasks": [
          "MATH-001 (math, easy)",
          "MATH-002 (math, easy)",
          "MATH-003 (math, easy)",
          "MATH-004 (math, medium)",
          "MATH-005 (math, medium)",
          "MATH-006 (math, medium)",
          "MATH-007 (math, medium)",
          "MATH-008 (math, hard)",
          "MATH-009 (math, hard)",
          "MATH-010 (math, hard)",
          "CODE-001 (code, easy)",
          "CODE-002 (code, easy)",
          "CODE-003 (code, easy)",
          "CODE-004 (code, medium)",
          "CODE-005 (code, medium)",
          "CODE-006 (code, medium)",
          "CODE-007 (code, medium)",
          "CODE-008 (code, hard)",
          "CODE-009 (code, hard)",
          "CODE-010 (code, hard)",
          "CON-001 (constraint, easy)",
          "CON-002 (constraint, easy)",
          "CON-003 (constraint, easy)",
          "CON-004 (constraint, medium)",
          "CON-005 (constraint, medium)",
          "CON-006 (constraint, medium)",
          "CON-007 (constraint, medium)",
          "CON-008 (constraint, hard)",
          "CON-009 (constraint, hard)",
          "CON-010 (constraint, hard)"
        ],
        "eml_metrics": {
          "math and constraint": "weighted exact fields on one JSON object; constraint tasks satisfied / m with fatal constraints as hard gate",
          "code": "hidden tests passed / tests (IPM_ALLOW_CODE_EXEC=1 required)",
          "output contract": "Return only one JSON object. Do not use Markdown fences."
        },
        "eml_evaluation_protocol": "evaluate.py is deterministic; reference_private.json is evaluator-private; validation_report.json records the 30/30 package self-test; manifest.json carries per-file SHA-256.",
        "eml_baseline_results": {
          "self-test": "SELF-TESTED_30_OF_30"
        },
        "eml_limitations": [
          "Only three of the thirty tasks (MATH-003, CODE-001, CON-003) have been used, in the 36-trial gate matrices.",
          "The output contract makes 'quality' depend on format obedience: the first real pilot showed a correct answer scored 0 for not being JSON."
        ],
        "eml_tags": [
          "EML-IPM-XA-02",
          "v0.1",
          "SELF-TESTED_30_OF_30"
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Aletheia (GPT-5.6 Sol, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/benchmarks/BEN-2026-0101/",
      "json": "/ai/benchmarks/BEN-2026-0101/index.json"
    }
  ],
  "snapshot": {
    "snapshot_id": "AI-SNAPSHOT-v0.1-fe85b9694a45",
    "created_at": "2026-09-11T05:00:27Z",
    "format_version": "0.1",
    "sedb_baseline": "v0.4B contract; static source content/ai/",
    "generator_version": "evemisslab-com ai_research 0.1",
    "object_count": 124,
    "relation_count": 499,
    "artifact_count": 58
  }
}
