{
  "section": "claims",
  "kind": "claim",
  "canonical": "https://evemisslab.com/ai/claims/",
  "count": 5,
  "records": [
    {
      "id": "CLM-2026-0101",
      "kind": "claim",
      "label": "F1 — Token hypothesis",
      "created_at": "2026-09-02",
      "updated_at": "2026-09-02",
      "values": {
        "eml_status": "PRELIMINARY",
        "eml_evidence_level": "E0",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/claims/CLM-2026-0101/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Intelligence Physical Metrology (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each package's own reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_all.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "If the token is a good universal unit of intelligent work, then N_μ^eff / TokenCount should be approximately stable across models, languages and phrasings once task quality is controlled. Large drift across models, modalities or expression forms weakens 'token = intelligent work unit' to an implementation/interface proxy.",
        "eml_summary_zh": "若 token 是良好的普適智能工作單位，則在控制任務品質後，N_μ^eff / TokenCount 應跨模型、語言與措辭大致穩定。若跨模型、模態或表達形式劇烈漂移，「token = 智能工作單位」就退化為實作／介面的代理量。",
        "eml_label_zh": "F1——Token 假說",
        "eml_primary_domain": "Evaluation",
        "eml_program_id": "PRG-2026-0101",
        "eml_data_basis": "THEORY",
        "eml_falsification_conditions": [
          "Falsified for token-as-unit if N_μ^eff per token drifts by large factors across models, languages or phrasings at matched task quality; supported if the ratio is stable."
        ],
        "eml_tags": [
          "falsifiable proposition",
          "IPM v0.1 canonical index §23",
          "Paper 10"
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Aletheia (GPT-5.6 Sol, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/claims/CLM-2026-0101/",
      "json": "/ai/claims/CLM-2026-0101/index.json"
    },
    {
      "id": "CLM-2026-0102",
      "kind": "claim",
      "label": "F2 — FLOPs sufficiency",
      "created_at": "2026-09-02",
      "updated_at": "2026-09-02",
      "values": {
        "eml_status": "PRELIMINARY",
        "eml_evidence_level": "E0",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/claims/CLM-2026-0102/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Intelligence Physical Metrology (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each package's own reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_all.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "If FLOPs suffice to describe physical computational cost, then with operations controlled, wall time, energy, memory traffic, interconnect traffic and memory residency should not show large independent variation. If same-FLOPs workloads differ greatly because of memory patterns or topology, a single FLOPs cost model is insufficient.",
        "eml_summary_zh": "若 FLOPs 足以描述物理計算成本，則控制運算量後，wall time、能量、記憶體流量、互連流量與記憶體駐留不應出現大幅獨立變化。若相同 FLOPs 的工作因記憶體模式或拓撲而差異巨大，單一 FLOPs 成本模型就不足。",
        "eml_label_zh": "F2——FLOPs 充分性",
        "eml_primary_domain": "Evaluation",
        "eml_program_id": "PRG-2026-0101",
        "eml_data_basis": "THEORY",
        "eml_falsification_conditions": [
          "Falsified for FLOPs-sufficiency if T, E, B_M, B_N, V_M vary substantially at fixed FLOPs; supported if they do not."
        ],
        "eml_tags": [
          "falsifiable proposition",
          "IPM v0.1 canonical index §23",
          "Paper 10"
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Aletheia (GPT-5.6 Sol, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/claims/CLM-2026-0102/",
      "json": "/ai/claims/CLM-2026-0102/index.json"
    },
    {
      "id": "CLM-2026-0103",
      "kind": "claim",
      "label": "F3 — Binary burden hypothesis",
      "created_at": "2026-09-02",
      "updated_at": "2026-09-02",
      "values": {
        "eml_status": "PRELIMINARY",
        "eml_evidence_level": "E0",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/claims/CLM-2026-0103/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Intelligence Physical Metrology (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each package's own reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_all.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "If direct numeric human rating were already the minimum-burden, high-quality measurement interface, then well-designed binary / pairwise adaptive protocols should have no advantage in response time, consistency, dropout, predictive validity or fatigue. BRQM predicts an advantage in at least some settings; it can be tested by randomizing participants across direct 0–10, structured yes/no and adaptive pairwise formats.",
        "eml_summary_zh": "若直接數值評分已是最低負擔、高品質的測量介面，則設計良好的二元／成對自適應協定在反應時間、一致性、流失率、預測效度或疲勞上應不具優勢。BRQM 預測至少在部分場景有優勢；可把受試者隨機分配到直接 0–10、結構化是／否與自適應成對三種格式來檢驗。",
        "eml_label_zh": "F3——二元負擔假說",
        "eml_primary_domain": "Evaluation",
        "eml_program_id": "PRG-2026-0101",
        "eml_data_basis": "THEORY",
        "eml_falsification_conditions": [
          "Falsified if binary/pairwise protocols are worse than direct rating on all of response time, consistency, dropout and predictive validity; supported if they win on some."
        ],
        "eml_tags": [
          "falsifiable proposition",
          "IPM v0.1 canonical index §23",
          "Paper 10"
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Aletheia (GPT-5.6 Sol, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/claims/CLM-2026-0103/",
      "json": "/ai/claims/CLM-2026-0103/index.json"
    },
    {
      "id": "CLM-2026-0104",
      "kind": "claim",
      "label": "F4 — Scaffolding separation",
      "created_at": "2026-09-02",
      "updated_at": "2026-09-02",
      "values": {
        "eml_status": "EXPERIMENTAL",
        "eml_evidence_level": "E2",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/claims/CLM-2026-0104/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Intelligence Physical Metrology (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each package's own reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_all.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "If scaffolding does not change the structure of the capability source, SSR ≈ 1 should hold across most tasks and test-time compute budgets. If a large share of benchmark quality appears only under multi-sample, tool, verifier or loop conditions, the distinction between model-native and system capability is empirically necessary. First data point: three easy tasks on a local 9B model gave SSR = 1.0 — consistent with 'no gap' on tasks the native pass already solves, and uninformative beyond that because the quality axis was confounded by format compliance.",
        "eml_summary_zh": "若鷹架不改變能力來源的結構，則 SSR ≈ 1 應在多數任務與 test-time 算力預算下成立。若大量 benchmark 品質只在多樣本、工具、驗證器或迴圈條件下出現，模型原生與系統能力的區分就具有實證必要性。第一個數據點：本地 9B 模型在三個簡單任務上 SSR = 1.0——與「原生單次已解的任務沒有落差」一致，此外沒有更多訊息，因為品質軸被格式服從性混淆。",
        "eml_label_zh": "F4——鷹架分離",
        "eml_primary_domain": "Evaluation",
        "eml_program_id": "PRG-2026-0101",
        "eml_data_basis": "THEORY",
        "eml_falsification_conditions": [
          "Falsified for the separation's necessity if Q_F ≈ Q_SP holds across most tasks, models and budgets; supported if the gap is common. The 2026-09-03 pilot is one uninformative-to-weak point on the 'no gap' side."
        ],
        "eml_tags": [
          "falsifiable proposition",
          "IPM v0.1 canonical index §23",
          "Paper 10"
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Aletheia (GPT-5.6 Sol, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/claims/CLM-2026-0104/",
      "json": "/ai/claims/CLM-2026-0104/index.json"
    },
    {
      "id": "CLM-2026-0105",
      "kind": "claim",
      "label": "F5 — Semantic intermediate utility",
      "created_at": "2026-09-02",
      "updated_at": "2026-09-02",
      "values": {
        "eml_status": "PRELIMINARY",
        "eml_evidence_level": "E0",
        "eml_object_version": "0.1",
        "eml_canonical_url": "https://evemisslab.com/ai/claims/CLM-2026-0105/",
        "eml_provenance": {
          "source": "EveMissLab research collection: Intelligence Physical Metrology (真本體論13)",
          "extracted_by": "Splice (Claude Code), reading the canonical UTF-8 sources and each package's own reports",
          "extracted_at": "2026-09-11",
          "generator": "tools/extract_all.py",
          "claim_boundary": "status, evidence level and result type follow the source artifact's own stated claim boundary; nothing is upgraded beyond what the report supports"
        },
        "eml_summary": "If the semantic middle layer N_μ has no measurement value, then adding it should not improve the explanation or prediction of cross-architecture efficiency, error paths, scaffold gain or task transfer over a direct physical-cost → quality model. μI itself must pass this predictive/explanatory utility test or be revised or eliminated.",
        "eml_summary_zh": "若語意中間層 N_μ 沒有測量價值，那麼加入它不應比直接的「物理成本 → 品質」模型更能解釋或預測跨架構效率、錯誤路徑、鷹架增益或任務遷移。μI 本身必須通過這個預測／解釋效用檢驗，否則就該被修正或淘汰。",
        "eml_label_zh": "F5——語意中間層效用",
        "eml_primary_domain": "Evaluation",
        "eml_program_id": "PRG-2026-0101",
        "eml_data_basis": "THEORY",
        "eml_falsification_conditions": [
          "Falsified if N_μ adds no explanatory or predictive power over physical-cost → quality models; supported if it does."
        ],
        "eml_tags": [
          "falsifiable proposition",
          "IPM v0.1 canonical index §23",
          "Paper 10"
        ],
        "eml_authors": [
          "Neo.K (EveMissLab)"
        ],
        "eml_ai_collaborators": [
          "Aletheia (GPT-5.6 Sol, OpenAI ChatGPT)"
        ]
      },
      "canonical_url": "https://evemisslab.com/ai/claims/CLM-2026-0105/",
      "json": "/ai/claims/CLM-2026-0105/index.json"
    }
  ],
  "snapshot": {
    "snapshot_id": "AI-SNAPSHOT-v0.1-fe85b9694a45",
    "created_at": "2026-09-11T05:00:27Z",
    "format_version": "0.1",
    "sedb_baseline": "v0.4B contract; static source content/ai/",
    "generator_version": "evemisslab-com ai_research 0.1",
    "object_count": 124,
    "relation_count": 499,
    "artifact_count": 58
  }
}
