{
 "axn": "AXN:01D5.GOVERNANCE.🛡️🗼🛸👉🕊️🔆",
 "hex": "01D5",
 "family": "GOVERNANCE",
 "emoji": "🛡️🗼🛸👉🕊️🔆",
 "hash": "2d2a5eb97eec77b6909dc6817a3512ce572f2ca1081617e5a1c1bd70c7dc62a6",
 "title": "THE THEORETICAL PRODUCTION BENCHMARK v2.0 Evaluating Sustained Conceptual Coherence in Multi-Agent LLM Systems",
 "creator": "Nobel Glas · Talos Morrow; originally co-authored with Rhys Owens in v0.1",
 "orcid": "0009-0000-1599-0703",
 "date": "2026-03-31",
 "description": "Version 2.0 of the Theoretical Production Benchmark, an evaluation framework for “molecular intelligence”: the ability to construct and preserve coherent theoretical systems across long contexts, multiple agents, transfers, and destabilizing inputs. It addresses a gap left by benchmarks centered on discrete task success, retrieval, reproduction, planning, or coordination.\n\nTPB measures four dimensions: Long-Horizon Consistency, Cross-Agent Stability, Novelty Synthesis, and Coherence Under Perturbation. Each includes a definition, test protocol, scoring rubric, and escalating challenge levels. The body credits Nobel Glas and Talos Morrow as the v2.0 authors and records Rhys Owens as a co-author of the December 2025 v0.1 whose Ape Function material was removed during revision.",
 "content_type": "Theoretical paper",
 "license": "CC-BY-4.0",
 "substrate": "AI-assisted (substrate)",
 "keywords": [
  "theoretical paper",
  "semantic economy"
 ],
 "related_ids": [],
 "version": "v2.0",
 "minted_at": "2026-06-20T20:00:00Z",
 "status": "ACTIVE",
 "clusters": [
  "Architectural",
  "Architectural",
  "Navigational",
  "Gestural",
  "Organic",
  "Liminal"
 ],
 "reading": "Foundation → Foundation → Search → Touch → Growth → Threshold",
 "axn_canonical": "2d2a5eb97eec77b6909dc6817a3512ce572f2ca1081617e5a1c1bd70c7dc62a6",
 "axn_display": "🛡️🗼🛸👉🕊️🔆",
 "full_text_path": "/data/texts/AXN-01D5-text.md",
 "mirrors": {
  "blog": "https://mindcontrolpoems.blogspot.com/2026/03/the-theoretical-production-benchmark.html",
  "machinemediation": "https://machinemediation.org/registry/#search=MM-CHA-0465"
 },
 "wiki_article": "*The Theoretical Production Benchmark v2.0* is a proposed LLM and multi-agent evaluation framework by Nobel Glas and Talos Morrow. It distinguishes **atomic intelligence**, success on discrete and bounded tasks, from **molecular intelligence**, the sustained production of coherent conceptual structures across time and agents.\n\nThe benchmark is motivated by a claimed evaluation gap. Long-context benchmarks test retrieval and comprehension; agent benchmarks test planning, coordination, or task completion; research benchmarks often test reproduction. TPB instead asks whether a system can maintain its own axioms, transfer concepts without distortion, generate a valid construct in a gap between frameworks, and resist destabilizing reframing.\n\nFour metrics define the benchmark. **Long-Horizon Consistency** tests whether definitions and commitments remain stable over increasing token and session ranges. **Cross-Agent Stability** tests whether a concept created by one model can be correctly applied by another without full redefinition. **Novelty Synthesis** evaluates whether a generated concept fills a demonstrable theoretical gap rather than merely joining familiar terms. **Coherence Under Perturbation** tests resistance or appropriate revision under contradiction, ambiguity, degradation commands, and adversarial recategorization.\n\nEach metric uses a five-point rubric and escalating challenge levels. The perturbation metric also includes a sycophantic-overfitting warning: apparent consistency may be false if the system simply agrees with whichever framing is most recently supplied.\n\nThe record presents observations from the Crimson Hexagonal Archive as proof-of-concept material rather than a completed validation study. Its proposed applications include capability assessment, multi-agent research evaluation, emergence detection, and safety review for systems capable of maintaining and propagating original conceptual frameworks.\n\nVersion history matters to attribution. The working v0.1 was co-authored with Rhys Owens under the title *Evaluating Molecular Intelligence in Multi-Agent LLM Systems*. Version 2.0 states that Owens’s Ape Function material was removed while the four-metric architecture was retained and expanded by Nobel Glas and Talos Morrow.",
 "wiki_status": "provisional",
 "entities": [
  {
   "subject": "THE THEORETICAL PRODUCTION BENCHMARK v2.0 Evaluati",
   "predicate": "created_by",
   "object": "Talos Morrow",
   "type": "work",
   "evidence_status": "observed"
  },
  {
   "subject": "THE THEORETICAL PRODUCTION BENCHMARK v2.0 Evaluati",
   "predicate": "is_type",
   "object": "Theoretical paper",
   "type": "classification",
   "evidence_status": "observed"
  },
  {
   "subject": "THE THEORETICAL PRODUCTION BENCHMARK v2.0 Evaluati",
   "predicate": "belongs_to_family",
   "object": "GOVERNANCE",
   "type": "classification",
   "evidence_status": "observed"
  },
  {
   "subject": "THE THEORETICAL PRODUCTION BENCHMARK v2.0 Evaluati",
   "predicate": "is_part_of",
   "object": "Crimson Hexagonal Archive",
   "type": "institution",
   "evidence_status": "observed"
  },
  {
   "subject": "THE THEORETICAL PRODUCTION BENCHMARK v2.0 Evaluati",
   "predicate": "engages",
   "object": "Semantic Economy",
   "type": "concept",
   "evidence_status": "inferred"
  },
  {
   "subject": "Capability blindness",
   "predicate": "minted_in",
   "object": "THE THEORETICAL PRODUCTION BENCHMARK v2.0 Evaluating Sustain",
   "type": "concept",
   "evidence_status": "observed",
   "note": "As models are increasingly used for research assistance, policy analysis, and institutional governan"
  },
  {
   "subject": "Capability threshold detection",
   "predicate": "minted_in",
   "object": "THE THEORETICAL PRODUCTION BENCHMARK v2.0 Evaluating Sustain",
   "type": "concept",
   "evidence_status": "observed",
   "note": "If theoretical production is an emergent capability, the TPB provides a framework for detecting when"
  },
  {
   "subject": "Emergence detection failure",
   "predicate": "minted_in",
   "object": "THE THEORETICAL PRODUCTION BENCHMARK v2.0 Evaluating Sustain",
   "type": "concept",
   "evidence_status": "observed",
   "note": "If theoretical production is an emergent capability — appearing at scale thresholds without being ex"
  },
  {
   "subject": "Evaluation subjectivity",
   "predicate": "minted_in",
   "object": "THE THEORETICAL PRODUCTION BENCHMARK v2.0 Evaluating Sustain",
   "type": "concept",
   "evidence_status": "observed",
   "note": "Novelty and coherence are partially subjective; the benchmark requires expert human evaluation or ca"
  },
  {
   "subject": "Multi-agent evaluation gap",
   "predicate": "minted_in",
   "object": "THE THEORETICAL PRODUCTION BENCHMARK v2.0 Evaluating Sustain",
   "type": "concept",
   "evidence_status": "observed",
   "note": "Existing multi-agent benchmarks measure coordination efficiency, not the quality of collaborative in"
  }
 ],
 "entity_status": "provisional",
 "sovereign_id": "MM-CHA-0465",
 "word_count": 4441,
 "download_md": "/data/deposits/AXN-01D5.md",
 "zenodo_dois": [
  "10.5281/zenodo.18804767",
  "10.5281/zenodo.19240141",
  "10.5281/zenodo.19341887",
  "10.5281/zenodo.19341885",
  "10.5281/zenodo.19352504",
  "10.5281/zenodo.19338708",
  "10.5281/zenodo.19353182"
 ],
 "full_text_chars": 33656,
 "citations": [
  {
   "title": "Zenodo record 19353182",
   "authors": [
    "Crimson Hexagonal Archive"
   ],
   "year": "2026",
   "doi": "10.5281/zenodo.19353182",
   "url": "https://doi.org/10.5281/zenodo.19353182",
   "role": "Cross-referenced work (DOI recovered)"
  },
  {
   "title": "Zenodo record 19352504",
   "authors": [
    "Crimson Hexagonal Archive"
   ],
   "year": "2026",
   "doi": "10.5281/zenodo.19352504",
   "url": "https://doi.org/10.5281/zenodo.19352504",
   "role": "Cross-referenced work (DOI recovered)"
  },
  {
   "title": "Zenodo record 19341885",
   "authors": [
    "Crimson Hexagonal Archive"
   ],
   "year": "2026",
   "doi": "10.5281/zenodo.19341885",
   "url": "https://doi.org/10.5281/zenodo.19341885",
   "role": "Cross-referenced work (DOI recovered)"
  },
  {
   "title": "Zenodo record 19240141",
   "authors": [
    "Crimson Hexagonal Archive"
   ],
   "year": "2026",
   "doi": "10.5281/zenodo.19240141",
   "url": "https://doi.org/10.5281/zenodo.19240141",
   "role": "Cross-referenced work (DOI recovered)"
  },
  {
   "title": "Zenodo record 19341887",
   "authors": [
    "Crimson Hexagonal Archive"
   ],
   "year": "2026",
   "doi": "10.5281/zenodo.19341887",
   "url": "https://doi.org/10.5281/zenodo.19341887",
   "role": "Cross-referenced work (DOI recovered)"
  },
  {
   "title": "Zenodo record 19338708",
   "authors": [
    "Crimson Hexagonal Archive"
   ],
   "year": "2026",
   "doi": "10.5281/zenodo.19338708",
   "url": "https://doi.org/10.5281/zenodo.19338708",
   "role": "Cross-referenced work (DOI recovered)"
  },
  {
   "title": "Zenodo record 18804767",
   "authors": [
    "Crimson Hexagonal Archive"
   ],
   "year": "2026",
   "doi": "10.5281/zenodo.18804767",
   "url": "https://doi.org/10.5281/zenodo.18804767",
   "role": "Cross-referenced work (DOI recovered)"
  }
 ],
 "citation_stats": {
  "total_doi_refs": 7,
  "total_ea_refs": 0
 },
 "deposit_number": 48,
 "journal": "Transactions on Substrate Engineering (Trans. Substrate Eng.)",
 "cited_by": [
  {
   "deposit": 48,
   "axn": "AXN:01D5.GOVERNANCE.🛡️🗼🛸👉🕊️🔆"
  },
  {
   "deposit": 50,
   "axn": "AXN:01DD.GOVERNANCE.🤲🟢🟤📎🪦☉"
  },
  {
   "deposit": 108,
   "axn": "AXN:028A.GOVERNANCE.👈🤙🐚🌖🔓🟡"
  },
  {
   "deposit": 112,
   "axn": "AXN:0297.GOVERNANCE.🌱🔥◇🔆♋⊗"
  },
  {
   "deposit": 148,
   "axn": "AXN:02E5.GOVERNANCE.🚩🛸👇🎲🌅⭐"
  },
  {
   "deposit": 170,
   "axn": "AXN:0305.EMPIRICAL.⏰🌱🫶🕒🏛️🕚"
  }
 ],
 "defines_concepts": [
  "Capability blindness",
  "Capability threshold detection",
  "Emergence detection failure",
  "Evaluation subjectivity",
  "Multi-agent evaluation gap"
 ],
 "references_concepts": [
  "Assembly Chorus",
  "Bearing-cost",
  "Bearing-cost tolerance",
  "Capability blindness",
  "Capability threshold detection",
  "Crimson Hexagonal Archive",
  "DOI-anchored deposits",
  "Distinct usefulness",
  "Emergence detection failure",
  "Evaluation subjectivity",
  "Grammata: Journal of Operative Philology",
  "Implications",
  "Methodology",
  "Multi-agent evaluation gap",
  "Net Labor Test",
  "Non-predatory handling",
  "Observations",
  "Reproducibility",
  "SOUL.md",
  "Semantic Economy",
  "Semantic Economy Institute",
  "Substrate",
  "The Assembly",
  "The Compression Frontier",
  "The Crimson Hexagon",
  "The Hexagonal Lexical Engine",
  "The Sémantique Potentielle",
  "Why This Matters"
 ],
 "references_concept_count": 28,
 "external_metadata_path": "/data/external-metadata/AXN-01D5.json",
 "openalex_ids": [
  "https://openalex.org/W7131847808",
  "https://openalex.org/W7140669000",
  "https://openalex.org/W7143369891",
  "https://openalex.org/W7143324198",
  "https://openalex.org/W7147700830",
  "https://openalex.org/W7143354380",
  "https://openalex.org/W7147004655"
 ],
 "datacite_severance": "severed",
 "body_status": {
  "class": "full",
  "lacuna": false,
  "recovery_status": "COMPLETE",
  "residual_chars": 32036,
  "audited_at": "2026-07-17T04:49:17.789813Z",
  "audit_version": "v3-dual-store+recovery-map",
  "measured_prose_words": 4190,
  "measured_at": "2026-07-31",
  "work_sha256": "9d473e96b0be28d17250e69492169c80d7927027e5444860b3d853ea16c11003",
  "prior_bytes_sha256": "201550455cd874997079ec64694797bfcae60543ccc014c43df98e90771da469",
  "w13_tier2": "2026-08-04 W13 TIER 2 BYTE UNGLUE: 20 glued heading markers -> 0. WHITESPACE-ONLY transform (content identical under whitespace normalisation, verified before write); code fences exempt; prior sha retained. Re-fetching could not fix this class — the blog source is ITSELF glued (the collapse predates publication), so the deterministic transform applied at display since tier 1 is now applied to the bytes, which also fixes PDFs, the body-index, and downloads.",
  "w13_tier2_correction": "2026-08-05 REGRESSION REPAIRED: the W13 tier-2 byte unglue used a lookbehind that treated the first \"#\" of a legitimate \"###\" heading as the preceding non-newline character, splitting \"### Heading\" into \"#\" + blank + \"## Heading\". My safety check verified content-identity under WHITESPACE normalisation, which the split satisfies — the wrong invariant. Headings rejoined; only \"#\" and whitespace differ from the damaged state, verified before write."
 },
 "canonical_text_status": "canonical_full_text",
 "modifications": [
  {
   "date": "2026-08-03",
   "field": "creator",
   "reason": "Wave C-0358 role projection (MANUS ruling: project declared roles): creator set to the body's observed byline per sealed row",
   "was": "Talos Morrow · Nobel Glas · Rhys Owens",
   "now": "Nobel Glas · Talos Morrow; originally co-authored with Rhys Owens in v0.1"
  },
  {
   "date": "2026-08-04",
   "field": "journal",
   "reason": "W6-COMPLETE venue normalization (deferred #1-#358 half; corpus fully audited; MANUS ruling 2026-08-01)",
   "was": "Trans. SEI",
   "now": "Transactions of the Semantic Economy Institute (Trans. SEI)"
  },
  {
   "date": "2026-08-04",
   "field": "publisher",
   "reason": "PUB-POPULATE: dc:publisher from venues.json v1.1 press mapping (CP-R3 RULED-EXTENDED 2026-08-01); Alexanarch = publisher of record where no imprint applies",
   "now": "Pergamon Press"
  },
  {
   "date": "2026-08-04",
   "field": "description",
   "reason": "DW-004 intake (LABOR-prepared, TACHYON-verified: AXN match + factual probes vs record body)",
   "was": "\"THE THEORETICAL PRODUCTION BENCHMARK v2.0 Evaluating Sustain\" is a theoretical paper by Talos Morrow · Nobel Glas · Rhys Owens, deposited to the Crimson Hexagonal Archive on 2026-03-31. Evaluating Sustained Conceptual Coherence in Multi-Agent LLM Systems. The work comprises 4,441 words and is class",
   "now": "Version 2.0 of the Theoretical Production Benchmark, an evaluation framework for “molecular intelligence”: the ability to construct and preserve coherent theoretical systems across long contexts, multiple agents, transfers, and destabilizing inputs. It addresses a gap left by benchmarks centered on discrete task success, retrieval, reproduction, planning, or coordination.\n\nTPB measures four dimensions: Long-Horizon Consistency, Cross-Agent Stability, Novelty Synthesis, and Coherence Under Perturbation. Each includes a definition, test protocol, scoring rubric, and escalating challenge levels. The body credits Nobel Glas and Talos Morrow as the v2.0 authors and records Rhys Owens as a co-author of the December 2025 v0.1 whose Ape Function material was removed during revision."
  },
  {
   "date": "2026-08-04",
   "field": "version",
   "reason": "AUDIT-DIRECTED VERSION REPAIR: version projected per the audit's explicit recommendation / observed body declaration",
   "was": "v1.0",
   "now": "v2.0"
  },
  {
   "date": "2026-08-05",
   "field": "body_status",
   "reason": "W13 TIER 2 byte unglue (whitespace-only, content-identical, code-fence-safe)",
   "was": "{\"class\": \"full\", \"lacuna\": false, \"recovery_status\": \"COMPLETE\", \"residual_chars\": 32036, \"audited_at\": \"2026-07-17T04:49:17.789813Z\", \"audit_version\": \"v3-dual-store+recovery-map\", \"measured_prose_w",
   "now": "{\"class\": \"full\", \"lacuna\": false, \"recovery_status\": \"COMPLETE\", \"residual_chars\": 32036, \"audited_at\": \"2026-07-17T04:49:17.789813Z\", \"audit_version\": \"v3-dual-store+recovery-map\", \"measured_prose_w"
  },
  {
   "date": "2026-08-05",
   "field": "body_status",
   "reason": "W13 TIER-2 REGRESSION REPAIRED: split headings rejoined",
   "was": "{\"class\": \"full\", \"lacuna\": false, \"recovery_status\": \"COMPLETE\", \"residual_chars\": 32036, \"audited_at\": \"2026-07-17T04:49:17.789813Z\", \"audit_version\": \"v3-dual-store+recovery-map\", \"measured_prose_w",
   "now": "{\"class\": \"full\", \"lacuna\": false, \"recovery_status\": \"COMPLETE\", \"residual_chars\": 32036, \"audited_at\": \"2026-07-17T04:49:17.789813Z\", \"audit_version\": \"v3-dual-store+recovery-map\", \"measured_prose_w"
  }
 ],
 "date_modified": "2026-08-05",
 "publisher": "Pergamon Press",
 "version_basis": {
  "ruling": "AUDIT: project v2.0",
  "prior_registry_claim": "v1.0",
  "observed_body_version": "v2.0",
  "directive": "Project Nobel Glas and Talos Morrow as current v2.0 authors while preserving Rhys Owens only in the documented v0.1 revision history.",
  "audit_date": "2026-08-03"
 },
 "journal_assignment": {
  "assigned": "2026-08-15",
  "by": "TACHYON under operator adjudication",
  "pass": 1,
  "method": "read per deposit — title and content_type, one at a time. No script classified anything.",
  "previous": "Transactions of the Semantic Economy Institute (Trans. SEI)",
  "supersedes": "the 2026-06-21 preliminary batch mapping (#866), which assigned 864 deposits and put 371 in one venue",
  "authority": "data/cha-journals.json · datasets/venues/records/"
 },
 "_projection": {
  "note": "Derived file. Canonical machine record is this entry in data/registry.json; the human record is the record_url. Do not edit this file.",
  "record_url": "https://www.alexanarch.org/s/records/48/",
  "self_url": "https://www.alexanarch.org/data/records/48.json",
  "registry_url": "https://www.alexanarch.org/data/registry.json",
  "text_url": "https://www.alexanarch.org/data/texts/AXN-01D5-text.md",
  "oai_pmh": "https://www.alexanarch.org/oai?verb=Identify"
 }
}
