{
  "study_id": "HNI-2026-001",
  "protocol_sha256": "32b237b6e05c6a597fd2358430684131ee8ca01e0f66a627daa0e477005f3e43",
  "completed_at": "2026-07-09T20:29:00+00:00",
  "outcome": "supports",
  "summary": "Across three bounded local task types, reducing the exposed tool surface from the broad default to file and terminal while preserving rules reduced physical tokens in every task and preserved all locked quality checks. Skipping rules alone was approximately neutral at the median and was not uniformly beneficial.",
  "primary_results": [
    {
      "name": "median_tool_scoping_reduction_percent",
      "value": 77.916,
      "unit": "percent",
      "interpretation": "The median within-task reduction from broad tools plus rules to scoped tools plus rules was 77.916%. Per-task reductions were 77.916%, 78.740%, and 20.787%."
    },
    {
      "name": "locked_quality_pass_rate",
      "value": 1.0,
      "unit": "proportion",
      "interpretation": "All 12 task-by-harness cells passed their locked task-specific validators and independent archived-output revalidation."
    }
  ],
  "secondary_results": [
    {
      "name": "median_skip_rules_alone_reduction_percent",
      "value": 0.014,
      "unit": "percent",
      "interpretation": "Skipping rules/context with broad tools was approximately neutral at the median and increased tokens on the asset-manifest task."
    },
    {
      "name": "combined_scoped_tools_and_skipped_rules_reduction_percent",
      "value": 78.101,
      "unit": "percent",
      "interpretation": "The combined arm was efficient, but the factorial contrasts attribute the dominant measured effect to tool scoping rather than rules removal."
    }
  ],
  "all_primary_outcomes_reported": true,
  "deviations": [
    "The HNI protocol wrapper and public research-note structure were created after the experiment completed. The underlying FACTORIAL_DESIGN.json was fixed before execution, but this study must not be described as publicly preregistered.",
    "The experiment used one run per task-by-harness cell, so the analysis is descriptive and cannot estimate stable population effects.",
    "Terminology was clarified after editorial review: locked automated task validators are internal evidence checks, not independent external review. The original locked protocol remains unchanged and the clarification is recorded in protocol.amendments.json."
  ],
  "limitations": [
    "Only three bounded local task types were tested.",
    "There was one run per factorial cell and no inferentially powered sample.",
    "One model and one provider were used.",
    "The tasks were achievable with local file and terminal operations; results do not generalize to web, browser, email, delegation, or other tool classes.",
    "Physical token reduction is not the same as marginal dollar savings under a subscription-included provider.",
    "Hypernovelty Institute develops and benefits from the tested workflow, creating a reputational conflict that readers should consider.",
    "With one run per cell, within-cell variance is unobservable and no confidence interval can be computed from these data.",
    "The automated quality validators were implemented with AI assistance and locked before execution. They were not an external replication, external audit, or peer review."
  ],
  "ai_use_actual": {
    "disclosed_roles": [
      "experimental design assistance",
      "Python implementation",
      "agent task execution",
      "receipt extraction",
      "task-quality-check implementation",
      "statistical summarization",
      "report drafting"
    ],
    "human_accountability": "Jordan Finneseth approved the research direction and public communication."
  },
  "disclosure_actual": {
    "funding": "Self-funded. Inference was available through an existing subscription. No external research grant supported this study.",
    "conflicts": "Hypernovelty Institute develops and uses the Hermes research harness and benefits reputationally from useful results. No Writer, Nous Research, or OpenAI funding supported the study."
  },
  "publication": {
    "publish_regardless_of_direction": true,
    "public_path": "/research/notes/hni-2026-001/",
    "publication_class": "Experiment Note",
    "peer_review_status": "not peer reviewed",
    "repository_status": "local candidate; DOI not yet minted"
  },
  "verification": {
    "status": "pass",
    "level": "internal_locked_automated",
    "external_party": false,
    "implementation_disclosure": "Task-quality validators were implemented with AI assistance and locked before runs. Evidence hashes were rechecked after execution. This is internal automated verification, not external review.",
    "evidence_manifest": [
      {
        "name": "FACTORIAL_DESIGN.json",
        "sha256": "63674a98e148edf2231cb2a6e38bda0f047f3a68b9b1b5df56ffb6f5e5db0544"
      },
      {
        "name": "FACTORIAL_SUMMARY.json",
        "sha256": "8dc11a30fb5f2f5c7ae5c09fed07f82dfe216666f7c98ca35efc7c285ef5e7d2"
      },
      {
        "name": "FACTORIAL_INDEPENDENT_VALIDATION.json",
        "sha256": "c053e9362b24694f6999a8cfa6b89fbc38f846d55f9b77d8bdea7ff7d88b0dec"
      },
      {
        "name": "ALPHAXIV_COMMENT_RECEIPT.json",
        "sha256": "514f7aee3a35b20e365ea9ceb2979cc169cecdd0874a55ecb9ac77b693d737cd"
      }
    ]
  }
}
