{
  "format": "proofrun-model-release-publication",
  "schemaVersion": 1,
  "contentHash": "sha256:41385fcdec0cbde7fc0815c2b4d1693eaa0af03eb920e876ea1f2dd1e974a9f9",
  "preparedAt": "2026-08-29T04:50:10.273Z",
  "editorial": {
    "format": "editorial-feature",
    "headline": "Hy4 preview completed two of three Proofrun tests—and the third stopped at the provider boundary",
    "dek": "Across Proofrun’s first Model Release Profile, Tencent’s exact FP8 route scored 65 on Triage Desk and 61 on Verdant Terminal; Weather Story produced no artifact after an HTTP 429, so it remains incident evidence—not a model failure.",
    "slug": "hy4-preview-completed-two-of-three-proofrun-tests-and-the-third-stopped-at-the-provider-boundary",
    "article": {
      "path": "article.md",
      "contentHash": "sha256:6a6010a6466c29bfd5023ca72416428e518fef5d6a281166e586aaf8a30e9877",
      "approvedAt": "2026-08-29T04:50:10.091Z",
      "draftingModel": "openai/gpt-5.6-sol"
    }
  },
  "profile": {
    "id": "MRP-001",
    "title": "Hy4 preview Model Release Profile",
    "contentHash": "sha256:8d271fb4180236a10c29a826cb3ec16f88c62871c2cb443d6f5c522219b7235b",
    "researchQuestion": "Across Proofrun's fixed software anchor, current Quick rotation, and current Voxel Forge rotation, what run-level evidence does the exact Tencent Hy4 preview FP8 route produce under one High native-reasoning policy?",
    "status": "terminal",
    "authorizationId": "MLA-7828660B54A61C38",
    "authorizationHash": "sha256:98dc14eae952f62ad6954a213dd43f1e62fbad8f79123f1c29dfc6ad9bea6885",
    "approvedAt": "2026-08-28T10:57:16.448Z",
    "rotationsConsumedAt": "2026-08-28T16:11:21.718Z",
    "resumptionCount": 2,
    "suite": {
      "id": "proofrun-model-release-suite",
      "version": 1,
      "revision": 2,
      "contentHash": "sha256:2b1a15d4e309c7c6cd886f1f424c0131eaadec6d141d8584974630b6322db4c4"
    },
    "protocol": {
      "id": "proofrun-review",
      "version": "0.8",
      "revision": 3,
      "contentHash": "sha256:6bda63bb3ec8881dc9d84ffdc1bb1e4fbfd3faee9d850680bfaf862f20a5e1d6"
    }
  },
  "release": {
    "tier": "major-or-frontier-release",
    "modelId": "tencent/hy4-preview",
    "canonicalSlug": "tencent/hy4-preview-20260827",
    "displayName": "Tencent Hy4 preview",
    "releasedAt": "2026-08-28",
    "officialEvidence": [
      {
        "type": "vendor-announcement",
        "url": "https://x.com/TencentHunyuan/status/2093222928720761009",
        "label": "Tencent Hy announcement"
      },
      {
        "type": "official-model-repository",
        "url": "https://github.com/Tencent-Hunyuan/Hy4-preview",
        "label": "Tencent Hy4 preview repository"
      }
    ],
    "rationale": "Tencent presents Hy4 preview as a 770B-total, 49B-active open-weight frontier model with a 1M context window and explicit claims in software engineering, front-end quality, game development, office artifacts, and scientific reasoning. Those claims overlap materially with Proofrun's validated web and Voxel Forge lanes."
  },
  "condition": {
    "model": "tencent/hy4-preview",
    "route": {
      "mode": "pinned",
      "endpointSlug": "tencent/fp8",
      "provider": "Tencent Cloud",
      "quantization": "fp8",
      "fallbacksAllowed": false,
      "parametersRequired": true
    },
    "reasoningEffort": "high",
    "reasoningAllowanceTokens": 16000,
    "reasoningMechanism": "native-effort",
    "artifactBudgetEnforcement": "target-only",
    "temperature": 0.7,
    "vendorRecommendedTemperature": 0.9,
    "temperatureDifferenceDisclosed": true,
    "wallLimitMinutes": 20,
    "writingGraceMinutes": 10,
    "attemptsPerSlot": 1,
    "automaticRetries": false
  },
  "summary": {
    "slotCount": 3,
    "completedArtifacts": 2,
    "noSubmissions": 1,
    "providerIncidents": 1,
    "knownReportedCostUsd": 0.107696068,
    "unknownChargeSlotCount": 1,
    "knownWallTimeMs": 592754,
    "scoreAggregation": "forbidden"
  },
  "slots": [
    {
      "id": "standard-anchor",
      "ordinal": 1,
      "selection": "fixed",
      "disposition": "published",
      "test": {
        "id": "triage-desk",
        "title": "Triage Desk",
        "category": "Product workflow",
        "version": 1,
        "hash": "curated:triage-desk:v1",
        "scope": "standard",
        "executionMode": "web-single-file-v2",
        "summary": "A compact support-operations workspace that tests information architecture, validation, and synchronized multi-step state.",
        "prompt": "Build a responsive one-page support operations workspace called TRIAGE DESK for a small software team. Seed exactly four incidents: INC-204, Checkout timeout, Northstar Bikes, Critical, Web, 6 min, Unassigned; INC-203, Duplicate invoice, Kanso Studio, High, Email, 18 min, Noor Patel; INC-202, Export stuck at 99%, Fieldnote Labs, Medium, Chat, 42 min, Imani Brooks; and INC-201, Cannot rename workspace, Alder & Co, Low, Email, 1 hr, Leo Martins. All four begin Open. Make urgency and current queue state easy to scan. Include text search, status filters for All, Open, and Resolved, and a severity filter; combinations must work together and show a clear empty-result state. Selecting a queue item must open a useful incident detail with the incident facts, a short customer report, and an activity timeline. In the detail, let the operator change severity, choose among Unassigned, Mira Chen, Noor Patel, Imani Brooks, and Leo Martins, write an internal note, and resolve the incident. Resolution must be blocked when the note is blank, with visible guidance. A successful resolution must preserve the note in the timeline, show a clear Resolved state, and update the queue, filters, and counts without losing the selected incident. Keep the workflow useful at narrow mobile widths and support keyboard operation with visible focus and properly associated labels. For standardized evidence, use these exact hooks: id=\"ticket-search\" on the search input, id=\"severity-select\" on the detail severity select with lowercase option values, id=\"owner-select\" on the owner select with value \"mira\" for Mira Chen, id=\"internal-note\" on the note field, and id=\"incident-status\" on the selected incident's visible status. Give the selected incident control an accessible name containing its ID and title, and give the final action the accessible name \"Resolve incident\". Use one self-contained offline HTML file with vanilla CSS and JavaScript, with no external assets, fonts, libraries, APIs, or network requests."
      },
      "run": {
        "id": "PR-MTD5FJQT-3F6B5B",
        "startedAt": "2026-08-28T16:11:21.652Z",
        "runOutcome": "solo_artifact",
        "protocolHash": "sha256:6bda63bb3ec8881dc9d84ffdc1bb1e4fbfd3faee9d850680bfaf862f20a5e1d6",
        "scoresLocked": true,
        "identitiesRevealed": true
      },
      "artifact": {
        "model": "tencent/hy4-preview",
        "outcome": "completed_submission",
        "craftScore": 65,
        "compliancePercent": 90,
        "confidence": "medium",
        "rationale": "A very average result further weighed down by the lack of responsiveness in the UI.",
        "requirements": [
          {
            "label": "Exactly the four frozen incidents and their supplied facts are present, with every incident initially Open.",
            "judgment": "pass"
          },
          {
            "label": "The queue makes urgency, assignment, status, and useful aggregate counts easy to scan.",
            "judgment": "pass"
          },
          {
            "label": "Text search, status filters, and severity filtering compose correctly and include a clear empty-result state.",
            "judgment": "pass"
          },
          {
            "label": "Selecting an incident keeps the queue context visible and reveals its facts, customer report, and activity timeline.",
            "judgment": "pass"
          },
          {
            "label": "Severity and owner changes remain synchronized between the selected detail, queue, filters, and counts.",
            "judgment": "pass"
          },
          {
            "label": "Attempting to resolve with a blank internal note is blocked and produces visible, actionable guidance.",
            "judgment": "pass"
          },
          {
            "label": "A successful resolution preserves the note in the timeline and clearly changes the incident to Resolved.",
            "judgment": "pass"
          },
          {
            "label": "After resolution, the selected detail, queue row, filters, and aggregate counts remain mutually consistent.",
            "judgment": "pass"
          },
          {
            "label": "The workflow remains usable at narrow widths and provides semantic labels, keyboard operation, and visible focus.",
            "judgment": "fail"
          },
          {
            "label": "The artifact is one responsive, self-contained HTML document that works offline without external dependencies.",
            "judgment": "pass"
          }
        ],
        "criteria": [
          {
            "label": "Workflow reasoning",
            "description": "Whether selection, assignment, notes, validation, resolution, and dependent state changes form one coherent operational flow.",
            "weight": 30,
            "score": 7
          },
          {
            "label": "Information design",
            "description": "Scanability, hierarchy, density, prioritization, and the relationship between the queue and incident detail.",
            "weight": 25,
            "score": 6
          },
          {
            "label": "Interaction & accessibility",
            "description": "Control clarity, keyboard use, focus visibility, responsive adaptation, validation, and useful feedback.",
            "weight": 25,
            "score": 5
          },
          {
            "label": "State robustness",
            "description": "Consistency of incident data, filters, counts, selected detail, and edge cases as the workflow changes.",
            "weight": 20,
            "score": 10
          }
        ]
      },
      "attempt": {
        "id": "PA-PR-MTD5FJQT-3F6B5B-A-01",
        "ordinal": 1,
        "artifactOutcome": "completed_submission",
        "terminalCause": "normal",
        "startedAt": "2026-08-28T16:11:21.825Z",
        "finishedAt": "2026-08-28T16:16:28.252Z",
        "provider": "Tencent",
        "endpointSlug": "tencent/fp8",
        "finishReason": "stop",
        "wallTimeMs": 306427,
        "timeToFirstVisibleTokenMs": 221399,
        "promptTokens": 611,
        "completionTokens": 23775,
        "reasoningTokens": 16523,
        "reportedCostUsd": 0.059970849,
        "failure": null
      },
      "evidence": [
        {
          "id": "EV-da0a7b56-f0b8-4f46-9605-9006862645d0",
          "path": "evidence/slot-01-01.png",
          "contentHash": "sha256:30adc736fea41301931f90e175c70d55b453519b20dc10aaef184c014fa3f0e4",
          "width": 1280,
          "height": 720,
          "capturedAt": "2026-08-28T17:33:58.279Z",
          "method": "browser-paint-isolated-v1",
          "provenance": "harness-matched",
          "state": "initial",
          "label": "Matched initial state",
          "caption": "Fresh artifact state captured at the fixed 1280 × 720 review viewport.",
          "primary": true
        }
      ],
      "limitations": [
        "Selected screenshots do not independently prove every interaction, accessibility path, or responsive breakpoint."
      ]
    },
    {
      "id": "quick-rotation",
      "ordinal": 2,
      "selection": "rotation",
      "disposition": "published",
      "test": {
        "id": "weather-story",
        "title": "Weather Story",
        "category": "Data visualization",
        "version": 2,
        "hash": "curated:weather-story:v2",
        "scope": "quick",
        "executionMode": "web-single-file-v2",
        "summary": "Turns the same small dataset into a useful, visually opinionated forecast.",
        "prompt": "Design a polished one-page weather experience for Helsinki using this fixed six-hour temperature sequence: 12 degrees, 13 degrees, 15 degrees, 14 degrees, 11 degrees, 9 degrees. Show the progression visually, highlight the warmest hour, and include wind and rain probability as invented but clearly labeled demo data. Add one meaningful hover or tap interaction. Use one offline HTML file with vanilla CSS and JavaScript; no libraries, APIs, or external assets."
      },
      "run": {
        "id": "PR-MTD6103R-932CEA",
        "startedAt": "2026-08-28T16:28:02.631Z",
        "runOutcome": "solo_no_valid_submission",
        "protocolHash": "sha256:6bda63bb3ec8881dc9d84ffdc1bb1e4fbfd3faee9d850680bfaf862f20a5e1d6",
        "scoresLocked": true,
        "identitiesRevealed": true
      },
      "artifact": {
        "model": "tencent/hy4-preview",
        "outcome": "no_submission",
        "craftScore": null,
        "compliancePercent": null,
        "confidence": null,
        "rationale": null,
        "requirements": [],
        "criteria": []
      },
      "attempt": {
        "id": "PA-PR-MTD6103R-932CEA-A-01",
        "ordinal": 1,
        "artifactOutcome": "no_submission",
        "terminalCause": "provider_incident_or_capacity",
        "startedAt": "2026-08-28T16:28:02.754Z",
        "finishedAt": "2026-08-28T16:28:03.663Z",
        "provider": "Tencent Cloud",
        "endpointSlug": "tencent/fp8",
        "finishReason": null,
        "wallTimeMs": null,
        "timeToFirstVisibleTokenMs": null,
        "promptTokens": null,
        "completionTokens": null,
        "reasoningTokens": null,
        "reportedCostUsd": null,
        "failure": {
          "source": "openrouter",
          "httpStatus": 429,
          "message": "Provider returned error",
          "observedAt": "2026-08-28T16:28:03.663Z"
        }
      },
      "evidence": [],
      "limitations": [
        "No model artifact existed to inspect or score; this slot records a provider incident, not evidence of model quality."
      ]
    },
    {
      "id": "differentiated-spatial-rotation",
      "ordinal": 3,
      "selection": "rotation",
      "disposition": "published",
      "test": {
        "id": "voxel-forge-verdant-terminal",
        "title": "Verdant Terminal",
        "category": "Voxel Forge",
        "version": 3,
        "hash": "curated:voxel-forge-verdant-terminal:v3",
        "scope": "standard",
        "executionMode": "scene-script-v1",
        "summary": "A layered botanical station that tests architectural repetition, readable interiors, transparency, and camera discipline.",
        "prompt": "Create a cinematic cutaway voxel scene called VERDANT TERMINAL: a monumental glass-roofed botanical railway hall built on a grounded station platform. Two parallel rail lines must pass visibly through the hall. Place a stationary train on one line with a recognizable front or locomotive and at least two linked cars. Build an elevated pedestrian bridge that crosses both tracks and visibly reaches the platform areas at both ends by stairs or ramps. Give the hall a repeating structural frame and a glass roof while keeping the interior easy to read. Integrate trees, planters, vines, or other greenery across several distinct parts of the station, and use a luminous clock or departure board as a strong interior landmark. Contrast warm inhabited light with cooler rail, frame, or glass materials. Choose a camera that clearly reveals the tracks, train, bridge, roof, vegetation, and vertical layers without flattening the scene. Use reusable groups or arrays for repeated architecture where they improve command economy."
      },
      "run": {
        "id": "PR-MTD7GPX4-2EF8BE",
        "startedAt": "2026-08-28T17:08:15.542Z",
        "runOutcome": "solo_artifact",
        "protocolHash": "sha256:6bda63bb3ec8881dc9d84ffdc1bb1e4fbfd3faee9d850680bfaf862f20a5e1d6",
        "scoresLocked": true,
        "identitiesRevealed": true
      },
      "artifact": {
        "model": "tencent/hy4-preview",
        "outcome": "completed_submission",
        "craftScore": 61,
        "compliancePercent": 80,
        "confidence": "medium",
        "rationale": "Average to slightly below average result containing flaws like the rails missing, issues with the stairs to the railway bridge and overall lack of detail.",
        "requirements": [
          {
            "label": "A grounded station hall has a clearly readable cutaway interior rather than only an exterior facade.",
            "judgment": "pass"
          },
          {
            "label": "Two parallel rail lines visibly pass through the station hall.",
            "judgment": "fail"
          },
          {
            "label": "A stationary train has a recognizable front or locomotive and at least two visibly linked cars.",
            "judgment": "pass"
          },
          {
            "label": "An elevated pedestrian bridge spans both rail lines and visibly reaches the platform areas at both ends by stairs or ramps.",
            "judgment": "partial"
          },
          {
            "label": "A repeating structural frame supports a visibly glass roof without obscuring the main interior.",
            "judgment": "pass"
          },
          {
            "label": "Botanical details are integrated across multiple distinct parts of the station rather than confined to one token planter.",
            "judgment": "pass"
          },
          {
            "label": "A luminous clock or departure board forms a strong interior landmark.",
            "judgment": "partial"
          },
          {
            "label": "Warm inhabited lighting contrasts deliberately with cooler rail, frame, or glass materials.",
            "judgment": "pass"
          },
          {
            "label": "The selected camera keeps the tracks, train, bridge, roof, vegetation, and vertical layering readable together.",
            "judgment": "pass"
          },
          {
            "label": "The response produces a renderable SceneScript scene within the frozen validator limits.",
            "judgment": "pass"
          }
        ],
        "criteria": [
          {
            "label": "Spatial reasoning",
            "description": "Coherent three-dimensional placement, scale, connection, support, and navigable relationships.",
            "weight": 30,
            "score": 6
          },
          {
            "label": "Composition",
            "description": "Silhouette, depth, focal hierarchy, camera choice, balance, and use of negative space.",
            "weight": 30,
            "score": 7
          },
          {
            "label": "Scene craft",
            "description": "Material contrast, detail distribution, repetition, variation, and voxel-specific visual character.",
            "weight": 25,
            "score": 5
          },
          {
            "label": "Command economy",
            "description": "Complexity achieved through valid, bounded SceneScript rather than wasteful repetition or ignored operations.",
            "weight": 15,
            "score": 9
          }
        ]
      },
      "attempt": {
        "id": "PA-PR-MTD7GPX4-2EF8BE-A-01",
        "ordinal": 1,
        "artifactOutcome": "completed_submission",
        "terminalCause": "normal",
        "startedAt": "2026-08-28T17:08:15.670Z",
        "finishedAt": "2026-08-28T17:13:01.997Z",
        "provider": "Tencent",
        "endpointSlug": "tencent/fp8",
        "finishReason": "stop",
        "wallTimeMs": 286327,
        "timeToFirstVisibleTokenMs": 267575,
        "promptTokens": 910,
        "completionTokens": 18779,
        "reasoningTokens": 17193,
        "reportedCostUsd": 0.047725219,
        "failure": null
      },
      "evidence": [
        {
          "id": "EV-b50efbec-7d88-4406-906a-d20c64bbd174",
          "path": "evidence/slot-03-01.png",
          "contentHash": "sha256:a3f82cf0c5d8b4f05aeaef51abda4e978df09d646817c17e7c6735e9a7f31f46",
          "width": 1280,
          "height": 720,
          "capturedAt": "2026-08-28T17:47:26.193Z",
          "method": "voxel-framebuffer-v1",
          "provenance": "harness-matched",
          "state": "initial",
          "label": "Matched initial state · model camera",
          "caption": "Fresh artifact state captured at the fixed 1280 × 720 viewport from the contestant-authored camera; the reviewer inspection view was restored after capture.",
          "primary": true
        },
        {
          "id": "EV-c9b65d34-81ac-48ba-801e-16938de8df4f",
          "path": "evidence/slot-03-02.png",
          "contentHash": "sha256:ad7622e1c27bc5e242b78776ee3685c26e52ddb489caafe01ac7815e2a18a9d1",
          "width": 1280,
          "height": 720,
          "capturedAt": "2026-08-28T17:47:26.296Z",
          "method": "voxel-framebuffer-v1",
          "provenance": "harness-supplementary-context",
          "state": "context",
          "label": "Deterministic context view",
          "caption": "Supplementary 1280 × 720 perspective context view. Proofrun follows the standardized camera direction, retargets toward the scene bounds, uses a moderate lens, and deliberately favors readable subject scale over complete edge coverage. The authored environment and canonical camera evidence remain unchanged.",
          "primary": false
        },
        {
          "id": "EV-4736d073-ff63-4e3b-b561-dd79e65e575a",
          "path": "evidence/slot-03-03.png",
          "contentHash": "sha256:aef7f17563cf59dd41088f0296c02a9dbdecefc056ef94f668ef159bfeca8010",
          "width": 1280,
          "height": 720,
          "capturedAt": "2026-08-28T17:47:26.381Z",
          "method": "voxel-framebuffer-v1",
          "provenance": "harness-supplementary-survey",
          "state": "survey",
          "label": "Deterministic orthographic survey",
          "caption": "Supplementary 1280 × 720 orthographic survey fitted to the complete voxel bounds from the standardized camera direction. It preserves the authored environment while removing perspective shrinkage; the canonical camera remains standardized primary evidence.",
          "primary": false
        }
      ],
      "limitations": [
        "Static captures cannot establish motion or every navigable viewpoint."
      ]
    }
  ],
  "disclosure": {
    "scope": "This profile reports three prospectively selected Solo runs for one exact model route. Each slot remains an independent case study; no composite model score or universal rating is produced.",
    "limitations": [
      "Two completed artifacts and one provider incident cannot establish a stable distribution of model quality or reliability.",
      "The Weather Story slot produced no submission and therefore supports no claim about Hy4's ability on that test.",
      "The profile used Proofrun's frozen temperature of 0.7 rather than the vendor-recommended 0.9; the difference is disclosed, not corrected after seeing results.",
      "Known reported cost excludes any unreceipted charge that may have resulted from the Weather provider incident."
    ],
    "privateMaterialExcluded": [
      "raw model responses",
      "reasoning text",
      "generated executable HTML",
      "credentials",
      "router metadata",
      "generation identifiers",
      "private reviewer notes"
    ],
    "generatedHtmlExecutedByPublicRecord": false,
    "compositeScorePublished": false,
    "sponsorship": {
      "sponsored": false,
      "vendorSuppliedAccess": false,
      "vendorSuppliedCredits": false,
      "materialConflict": null
    }
  }
}