From c482c630b14d00b4bee9456257ece7aee2f0bd8a Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sat, 19 Sep 2026 19:23:45 -0400 Subject: [PATCH 01/16] feat: add trusted measurement receipt controls before native qualification MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Land hosted source/publication issuance, exact immutable artifact transport, and independently tested Python client foundations without changing native pilot selection, matrix routing, or allocation. Document reader-first deployment and reviewed issuer/reader pins before a producer qualification change. 中文:在 native 验收前加入受信测量回执控制。提供托管源回执与发布记录 签发、准确的不可变产物传输,以及经过独立测试的 Python 客户端基础; 不改变 native pilot 选择、矩阵路由或资源分配。文档明确先部署读取端, 再部署已审查签发版本,最后通过独立变更开展产出端验收。 Validation: all 41 tests in the five requested suites pass, including installed wheel resources; Ruff checks and formatting pass for all 109 infx files. The four changed workflows pass the strict auditor-mode zizmor check locally (offline action-reference checks); no Slurm or production database execution. 验证:指定五组测试共 41 项全部通过,其中包含安装后的 wheel 资源验证; 全部 109 个 infx 文件通过 Ruff 检查和格式检查。四个变更工作流通过本地 严格 auditor 模式的 zizmor 检查(未进行在线 action 引用验证);未执行 Slurm 任务,也未写入生产数据库。 --- .github/workflows/phase1-receipt.yml | 79 + .github/workflows/recover-reused-ingest.yml | 22 + .github/workflows/run-sweep.yml | 44 + .github/workflows/stage-results.yml | 12 + docs/eval-agentx-procedures.md | 2 +- docs/eval-agentx-procedures_zh.md | 2 +- docs/measurement-receipts.md | 42 + docs/measurement-receipts_zh.md | 42 + infx/benchmarks/__init__.py | 1 + infx/benchmarks/agentx.py | 349 +++++ infx/benchmarks/cache.py | 169 +++ infx/benchmarks/common.py | 227 +++ infx/benchmarks/eval.py | 344 +++++ infx/benchmarks/identity.py | 112 ++ infx/benchmarks/identity_probe.py | 103 ++ infx/benchmarks/prepare.py | 196 +++ infx/benchmarks/resources/README.md | 9 + infx/benchmarks/resources/README_zh.md | 9 + .../resources/gsm8k-test-doc-hashes.json | 1321 +++++++++++++++++ infx/benchmarks/spec.py | 181 +++ infx/results/evals.py | 11 + infx/results/publication_receipt.py | 469 ++++++ infx/srt_slurm/contracts.py | 89 ++ infx/workflows/phase1_publication.py | 222 +++ infx/workflows/phase1_record.py | 143 ++ infx/workflows/receipt_transport.py | 278 ++++ utils/test_benchmark_preparation.py | 306 ++++ utils/test_phase1_receipt_control.py | 410 +++++ utils/test_publication_receipt.py | 276 ++++ utils/test_python_benchmark_clients.py | 594 ++++++++ utils/test_receipt_transport.py | 265 ++++ 31 files changed, 6327 insertions(+), 2 deletions(-) create mode 100644 .github/workflows/phase1-receipt.yml create mode 100644 docs/measurement-receipts.md create mode 100644 docs/measurement-receipts_zh.md create mode 100644 infx/benchmarks/__init__.py create mode 100644 infx/benchmarks/agentx.py create mode 100644 infx/benchmarks/cache.py create mode 100644 infx/benchmarks/common.py create mode 100644 infx/benchmarks/eval.py create mode 100644 infx/benchmarks/identity.py create mode 100644 infx/benchmarks/identity_probe.py create mode 100644 infx/benchmarks/prepare.py create mode 100644 infx/benchmarks/resources/README.md create mode 100644 infx/benchmarks/resources/README_zh.md create mode 100644 infx/benchmarks/resources/gsm8k-test-doc-hashes.json create mode 100644 infx/benchmarks/spec.py create mode 100644 infx/results/publication_receipt.py create mode 100644 infx/srt_slurm/contracts.py create mode 100644 infx/workflows/phase1_publication.py create mode 100644 infx/workflows/phase1_record.py create mode 100644 infx/workflows/receipt_transport.py create mode 100644 utils/test_benchmark_preparation.py create mode 100644 utils/test_phase1_receipt_control.py create mode 100644 utils/test_publication_receipt.py create mode 100644 utils/test_python_benchmark_clients.py create mode 100644 utils/test_receipt_transport.py diff --git a/.github/workflows/phase1-receipt.yml b/.github/workflows/phase1-receipt.yml new file mode 100644 index 0000000000..2f36ebeb31 --- /dev/null +++ b/.github/workflows/phase1-receipt.yml @@ -0,0 +1,79 @@ +name: Seal Phase 1 measurement receipt + +# Each reviewed approval produces an independent immutable record; later requests must not cancel pending issuances. +on: # zizmor: ignore[concurrency-limits] + workflow_dispatch: + inputs: + kind: + description: Seal source measurement or its later publication reference + required: true + default: measurement + type: choice + options: [measurement, publication] + approval-path: + description: Reviewed default-branch qualification/phase1/*.json expectation + required: true + type: string + +permissions: + contents: read + actions: read # Read source workflow attempts and download the exact artifacts being sealed. + +jobs: + seal: + name: Seal reviewed measurement or publication + if: github.ref == 'refs/heads/main' + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + ref: ${{ github.workflow_sha }} + persist-credentials: false + - uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1 + - name: Resolve reviewed expectation + env: + APPROVAL_PATH: ${{ inputs.approval-path }} + shell: python + run: | + import os + from pathlib import Path + path = Path(os.environ['APPROVAL_PATH']).resolve(strict=True) + root = Path('qualification/phase1').resolve() + if not path.is_relative_to(root) or path.suffix != '.json' or any(c in str(path) for c in '\r\n'): + raise ValueError('approval must be a reviewed qualification/phase1 JSON file') + with Path(os.environ['GITHUB_ENV']).open('a') as stream: + stream.write(f'REVIEWED_APPROVAL={path}\n') + - name: Seal independently approved complete source sweep + if: inputs.kind == 'measurement' + env: + GH_TOKEN: ${{ github.token }} + TRUSTED_WORKFLOW_SHA: ${{ github.workflow_sha }} + shell: python + run: | + import os, subprocess + subprocess.run(['uv', 'run', '--locked', 'python', '-m', 'infx.workflows.phase1_publication', + '--approval', os.environ['REVIEWED_APPROVAL'], '--archives', 'verified-archives', + '--output', 'receipt.json'], check=True) + - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + if: inputs.kind == 'measurement' + with: + name: measurement-receipt + path: receipt.json + if-no-files-found: error + - name: Bind accepted source to reviewed publication + if: inputs.kind == 'publication' + env: + GH_TOKEN: ${{ github.token }} + TRUSTED_ISSUER_SHAS: ${{ vars.INFX_RECEIPT_ISSUER_SHAS }} + DEPLOYED_READER_SHA: ${{ vars.INFX_PHASE1_READER_REVISION }} + shell: python + run: | + import os, subprocess + subprocess.run(['uv', 'run', '--locked', 'python', '-m', 'infx.workflows.phase1_record', + '--approval', os.environ['REVIEWED_APPROVAL'], '--output', 'publication.json'], check=True) + - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + if: inputs.kind == 'publication' + with: + name: publication-record + path: publication.json + if-no-files-found: error diff --git a/.github/workflows/recover-reused-ingest.yml b/.github/workflows/recover-reused-ingest.yml index a6225d44f6..6e5795dab4 100644 --- a/.github/workflows/recover-reused-ingest.yml +++ b/.github/workflows/recover-reused-ingest.yml @@ -21,13 +21,34 @@ concurrency: jobs: trigger-agentic-ingest: name: trigger-agentic-ingest + if: github.ref == 'refs/heads/main' runs-on: ubuntu-latest + permissions: + contents: read + actions: read # Verify issuer workflow attempts and download accepted receipt artifacts. steps: + - name: Checkout trusted receipt resolver + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + ref: ${{ github.workflow_sha }} + persist-credentials: false + - name: Set up uv + uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1 + - name: Resolve accepted immutable publication + id: receipt + env: + GH_TOKEN: ${{ github.token }} + SOURCE_RUN_ID: ${{ inputs.source-run-id }} + MERGE_RUN_ID: ${{ inputs.merge-run-id }} + INFX_RECEIPT_ISSUER_SHAS: ${{ vars.INFX_RECEIPT_ISSUER_SHAS }} + INFX_RECEIPT_ISSUER_WORKFLOW: ${{ vars.INFX_RECEIPT_ISSUER_WORKFLOW }} + run: uv run --locked python -m infx.workflows.receipt_transport --publication-required - name: Trigger agentic database ingest uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 env: SOURCE_RUN_ID: ${{ inputs.source-run-id }} MERGE_RUN_ID: ${{ inputs.merge-run-id }} + RECEIPT_TRANSPORT: ${{ steps.receipt.outputs.payload }} with: # Cross-repository dispatch credential for maintainer-triggered ingest recovery. github-token: ${{ secrets.FRONTEND_PAT }} # zizmor: ignore[secrets-outside-env] @@ -37,6 +58,7 @@ jobs: repo: "InferenceX-app", event_type: "ingest-agentic-results", client_payload: { + ...JSON.parse(process.env.RECEIPT_TRANSPORT), "source-run-id": process.env.SOURCE_RUN_ID, "merge-run-id": process.env.MERGE_RUN_ID, "database-target": "production" diff --git a/.github/workflows/run-sweep.yml b/.github/workflows/run-sweep.yml index 6e84196d31..ae7c35c038 100644 --- a/.github/workflows/run-sweep.yml +++ b/.github/workflows/run-sweep.yml @@ -1000,12 +1000,33 @@ jobs: needs.setup.outputs.reuse-enabled == 'true' ) runs-on: ubuntu-latest + permissions: + contents: read + actions: read # Verify issuer workflow attempts and download accepted receipt artifacts. steps: + - name: Checkout trusted receipt resolver + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + ref: ${{ needs.setup.outputs.tooling-ref }} + persist-credentials: false + - name: Set up uv + uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1 + - name: Resolve accepted immutable publication + id: receipt + env: + GH_TOKEN: ${{ github.token }} + SOURCE_RUN_ID: ${{ needs.setup.outputs.reuse-enabled == 'true' && needs.setup.outputs.reuse-source-run-id || github.run_id }} + MERGE_RUN_ID: ${{ github.run_id }} + INFX_RECEIPT_ISSUER_SHAS: ${{ vars.INFX_RECEIPT_ISSUER_SHAS }} + INFX_RECEIPT_ISSUER_WORKFLOW: ${{ vars.INFX_RECEIPT_ISSUER_WORKFLOW }} + run: uv run --locked python -m infx.workflows.receipt_transport --publication-required --defer-unsealed - name: Trigger database ingest + if: steps.receipt.outputs.ready == 'true' uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 env: SOURCE_RUN_ID: ${{ needs.setup.outputs.reuse-enabled == 'true' && needs.setup.outputs.reuse-source-run-id || github.run_id }} MERGE_RUN_ID: ${{ github.run_id }} + RECEIPT_TRANSPORT: ${{ steps.receipt.outputs.payload }} with: # Repository integration credential for the scoped sweep/ingest job. github-token: ${{ secrets.FRONTEND_PAT }} # zizmor: ignore[secrets-outside-env] @@ -1015,6 +1036,7 @@ jobs: repo: "InferenceX-app", event_type: "ingest-results", client_payload: { + ...JSON.parse(process.env.RECEIPT_TRANSPORT), "source-run-id": process.env.SOURCE_RUN_ID, "merge-run-id": process.env.MERGE_RUN_ID } @@ -1065,12 +1087,33 @@ jobs: ) ) runs-on: ubuntu-latest + permissions: + contents: read + actions: read # Verify issuer workflow attempts and download accepted receipt artifacts. steps: + - name: Checkout trusted receipt resolver + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + ref: ${{ needs.setup.outputs.tooling-ref }} + persist-credentials: false + - name: Set up uv + uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1 + - name: Resolve accepted immutable publication + id: receipt + env: + GH_TOKEN: ${{ github.token }} + SOURCE_RUN_ID: ${{ needs.setup.outputs.reuse-enabled == 'true' && needs.setup.outputs.reuse-source-run-id || github.run_id }} + MERGE_RUN_ID: ${{ github.run_id }} + INFX_RECEIPT_ISSUER_SHAS: ${{ vars.INFX_RECEIPT_ISSUER_SHAS }} + INFX_RECEIPT_ISSUER_WORKFLOW: ${{ vars.INFX_RECEIPT_ISSUER_WORKFLOW }} + run: uv run --locked python -m infx.workflows.receipt_transport --publication-required --defer-unsealed - name: Trigger agentic database ingest + if: steps.receipt.outputs.ready == 'true' uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 env: SOURCE_RUN_ID: ${{ needs.setup.outputs.reuse-enabled == 'true' && needs.setup.outputs.reuse-source-run-id || github.run_id }} MERGE_RUN_ID: ${{ github.run_id }} + RECEIPT_TRANSPORT: ${{ steps.receipt.outputs.payload }} with: # Repository integration credential for the scoped sweep/ingest job. github-token: ${{ secrets.FRONTEND_PAT }} # zizmor: ignore[secrets-outside-env] @@ -1080,6 +1123,7 @@ jobs: repo: "InferenceX-app", event_type: "ingest-agentic-results", client_payload: { + ...JSON.parse(process.env.RECEIPT_TRANSPORT), "source-run-id": process.env.SOURCE_RUN_ID, "merge-run-id": process.env.MERGE_RUN_ID, "database-target": "production" diff --git a/.github/workflows/stage-results.yml b/.github/workflows/stage-results.yml index 5145909df0..1f62c9759e 100644 --- a/.github/workflows/stage-results.yml +++ b/.github/workflows/stage-results.yml @@ -40,6 +40,16 @@ jobs: GH_TOKEN: ${{ github.token }} run: uv run --locked python -m infx.workflows.stage_results + - name: Resolve accepted immutable source receipt + id: receipt + env: + GH_TOKEN: ${{ github.token }} + SOURCE_RUN_ID: ${{ steps.request.outputs.run-id }} + MERGE_RUN_ID: ${{ steps.request.outputs.run-id }} + INFX_RECEIPT_ISSUER_SHAS: ${{ vars.INFX_RECEIPT_ISSUER_SHAS }} + INFX_RECEIPT_ISSUER_WORKFLOW: ${{ vars.INFX_RECEIPT_ISSUER_WORKFLOW }} + run: uv run --locked python -m infx.workflows.receipt_transport + - name: Acknowledge staging request id: acknowledge uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 @@ -67,6 +77,7 @@ jobs: RUN_DATE: ${{ steps.request.outputs.run-date }} REQUESTED_BY: ${{ steps.request.outputs.requested-by }} COMMENT_ID: ${{ steps.acknowledge.outputs.comment-id }} + RECEIPT_TRANSPORT: ${{ steps.receipt.outputs.payload }} with: # Cross-repository dispatch credential; trusted control code validates maintainer authorization. github-token: ${{ secrets.FRONTEND_PAT }} # zizmor: ignore[secrets-outside-env] @@ -76,6 +87,7 @@ jobs: repo: 'InferenceX-app', event_type: 'stage-results', client_payload: { + ...JSON.parse(process.env.RECEIPT_TRANSPORT), 'source-repository': `${context.repo.owner}/${context.repo.repo}`, 'pr-number': String(context.issue.number), 'run-id': process.env.RUN_ID, diff --git a/docs/eval-agentx-procedures.md b/docs/eval-agentx-procedures.md index 06c9ef6cf9..4b25bb47da 100644 --- a/docs/eval-agentx-procedures.md +++ b/docs/eval-agentx-procedures.md @@ -230,7 +230,7 @@ Treat fast results as bring-up evidence, never as a replacement for the canonica ## 8. Preserve trace and run provenance -AgentX defaults to recorded assistant-response replay. Live server outputs are measured but discarded when constructing later turns. Set `AIPERF_DATASET_WEKA_LIVE_ASSISTANT_RESPONSES=1` only for an explicitly different live-assistant experiment. The selected trace corpus is model-family dependent unless `WEKA_LOADER_OVERRIDE` pins it. The resolver logs both loader and Hugging Face dataset ([trace resolution](../benchmarks/benchmark_lib.sh#L2023-L2102), [replay semantics](../benchmarks/benchmark_lib.sh#L2104-L2270)). +At the pilot client revision `754356e9a39acc6cc6afb242d123bb57c3fb6f75`, AgentX always constructs later turns from recorded assistant-response deltas. Live server outputs are measured; `AIPERF_DATASET_WEKA_LIVE_ASSISTANT_RESPONSES` does not change this loader's behavior. A different replay methodology requires its own qualification. The legacy resolver chooses a model-family-dependent corpus unless `WEKA_LOADER_OVERRIDE` pins it. The prepared H100 client explicitly uses `semianalysis_cc_traces_weka_062126`, 393 entries, and no replay context filter ([Python client](../infx/benchmarks/agentx.py)). Capture orchestration provenance immediately: diff --git a/docs/eval-agentx-procedures_zh.md b/docs/eval-agentx-procedures_zh.md index 03c0580439..a967b1fb97 100644 --- a/docs/eval-agentx-procedures_zh.md +++ b/docs/eval-agentx-procedures_zh.md @@ -228,7 +228,7 @@ Fast 结果只能作为 bring-up 证据,绝不能替代 canonical candidate。 ## 8. 保留 trace 与运行 provenance -AgentX 默认 replay 已记录的 assistant response。实时服务输出会被测量,但构造后续 turn 时会丢弃。只有在明确要进行不同的 live-assistant 实验时,才设置 `AIPERF_DATASET_WEKA_LIVE_ASSISTANT_RESPONSES=1`。除非用 `WEKA_LOADER_OVERRIDE` 固定,否则所选 trace corpus 依赖模型 family;resolver 会同时记录 loader 与 Hugging Face dataset([trace 解析](../benchmarks/benchmark_lib.sh#L2023-L2102)、[replay 语义](../benchmarks/benchmark_lib.sh#L2104-L2270))。 +在试点固定的客户端版本 `754356e9a39acc6cc6afb242d123bb57c3fb6f75` 中,AgentX 始终使用已记录的 assistant response 增量构造后续轮次。实时服务输出会被测量;`AIPERF_DATASET_WEKA_LIVE_ASSISTANT_RESPONSES` 不会改变此 loader 的行为。不同的回放方法需要单独验证。旧版 resolver 会根据模型系列选择语料,除非用 `WEKA_LOADER_OVERRIDE` 固定。预先准备的 H100 客户端明确使用 `semianalysis_cc_traces_weka_062126`、393 条记录,并且不按上下文长度过滤回放数据([Python 客户端](../infx/benchmarks/agentx.py))。 立即记录 orchestration provenance: diff --git a/docs/measurement-receipts.md b/docs/measurement-receipts.md new file mode 100644 index 0000000000..b7999debac --- /dev/null +++ b/docs/measurement-receipts.md @@ -0,0 +1,42 @@ +# Trusted measurement receipts + +[中文](./measurement-receipts_zh.md) + +This prerequisite makes the hosted receipt issuer and immutable artifact transport available on trusted `main` before a native producer is qualified. It does not select a native pilot, modify matrix routing, add a runner launcher, or allocate a new GPU lane. The included Python clients and contract types are reusable validation/preparation foundations; existing benchmark selection remains unchanged. + +## Landing order + +1. Deploy the InferenceX-app receipt reader and migration `016_measurement_snapshots.sql`. Retain its actual deployment commit and verify the required receipt versions before enabling native publication. +2. Land this control prerequisite on InferenceX `main`. Its [issuer workflow](../.github/workflows/phase1-receipt.yml) checks out its own trusted workflow commit; it never runs candidate code to decide what measurements are acceptable. +3. Configure the reviewed issuer and deployed-reader revisions below. Native producer selection and allocation belong to a separately reviewed change, with its own hardware qualification evidence. +4. After the complete source run succeeds, issue its source receipt, validate staging, then issue a separate publication record after the approved merge run succeeds. Local control tests do not establish reader deployment, GPU qualification, or production publication. + +## Policy and reviewed inputs + +Configure `INFX_RECEIPT_ISSUER_SHAS` in both repositories as a comma-separated allowlist of reviewed InferenceX issuer commits, and set `INFX_RECEIPT_ISSUER_WORKFLOW` to `.github/workflows/phase1-receipt.yml`. Keep previously accepted issuer revisions while their immutable receipts remain supported. In InferenceX, set `INFX_PHASE1_READER_REVISION` to the final deployed app commit and record this prerequisite's trusted commit in `INFX_PHASE1_COLLECTOR_REVISION` for later producer qualification. These settings are deployment prerequisites, not values inferred from dispatch payloads. + +Maintainers review `qualification/phase1/*.json` inputs on trusted `main` before invoking the issuer with `kind: measurement`. The [Approval schema](../infx/workflows/phase1_publication.py) requires eight throughput points at concurrency 1, 2, 4, 8, 16, 20, 24, and 28 plus the real c28 GSM8K evaluation, with original source run/attempt/head, per-point execution and bundle identities, native manifest digests, and the actual corpus revision. Obtain these identities from independently prepared control records; worker `execution.json` files and artifact names cannot authorize their own expected contract. This prerequisite deliberately includes no fabricated approval file. + +The [receipt validator](../infx/results/publication_receipt.py) checks exact GitHub artifact IDs and ownership, API and ZIP digests, contained members, expected execution identities, physical topology, canonical configuration, required metrics, dataset metadata, and complete evaluation sample/filter coverage. Per-job raw lm-eval results and metadata remain valid inputs; aggregate deployments retain explicit zero split-worker counts. The compact version-1 receipt preserves original measurements when staging later becomes production. + +## Staging, publication, and recovery + +The [transport resolver](../infx/workflows/receipt_transport.py) uses read-only APIs to locate a unique accepted receipt from a successful `workflow_dispatch` issuer on `main` at an allowed revision. It verifies the original source attempt rather than substituting the latest rerun. An API inventory containing `native-execution-*` requires a receipt; missing or invalid evidence never enters legacy native ingestion. Ordinary legacy inventories retain their existing path. + +Use the existing authorized staging flow after source sealing. For production, review a [PublicationRecord](../infx/workflows/phase1_record.py) that references the original receipt artifact/digest and separately binds the merge run/SHA, changelog artifact/digest, and deployed app/ingest revisions. Issue it with `kind: publication`. Its `ingest_sha` must equal the app checkout that will execute ingestion; retain deployment evidence for `app_sha`. The source receipt is not rewritten. + +The sweep's automatic ingest jobs defer while required sealing is pending so the source or merge workflow can finish successfully. Once both issuer runs finish, the existing [recovery workflow](../.github/workflows/recover-reused-ingest.yml) dispatches exact receipt/publication IDs and both ZIP/JSON digests. Native production always requires the later record, including when source and merge run IDs are equal. The deployed app independently validates transport and accepted snapshots before import; interrupted imports may resume only the same accepted snapshot. Do not select a newer same-named artifact to repair a missing accepted one. + +## Local verification + +Run the behavior suites from this prerequisite checkout: + +```bash +uv run --locked --group test pytest -q utils/test_benchmark_preparation.py \ + utils/test_python_benchmark_clients.py utils/test_phase1_receipt_control.py \ + utils/test_publication_receipt.py utils/test_receipt_transport.py +uvx --exclude-newer PT12H ruff@latest check infx +uvx --exclude-newer PT12H ruff@latest format --check infx +``` + +The suites exercise installed package resources, prepared client identities, child-process cleanup, nine-point approval construction, artifact verification, and immutable source/publication transport without submitting a Slurm job or writing a production database. See [testing](./testing.md) and [eval/AgentX procedures](./eval-agentx-procedures.md) for the existing validation boundaries. diff --git a/docs/measurement-receipts_zh.md b/docs/measurement-receipts_zh.md new file mode 100644 index 0000000000..0f5e0f92d4 --- /dev/null +++ b/docs/measurement-receipts_zh.md @@ -0,0 +1,42 @@ +# 受信测量回执 + +[English](./measurement-receipts.md) + +这个前置变更让托管回执签发流程和不可变产物传输先部署到受信的 `main`,随后才能对 native 产出端进行验收。它不选择 native pilot,不修改矩阵路由,不新增 runner launcher,也不分配新的 GPU 执行路径。随附的 Python 客户端和契约类型是可复用的验证及准备基础;现有基准测试选择逻辑保持不变。 + +## 合入顺序 + +1. 部署 InferenceX-app 回执读取端及迁移 `016_measurement_snapshots.sql`。记录实际部署 commit,并在启用 native 发布前验证所需回执版本。 +2. 将这个控制端前置变更合入 InferenceX `main`。[签发工作流](../.github/workflows/phase1-receipt.yml) checkout 自身受信的工作流 commit,不执行候选代码来决定哪些测量结果可被接受。 +3. 配置下文列出的已审查签发版本和已部署读取端版本。native 产出端的选择与分配属于另一个独立审查的变更,必须提供相应硬件验收证据。 +4. 完整源运行成功后,先签发源回执并验证 staging;经批准的 merge 运行成功后,再签发独立发布记录。本地控制端测试不能证明读取端已部署、GPU 验收已通过或生产发布已完成。 + +## 策略与已审查输入 + +两个仓库都应配置 `INFX_RECEIPT_ISSUER_SHAS`,以逗号分隔已审查的 InferenceX 签发 commit,并将 `INFX_RECEIPT_ISSUER_WORKFLOW` 设置为 `.github/workflows/phase1-receipt.yml`。只要旧的不可变回执仍受支持,就应保留相应签发版本。在 InferenceX 中,将 `INFX_PHASE1_READER_REVISION` 固定到最终部署的应用 commit,并通过 `INFX_PHASE1_COLLECTOR_REVISION` 记录这个前置变更的受信 commit,供后续产出端验收使用。这些配置是部署前置条件,不能从调度 payload 推断。 + +维护者应先审查受信 `main` 上的 `qualification/phase1/*.json` 输入,再以 `kind: measurement` 调用签发流程。[Approval schema](../infx/workflows/phase1_publication.py) 要求 concurrency 为 1、2、4、8、16、20、24、28 的八个吞吐量点,以及实际 c28 GSM8K 评估,并记录源 run/attempt/head、逐点 execution 和 bundle 身份、native manifest digest 及实际语料版本。这些身份必须来自独立准备的控制记录;worker 的 `execution.json` 和产物名称不能自行授权预期契约。这个前置变更不包含虚构的批准文件。 + +[回执验证器](../infx/results/publication_receipt.py) 检查准确的 GitHub 产物 ID 及归属、API 与 ZIP digest、安全成员路径、预期执行身份、物理拓扑、规范化配置、必需指标、数据集元数据,以及完整的评估样本和过滤器覆盖。输入支持每个任务的原始 lm-eval 结果与元数据;聚合部署保留明确为零的拆分 worker 计数。紧凑的 version 1 回执在结果从 staging 进入生产时保留原始测量身份。 + +## Staging、发布与恢复 + +[传输解析器](../infx/workflows/receipt_transport.py) 使用只读 API,从允许版本在 `main` 上成功完成的 `workflow_dispatch` 签发运行中查找唯一的已接受回执。它验证源运行原始 attempt,不用最近一次 rerun 替换。只要 API 清单包含 `native-execution-*`,就必须提供回执;证据缺失或无效时,不能回退到旧 native 导入方式。普通旧产物清单继续使用现有路径。 + +源回执签发后,使用现有经过授权的 staging 流程。生产发布前,审查 [PublicationRecord](../infx/workflows/phase1_record.py):它引用原始回执产物及 digest,并独立绑定 merge run/SHA、changelog 产物及 digest、已部署的 app/ingest 版本。随后以 `kind: publication` 签发该记录。记录的 `ingest_sha` 必须匹配实际执行导入的应用 checkout,并应保留 `app_sha` 对应的部署证据。源回执不随发布而重写。 + +sweep 的自动导入任务在必需签发尚未完成时延后调度,让源或 merge 工作流能够成功结束。两个签发运行均完成后,现有[恢复工作流](../.github/workflows/recover-reused-ingest.yml) 会传递准确的回执和发布记录 ID,以及各自的 ZIP/JSON digest。native 生产发布始终需要后续记录,即使源运行和 merge 运行 ID 相同。已部署的应用在导入前独立验证传输和已接受 snapshot;中断恢复只能使用同一 snapshot。不要通过选择更新的同名产物来替代缺失的已接受产物。 + +## 本地验证 + +在这个前置变更的 checkout 中运行行为测试: + +```bash +uv run --locked --group test pytest -q utils/test_benchmark_preparation.py \ + utils/test_python_benchmark_clients.py utils/test_phase1_receipt_control.py \ + utils/test_publication_receipt.py utils/test_receipt_transport.py +uvx --exclude-newer PT12H ruff@latest check infx +uvx --exclude-newer PT12H ruff@latest format --check infx +``` + +这些测试覆盖已安装包资源、prepared client 身份、子进程清理、九点批准契约构建、产物验证及不可变源回执/发布传输,不提交 Slurm 任务,也不写入生产数据库。现有验证边界见[测试说明](./testing_zh.md)和 [eval/AgentX 操作流程](./eval-agentx-procedures_zh.md)。 diff --git a/infx/benchmarks/__init__.py b/infx/benchmarks/__init__.py new file mode 100644 index 0000000000..dec82a4381 --- /dev/null +++ b/infx/benchmarks/__init__.py @@ -0,0 +1 @@ +"""Prepared Python clients for srt-owned serving processes.""" diff --git a/infx/benchmarks/agentx.py b/infx/benchmarks/agentx.py new file mode 100644 index 0000000000..124993a702 --- /dev/null +++ b/infx/benchmarks/agentx.py @@ -0,0 +1,349 @@ +"""One pinned AgentX replay, followed by the existing InferenceX normalization.""" + +from __future__ import annotations + +import argparse +import math +import os +import shlex +from datetime import datetime +from pathlib import Path +from typing import Any + +from infx.results.agentic import build_result +from infx.results.agentic.artifacts import ( + find_server_log_paths, + iter_trace_blobs, + load_records_with_accounting, + load_server_log_head, + resolve_artifact_dir, +) +from infx.results.agentic.common import round_floats +from infx.results.agentic.validate_agentic_result import validate_result + +from .cache import MmapCache +from .common import ( + child_environment, + child_failed, + read_json, + run_child, + validate_endpoint, + verify_snapshot_assets, + write_json, +) +from .identity import AGENTX_REVISION, validate_cache_manifests, verify_runtime +from .spec import AgentXSpec, RuntimeSpec + + +def build_argv(spec: AgentXSpec, endpoint: str, artifact_root: Path) -> list[str]: + origin = validate_endpoint(endpoint) + return [ + spec.runtime.python, + "-I", + "-m", + "aiperf", + "profile", + "--scenario", + "inferencex-agentx-mvp", + "--url", + origin, + "--endpoint", + "/v1/chat/completions", + "--endpoint-type", + "chat", + "--streaming", + "--model", + spec.metadata.model, + "--tokenizer", + spec.tokenizer, + "--concurrency", + str(spec.concurrency), + "--benchmark-duration", + str(spec.duration_seconds), + "--stats-interval", + "30", + "--random-seed", + str(spec.random_seed), + "--failed-request-threshold", + str(spec.live_failed_request_threshold), + "--trajectory-start-min-ratio", + "0.25", + "--trajectory-start-max-ratio", + "0.75", + "--warmup-requests-per-lane", + str(spec.warmup_requests_per_lane), + "--trace-idle-gap-cap-seconds", + str(spec.trace_idle_gap_cap_seconds), + "--warmup-grace-period", + str(spec.warmup_grace_seconds), + "--use-server-token-count", + "--no-gpu-telemetry", + "--tokenizer-trust-remote-code", + "--num-dataset-entries", + str(spec.dataset_entries), + "--slice-duration", + "1.0", + "--server-metrics", + f"{origin}/metrics", + "--output-artifact-dir", + str(artifact_root / "aiperf_artifacts"), + "--public-dataset", + spec.dataset_loader, + ] + + +def replay_environment(spec: AgentXSpec) -> dict[str, str]: + allowed = {"AIPERF_DATASET_MMAP_CACHE_DIR"} + unexpected = {key for key in spec.runtime.env if key.startswith("AIPERF_")} - allowed + if unexpected: + raise ValueError(f"unqualified AIPerf environment overrides: {sorted(unexpected)}") + env = child_environment(spec.runtime) + env.update( + { + "AIPERF_DATASET_CONFIGURATION_TIMEOUT": "1800", + "AIPERF_SERVICE_PROFILE_CONFIGURE_TIMEOUT": "1800", + "AIPERF_UI_REALTIME_METRICS_ENABLED": "true", + } + ) + return env + + +def verify_corpus(spec: AgentXSpec) -> None: + verify_prepared_corpus(spec.runtime, spec.dataset_revision) + + +def verify_prepared_corpus(runtime: RuntimeSpec, revision: str) -> None: + """The pinned loader reads main; its offline cache must bind that ref to one snapshot.""" + verify_snapshot_assets( + runtime, + "semianalysisai/cc-traces-weka-062126", + expected_revision=revision, + only_snapshot=True, + ) + + +def _expect(actual: Any, expected: Any, label: str, errors: list[str]) -> None: + if actual != expected: + errors.append(f"{label}: expected {expected!r}, received {actual!r}") + + +def validate_scenario(aggregate: dict[str, Any], spec: AgentXSpec, endpoint: str) -> list[str]: + """Check effective settings independently from the scenario's own validity verdict.""" + errors: list[str] = [] + metadata = aggregate.get("metadata", {}) + _expect(metadata.get("submission_valid"), True, "scenario validity", errors) + _expect(metadata.get("scenario"), "inferencex-agentx-mvp", "scenario", errors) + dataset = metadata.get("dataset", {}) + for key, expected in { + "source_type": "public_dataset", + "loader": spec.dataset_loader, + "hf_dataset_name": spec.dataset_repository, + "hf_split": "train", + "num_dataset_entries": spec.dataset_entries, + }.items(): + _expect(dataset.get(key), expected, f"dataset.{key}", errors) + config = aggregate.get("input_config", {}) + endpoint_config = config.get("endpoint", {}) + for key, expected in { + "urls": [validate_endpoint(endpoint)], + "type": "chat", + "path": "/v1/chat/completions", + "streaming": True, + "use_server_token_count": True, + }.items(): + _expect(endpoint_config.get(key), expected, f"endpoint.{key}", errors) + _expect(config.get("models", {}).get("items"), [{"name": spec.metadata.model}], "model", errors) + _expect(config.get("tokenizer", {}).get("name"), spec.tokenizer, "tokenizer", errors) + phases = config.get("phases", []) + if len(phases) != 1 or not isinstance(phases[0], dict): + errors.append("expected one client-owned profiling phase with its own warmup/drain") + else: + for key, expected in { + "kind": "profiling", + "type": "concurrency", + "timing_mode": "agentic_replay", + "duration": spec.duration_seconds, + "concurrency": spec.concurrency, + "trajectory_start_min_ratio": 0.25, + "trajectory_start_max_ratio": 0.75, + "system_idle_gap_cap_seconds": 10.0, + "warmup_requests_per_lane": spec.warmup_requests_per_lane, + "agentic_warmup_grace_period": spec.warmup_grace_seconds, + "failed_request_threshold": spec.live_failed_request_threshold, + }.items(): + _expect(phases[0].get(key), expected, f"profiling.{key}", errors) + datasets = config.get("datasets", []) + if len(datasets) != 1 or not isinstance(datasets[0], dict): + errors.append("expected exactly one prepared dataset") + else: + for key, expected in { + "dataset": spec.dataset_loader, + "entries": spec.dataset_entries, + "random_seed": spec.random_seed, + "trace_idle_gap_cap_seconds": spec.trace_idle_gap_cap_seconds, + }.items(): + _expect(datasets[0].get(key), expected, f"effective dataset.{key}", errors) + if datasets[0].get("max_context_length") is not None: + errors.append("pilot replay must not filter traces by a context cap") + coverage = metadata.get("metric_duration_coverage", []) + if len(coverage) != 1: + errors.append("profiling temporal coverage is missing or ambiguous") + else: + reach = coverage[0] + _expect(reach.get("expected_duration_seconds"), 3600.0, "coverage duration", errors) + _expect(reach.get("required_ratio"), 0.95, "coverage requirement", errors) + ratios = [reach.get("ttft_ratio"), reach.get("inter_token_latency_ratio")] + if not any( + isinstance(value, int | float) + and not isinstance(value, bool) + and math.isfinite(value) + and value >= 0.95 + for value in ratios + ): + errors.append("neither TTFT nor successful ITL reaches 95% of the profiling duration") + return errors + + +def normalize(spec: AgentXSpec, artifact_root: Path) -> Path: + """Use the existing normalizer's successful-record spans and exclusion rules unchanged.""" + artifact_dir = resolve_artifact_dir(artifact_root) + records, accounting = load_records_with_accounting(artifact_dir / "profile_export.jsonl") + aggregate = read_json(artifact_dir / "profile_export_aiperf.json") + server_metrics = read_json(artifact_dir / "server_metrics_export.json") + env = {**spec.runtime.env, **spec.metadata.normalizer_env(spec.concurrency)} + log_directory = os.environ.get("SRT_LOG_DIR") + log_paths = ( + sorted(Path(log_directory).glob("*_agg_w*.out")) + if log_directory + else find_server_log_paths(artifact_root) + ) + result = round_floats( + build_result( + records, + aggregate, + server_metrics, + env, + request_accounting=accounting, + traces=iter_trace_blobs(aggregate, env), + server_logs=(load_server_log_head(path) for path in log_paths), + ) + ) + result.setdefault("dataset", {})["hf_revision"] = spec.dataset_revision + output = artifact_root / f"{spec.result_filename}.json" + write_json(output, result) + return output + + +def _has_metric(value: Any, prefix: str) -> bool: + if isinstance(value, dict): + return any( + key.startswith(prefix) or _has_metric(item, prefix) for key, item in value.items() + ) + if isinstance(value, list): + return any(_has_metric(item, prefix) for item in value) + return False + + +def finalize(spec: AgentXSpec, endpoint: str, artifact_root: Path) -> list[str]: + errors: list[str] = [] + artifact_dir = resolve_artifact_dir(artifact_root) + try: + # Preserve available aggregate diagnostics even when replay validation fails. + normalize(spec, artifact_root) + except (OSError, ValueError, TypeError, KeyError, SystemExit) as exc: + errors.append(f"normalization failed: {exc}") + try: + aggregate = read_json(artifact_dir / "profile_export_aiperf.json") + errors.extend(validate_scenario(aggregate, spec, endpoint)) + errors.extend(validate_result(artifact_dir, spec.failed_request_threshold)) + metrics = read_json(artifact_dir / "server_metrics_export.json") + csv = artifact_dir / "server_metrics_export.csv" + if ( + not csv.is_file() + or csv.stat().st_size == 0 + or not _has_metric(metrics, spec.required_server_metric_prefix) + ): + errors.append("required vLLM JSON/CSV server metrics are absent") + except (OSError, ValueError, TypeError, KeyError) as exc: + errors.append(f"artifact validation failed: {exc}") + return errors + + +def run(spec: AgentXSpec, endpoint: str, artifact_root: Path) -> int: + endpoint = validate_endpoint(endpoint) + env = replay_environment(spec) + identity = verify_runtime( + spec.runtime, dataset_loader=spec.dataset_loader, source_pins={"aiperf": AGENTX_REVISION} + ) + resolution = identity.get("dataset_resolution", {}) + if resolution.get("metadata", {}).get("hf_dataset_name") != spec.dataset_repository: + raise ValueError("installed dataset plugin resolves a different corpus") + verify_corpus(spec) + artifact_root.mkdir(parents=True, exist_ok=True) + if (artifact_root / "aiperf_artifacts").exists(): + raise ValueError("client artifact root already contains replay output") + argv = build_argv(spec, endpoint, artifact_root) + (artifact_root / "benchmark_command.txt").write_text(shlex.join(argv) + "\n") + # The historical power adapter needs the producer timezone for naive AIPerf timestamps. + (artifact_root / "agentic_power_timezone_offset.txt").write_text( + datetime.now().astimezone().strftime("%z") + "\n" + ) + cache = MmapCache( + Path(env["AIPERF_DATASET_MMAP_CACHE_DIR"]), + { + "client": spec.runtime.identity.sha256, + "assets": {asset.path: asset.sha256 for asset in spec.runtime.assets}, + "dataset_revision": spec.dataset_revision, + "tokenizer": spec.tokenizer, + "entries": spec.dataset_entries, + "seed": spec.random_seed, + }, + lock_timeout_seconds=spec.runtime.terminate_grace_seconds, + validate_manifests=lambda path: validate_cache_manifests(spec.runtime, path), + ) + try: + env["AIPERF_DATASET_MMAP_CACHE_DIR"] = str(cache.prepare()) + status = run_child( + argv, + env=env, + cwd=artifact_root, + log=artifact_root / "benchmark.log", + timeout_seconds=spec.runtime.timeout_seconds, + terminate_grace_seconds=spec.runtime.terminate_grace_seconds, + ) + errors = finalize(spec, endpoint, artifact_root) + if not child_failed(status) and not errors: + cache.publish() + finally: + cache.close() + write_json( + artifact_root / "diagnostics" / "client-audit.json", + { + "schema_version": 1, + "client": "agentx", + "status": status, + "errors": errors, + "prepared_identity_sha256": spec.runtime.identity.sha256, + "dataset_revision": spec.dataset_revision, + "requested": spec.model_dump(mode="json", exclude={"runtime"}), + "effective_recorded_response_deltas": True, + "endpoint": endpoint, + "cache_events": cache.events, + }, + ) + return 1 if child_failed(status) or errors else 0 + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--spec", type=Path, required=True) + parser.add_argument("--endpoint", default=os.environ.get("SRT_ENDPOINT")) + parser.add_argument("--artifact-root", type=Path, required=True) + args = parser.parse_args() + if not args.endpoint: + parser.error("--endpoint or runtime-provided SRT_ENDPOINT is required") + return run(AgentXSpec.model_validate(read_json(args.spec)), args.endpoint, args.artifact_root) + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/infx/benchmarks/cache.py b/infx/benchmarks/cache.py new file mode 100644 index 0000000000..6164f3015b --- /dev/null +++ b/infx/benchmarks/cache.py @@ -0,0 +1,169 @@ +"""Integrity-checked, disposable mmap snapshots around the unchanged pinned client. + +Only the owning UID shares snapshots. A client receives an independent copy, never +a hardlink to shared data. Duplicate cold preparation is permitted on contention; +canonical publication is always locked and atomic. +""" + +from __future__ import annotations + +import fcntl +import hashlib +import json +import os +import re +import shutil +import stat +import time +import uuid +from collections.abc import Callable, Iterator +from contextlib import contextmanager +from pathlib import Path +from typing import Any + +from .common import read_json, sha256_file, write_json + + +@contextmanager +def _lock(path: Path, timeout_seconds: float) -> Iterator[None]: + with path.open("a+b") as stream: + deadline = time.monotonic() + timeout_seconds + while True: + try: + fcntl.flock(stream.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB) + break + except BlockingIOError: + if time.monotonic() >= deadline: + raise TimeoutError("derived-cache lock contention deadline exceeded") from None + time.sleep(0.02) + try: + yield + finally: + fcntl.flock(stream.fileno(), fcntl.LOCK_UN) + + +def _inventory(root: Path) -> dict[str, str]: + if root.is_symlink(): + raise ValueError("derived cache snapshot must not be a symlink") + inventory: dict[str, str] = {} + for entry in sorted(root.iterdir()): + if not entry.is_dir() or re.fullmatch(r"[0-9a-f]{32}", entry.name) is None: + continue + if entry.is_symlink(): + raise ValueError("derived cache entries must not be symlinks") + manifest = read_json(entry / "manifest.json") + if manifest.get("cache_key") != entry.name or manifest.get("compressed") is not False: + raise ValueError( + "derived cache has an unexpected key or unsupported compressed payload" + ) + for name in ("dataset.dat", "index.dat"): + payload = entry / name + if not payload.is_file() or payload.stat().st_size == 0: + raise ValueError("derived cache payload is incomplete") + for path in sorted(entry.rglob("*")): + if path.is_symlink(): + raise ValueError("derived cache payload must not be a symlink") + if path.is_file(): + inventory[path.relative_to(root).as_posix()] = sha256_file(path) + if not inventory: + raise ValueError("derived cache has no complete client-produced entry") + return inventory + + +class MmapCache: + def __init__( + self, + base: Path, + contract: dict[str, Any], + *, + lock_timeout_seconds: float, + validate_manifests: Callable[[Path], None] | None = None, + ) -> None: + if not base.is_absolute(): + raise ValueError("mmap cache base must be explicit and absolute") + identity = hashlib.sha256( + json.dumps(contract, sort_keys=True, separators=(",", ":")).encode() + ).hexdigest() + private = base / f"infx-uid-{os.getuid()}" + private.mkdir(mode=0o700, parents=True, exist_ok=True) + mode = private.stat() + if private.is_symlink() or mode.st_uid != os.getuid() or stat.S_IMODE(mode.st_mode) & 0o077: + raise ValueError( + "client cache namespace must be private and owned by the executing UID" + ) + self.root = private / identity + self.root.mkdir(mode=0o700, exist_ok=True) + if self.root.is_symlink(): + raise ValueError("derived cache namespace must not be a symlink") + self.canonical = self.root / "complete" + self.run_dir = self.root / f"run-{uuid.uuid4().hex}" + self.lock_timeout_seconds = lock_timeout_seconds + self.validate_manifests = validate_manifests + self.events: list[str] = [] + + def _check(self) -> dict[str, str]: + receipt = read_json(self.canonical / "infx-integrity.json") + current = _inventory(self.canonical) + if self.validate_manifests is not None: + self.validate_manifests(self.canonical) + if receipt != {"schema_version": 1, "files": current}: + raise ValueError("derived cache payload differs from its integrity receipt") + return current + + def _quarantine(self) -> None: + self.canonical.rename(self.root / f"quarantine-{uuid.uuid4().hex}") + self.events.append("invalid snapshot quarantined; rebuilding from authoritative inputs") + + def prepare(self) -> Path: + try: + with _lock(self.root / "publication.lock", self.lock_timeout_seconds): + if self.canonical.exists(): + try: + expected = self._check() + except (OSError, ValueError, TypeError, KeyError): + self._quarantine() + else: + shutil.copytree(self.canonical, self.run_dir, copy_function=shutil.copyfile) + (self.run_dir / "infx-integrity.json").unlink() + if _inventory(self.run_dir) != expected: + raise ValueError("derived cache changed while copying") + self.events.append("verified snapshot copied without hardlinks") + return self.run_dir + except TimeoutError: + self.events.append("lock contention: preparing an independent cold cache") + self.run_dir.mkdir(mode=0o700) + return self.run_dir + + def publish(self) -> None: + """Call only after the foreground client and its result validators succeeded.""" + try: + inventory = _inventory(self.run_dir) + if self.validate_manifests is not None: + self.validate_manifests(self.run_dir) + except (OSError, ValueError, TypeError, KeyError) as exc: + self.events.append(f"cache publication skipped: {exc}") + return + try: + with _lock(self.root / "publication.lock", self.lock_timeout_seconds): + if self.canonical.exists(): + try: + self._check() + except (OSError, ValueError, TypeError, KeyError): + self._quarantine() + else: + self.events.append("another completed snapshot already published") + return + write_json( + self.run_dir / "infx-integrity.json", {"schema_version": 1, "files": inventory} + ) + # This is an owned directory populated by one exited client. Rename + # publishes payload and integrity receipt together, never a half entry. + self.run_dir.rename(self.canonical) + self.events.append("verified snapshot published atomically") + except TimeoutError: + self.events.append("cache publication skipped on lock contention") + + def close(self) -> None: + # Quarantined shared evidence remains for diagnosis; only this run's copy is removed. + if self.run_dir.exists(): + shutil.rmtree(self.run_dir) diff --git a/infx/benchmarks/common.py b/infx/benchmarks/common.py new file mode 100644 index 0000000000..9506785eb8 --- /dev/null +++ b/infx/benchmarks/common.py @@ -0,0 +1,227 @@ +"""Filesystem and child-process boundaries shared by prepared clients.""" + +from __future__ import annotations + +import hashlib +import json +import math +import os +import re +import signal +import subprocess +import time +from collections.abc import Mapping, Sequence +from pathlib import Path +from typing import Any +from urllib.parse import urlsplit + +from .spec import PreparedFile, RuntimeSpec, secret_environment_key + + +def sha256_file(path: Path) -> str: + with path.open("rb") as stream: + return hashlib.file_digest(stream, "sha256").hexdigest() + + +def verify_file(file: PreparedFile) -> Path: + path = Path(file.path) + if not path.is_file() or sha256_file(path) != file.sha256: + raise ValueError(f"prepared file missing or changed: {path}") + return path + + +def verify_snapshot_assets( + runtime: RuntimeSpec, repository: str, *, expected_revision: str | None, only_snapshot: bool +) -> str: + dataset = Path(runtime.env["HF_HUB_CACHE"]) / ("datasets--" + repository.replace("/", "--")) + reference = dataset / "refs" / "main" + revision = reference.read_text().strip() + if re.fullmatch(r"[0-9a-f]{40}", revision) is None: + raise ValueError("prepared dataset revision must be an immutable snapshot SHA") + if expected_revision is not None and revision != expected_revision: + raise ValueError("offline dataset main ref does not match the prepared snapshot") + snapshot = dataset / "snapshots" / revision + if not snapshot.is_dir(): + raise ValueError("prepared dataset snapshot is unavailable") + files = [path for path in snapshot.rglob("*") if path.is_file()] + if not files: + raise ValueError("prepared dataset snapshot is empty") + bound = {Path(asset.path).resolve() for asset in runtime.assets} + if any(path.resolve() not in bound for path in (reference, *files)): + raise ValueError( + "dataset snapshot/ref contains content absent from the prepared asset list" + ) + if only_snapshot and [path for path in snapshot.parent.iterdir() if path.is_dir()] != [ + snapshot + ]: + raise ValueError("pilot dataset cache must contain only the prepared snapshot") + return revision + + +def _unique_object(pairs: list[tuple[str, Any]]) -> dict[str, Any]: + result: dict[str, Any] = {} + for key, value in pairs: + if key in result: + raise ValueError(f"duplicate JSON key: {key}") + result[key] = value + return result + + +def _invalid_constant(value: str) -> None: + raise ValueError(f"non-finite JSON number: {value}") + + +def read_json(path: Path) -> Any: + return decode_json(path.read_text()) + + +def decode_json(text: str) -> Any: + value = json.loads(text, object_pairs_hook=_unique_object, parse_constant=_invalid_constant) + require_finite(value) + return value + + +def require_finite(value: Any) -> None: + if isinstance(value, float) and not math.isfinite(value): + raise ValueError("non-finite numeric value") + if isinstance(value, dict): + for item in value.values(): + require_finite(item) + if isinstance(value, list): + for item in value: + require_finite(item) + + +def write_json(path: Path, payload: Any) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + temporary = path.with_name(f".{path.name}.{os.getpid()}.tmp") + with temporary.open("x") as stream: + json.dump(payload, stream, indent=2, allow_nan=False) + stream.write("\n") + stream.flush() + os.fsync(stream.fileno()) + temporary.replace(path) + + +def validate_endpoint(endpoint: str) -> str: + url = urlsplit(endpoint) + if ( + url.scheme not in {"http", "https"} + or not url.hostname + or url.username is not None + or url.password is not None + or url.query + or url.fragment + or url.path not in {"", "/"} + ): + raise ValueError("endpoint must be an HTTP(S) origin without credentials or a path") + # Accessing port also rejects malformed or out-of-range port values. + _ = url.port + return endpoint.rstrip("/") + + +def child_environment(runtime: RuntimeSpec) -> dict[str, str]: + env = dict(os.environ) + for key in (*runtime.env_unset, "PYTHONPATH", "PYTHONHOME"): + env.pop(key, None) + # Ambient AIPerf settings can silently change the registered scenario. + for key in list(env): + if key.startswith("AIPERF_") or secret_environment_key(key): + env.pop(key) + env.update(runtime.env) + return env + + +def run_child( + argv: Sequence[str], + *, + env: Mapping[str, str], + cwd: Path, + log: Path, + timeout_seconds: float, + terminate_grace_seconds: float, +) -> dict[str, Any]: + """Run one foreground client, forwarding cancellation to its entire process group.""" + stopped_by: int | None = None + + def stop(signum: int, _frame: Any) -> None: + nonlocal stopped_by + if stopped_by is None: + stopped_by = signum + + previous = {sig: signal.signal(sig, stop) for sig in (signal.SIGINT, signal.SIGTERM)} + started = time.monotonic() + child: subprocess.Popen[bytes] | None = None + try: + with log.open("xb") as output: + child = subprocess.Popen( + list(argv), + env=dict(env), + cwd=cwd, + stdout=output, + stderr=subprocess.STDOUT, + start_new_session=True, + ) + timed_out = False + cleanup_deadline: float | None = None + while child.poll() is None: + if stopped_by is not None or time.monotonic() - started >= timeout_seconds: + timed_out = stopped_by is None + with _process_gone_ok(): + os.killpg(child.pid, signal.SIGTERM) + # One deadline; repeated TERM does not restart or reenter cleanup. + cleanup_deadline = time.monotonic() + terminate_grace_seconds + while child.poll() is None and time.monotonic() < cleanup_deadline: + time.sleep(0.05) + # The leader may have exited while descendants still hold artifact files. + with _process_gone_ok(): + os.killpg(child.pid, signal.SIGKILL) + child.wait() + break + time.sleep(0.05) + orphaned_descendants = False + try: + os.killpg(child.pid, 0) + except ProcessLookupError: + pass + else: + orphaned_descendants = True + with _process_gone_ok(): + os.killpg(child.pid, signal.SIGTERM) + deadline = cleanup_deadline or time.monotonic() + terminate_grace_seconds + while time.monotonic() < deadline: + try: + os.killpg(child.pid, 0) + except ProcessLookupError: + break + time.sleep(0.05) + with _process_gone_ok(): + os.killpg(child.pid, signal.SIGKILL) + return { + "returncode": child.returncode, + "cancelled_by_signal": stopped_by, + "timed_out": timed_out, + "orphaned_descendants": orphaned_descendants, + } + finally: + if child is not None and child.poll() is None: + with _process_gone_ok(): + os.killpg(child.pid, signal.SIGKILL) + child.wait() + for sig, handler in previous.items(): + signal.signal(sig, handler) + + +def _process_gone_ok() -> Any: + from contextlib import suppress + + return suppress(ProcessLookupError) + + +def child_failed(status: Mapping[str, Any]) -> bool: + return bool( + status["returncode"] != 0 + or status["cancelled_by_signal"] + or status["timed_out"] + or status["orphaned_descendants"] + ) diff --git a/infx/benchmarks/eval.py b/infx/benchmarks/eval.py new file mode 100644 index 0000000000..0c9b02371e --- /dev/null +++ b/infx/benchmarks/eval.py @@ -0,0 +1,344 @@ +"""Real GSM8K verification against an srt-owned endpoint, retaining raw samples.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import math +import os +import shlex +from collections import Counter +from importlib.resources import files +from pathlib import Path +from typing import Any + +from infx.srt_slurm.contracts import load_mapping + +from .common import ( + child_environment, + child_failed, + decode_json, + read_json, + require_finite, + run_child, + validate_endpoint, + verify_file, + verify_snapshot_assets, + write_json, +) +from .identity import LM_EVAL_REVISION, verify_runtime +from .spec import EvalSpec + +FILTERS = ("strict-match", "flexible-extract") + + +def packaged_task_path() -> Path: + """Installed wheels include the task; this path never depends on the checkout cwd.""" + return Path(str(files("infx.evals").joinpath("gsm8k.yaml"))) + + +def build_argv(spec: EvalSpec, endpoint: str, artifact_root: Path) -> list[str]: + origin = validate_endpoint(endpoint) + patch = str(files("infx.evals.patches").joinpath("lm_eval_sitecustomize.py")) + # Apply the existing compatibility patch explicitly. Python -I must not rely on + # the legacy PYTHONPATH/sitecustomize injection or on an importable infx in 3.11. + bootstrap = ( + "import runpy,sys; " + "runpy.run_path(sys.argv.pop(1)); " + "runpy.run_module('lm_eval', run_name='__main__')" + ) + model_args = { + "model": spec.metadata.model, + "base_url": f"{origin}/v1/chat/completions", + "api_key": "EMPTY", + "eos_string": "", + "max_retries": 5, + "num_concurrent": spec.concurrency, + "timeout": 1800, + "tokenized_requests": False, + "max_length": spec.max_length, + } + # Harness model_args is a comma-delimited format; reject values that could + # change its parse instead of shell-quoting an unsafe semantic value. + if any( + any(character in str(value) for character in (",", "\n", "\r")) + for value in model_args.values() + ): + raise ValueError("eval model/endpoint cannot contain delimiters in harness model_args") + return [ + spec.runtime.python, + "-I", + "-c", + bootstrap, + patch, + "--model", + "local-chat-completions", + "--apply_chat_template", + "--tasks", + spec.task.path, + "--output_path", + str(artifact_root / "harness"), + "--log_samples", + "--model_args", + ",".join(f"{key}={value}" for key, value in model_args.items()), + "--gen_kwargs", + f"max_tokens={spec.max_tokens},temperature=0,top_p=1", + ] + + +def eval_metadata(spec: EvalSpec, *, complete: bool) -> dict[str, Any]: + metadata = spec.metadata + return { + "is_multinode": False, + "disagg": False, + "framework": metadata.framework, + "precision": metadata.precision, + "spec_decoding": metadata.spec_decoding, + "eval_suite": spec.task_name, + "recipe_fingerprint": metadata.recipe_fingerprint, + "tp": metadata.tp, + "pp": 1, + "dcp_size": 1, + "pcp_size": 1, + "conc": spec.concurrency, + "ep": 1, + "dp_attention": False, + "prefill_tp": metadata.tp, + "prefill_pp": 1, + "prefill_dcp_size": 1, + "prefill_pcp_size": 1, + "prefill_ep": 1, + "prefill_dp_attention": False, + "prefill_num_workers": 0, + "decode_tp": metadata.tp, + "decode_pp": 1, + "decode_dcp_size": 1, + "decode_pcp_size": 1, + "decode_ep": 1, + "decode_dp_attention": False, + "decode_num_workers": 0, + "num_gpus": metadata.tp, + "model": metadata.model, + "infmax_model_prefix": metadata.model_prefix, + "hw": metadata.hw, + "isl": "0", + "osl": "0", + "eval_concs": [spec.concurrency], + "completed_eval_concs": [spec.concurrency] if complete else [], + "failed_eval_concs": [] if complete else [spec.concurrency], + "deployment": { + "kind": "aggregate", + "nodes": 1, + "serving_gpus": metadata.tp, + "tp": metadata.tp, + "ep": 1, + }, + } + + +def stage_outputs(spec: EvalSpec, artifact_root: Path) -> tuple[list[Path], list[Path]]: + """Keep raw harness files and stage compatibility names without silent overwrite.""" + results: list[Path] = [] + samples: list[Path] = [] + for path in sorted((artifact_root / "harness").rglob("*")): + if not path.is_file(): + continue + if path.name.startswith("results") and path.suffix == ".json": + group = results + elif path.name.startswith("sample") and path.suffix == ".jsonl": + group = samples + else: + continue + target = artifact_root / f"{path.stem}_conc{spec.concurrency}{path.suffix}" + with target.open("xb") as output, path.open("rb") as source: + # Raw evidence is byte-for-byte identical; no reserialization. + import shutil + + shutil.copyfileobj(source, output) + group.append(target) + return results, samples + + +def _expected_task(spec: EvalSpec) -> dict[str, Any]: + task = load_mapping(verify_file(spec.task)) + expected = { + "task": "gsm8k", + "dataset_path": "openai/gsm8k", + "dataset_name": "main", + "training_split": "train", + "test_split": "test", + "fewshot_split": "train", + "num_fewshot": 5, + "repeats": 1, + "output_type": "generate_until", + } + if not isinstance(task, dict) or any(task.get(key) != value for key, value in expected.items()): + raise ValueError("prepared task does not describe the full canonical GSM8K contract") + if [item["name"] for item in task.get("filter_list", [])] != list(FILTERS): + raise ValueError("prepared task must contain strict-match and flexible-extract filters") + return task + + +def validate_outputs( + spec: EvalSpec, endpoint: str, result_files: list[Path], sample_files: list[Path] +) -> list[str]: + errors: list[str] = [] + if len(result_files) != 1 or len(sample_files) != 1: + return ["expected exactly one GSM8K result file and one complete sample file"] + result = read_json(result_files[0]) + task = _expected_task(spec) + identities = read_json(verify_file(spec.document_identities)) + if not isinstance(identities, dict) or set(identities) != { + str(index) for index in range(spec.expected_documents) + }: + raise ValueError("prepared document identities must cover the entire expected GSM8K split") + config = result.get("config", {}) + model_args = config.get("model_args", {}) + for key, value in { + "model": spec.metadata.model, + "base_url": f"{validate_endpoint(endpoint)}/v1/chat/completions", + "num_concurrent": spec.concurrency, + "max_length": spec.max_length, + "tokenized_requests": False, + }.items(): + if model_args.get(key) != value: + errors.append(f"eval model_args.{key} differs from the independently expected contract") + if config.get("model") != "local-chat-completions" or config.get("limit") is not None: + errors.append("eval must use the real full-split local-chat-completions adapter") + if config.get("gen_kwargs") != {"max_tokens": spec.max_tokens, "temperature": 0, "top_p": 1}: + errors.append("eval generation budget/sampling settings differ from the expected contract") + if result.get("n-samples") != { + spec.task_name: {"original": spec.expected_documents, "effective": spec.expected_documents} + }: + errors.append("eval n-samples does not contain the full expected task split") + configs = result.get("configs", {}) + if set(configs) != {spec.task_name}: + errors.append("eval result contains missing or unexpected tasks") + emitted_task = configs.get(spec.task_name, {}) + for key, value in task.items(): + if key in {"tag", "metadata", "generation_kwargs"}: + continue + if emitted_task.get(key) != value: + errors.append(f"eval task.{key} differs from the prepared task") + expected_generation = {**task["generation_kwargs"], **config.get("gen_kwargs", {})} + if emitted_task.get("generation_kwargs") != expected_generation: + errors.append("eval task generation settings differ from the prepared task") + scores = result.get("results", {}).get(spec.task_name, {}) + seen: set[tuple[int, str]] = set() + sums: Counter[str] = Counter() + with sample_files[0].open() as stream: + for line_number, line in enumerate(stream, 1): + sample = decode_json(line) + require_finite(sample) + doc_id, filter_name = sample.get("doc_id"), sample.get("filter") + if ( + not isinstance(doc_id, int) + or isinstance(doc_id, bool) + or str(doc_id) not in identities + or filter_name not in FILTERS + ): + errors.append(f"sample line {line_number} has unexpected document/filter identity") + continue + key = (doc_id, filter_name) + if key in seen: + errors.append(f"duplicate eval sample: {key}") + continue + seen.add(key) + digest = hashlib.sha256( + json.dumps(sample.get("doc"), indent=2, ensure_ascii=False).encode() + ).hexdigest() + if digest != identities[str(doc_id)] or sample.get("doc_hash") != digest: + errors.append(f"sample document bytes/hash differ from the prepared split: {key}") + target = sample.get("target") + if ( + target != sample.get("doc", {}).get("answer") + or sample.get("target_hash") != hashlib.sha256(str(target).encode()).hexdigest() + ): + errors.append(f"sample target/hash mismatch: {key}") + value = sample.get("exact_match") + if isinstance(value, bool) or value not in (0, 1): + errors.append(f"sample exact_match must be zero or one: {key}") + else: + sums[filter_name] += value + expected = {(index, name) for index in range(spec.expected_documents) for name in FILTERS} + if seen != expected: + errors.append( + f"eval sample coverage incomplete: {len(seen)}/{len(expected)} document/filter pairs" + ) + for name in FILTERS: + metric = scores.get(f"exact_match,{name}") + if ( + not isinstance(metric, int | float) + or isinstance(metric, bool) + or not math.isfinite(metric) + ): + errors.append(f"missing or non-finite exact_match,{name}") + elif metric < spec.minimum_score or not math.isclose( + metric, sums[name] / spec.expected_documents, rel_tol=0, abs_tol=1e-12 + ): + errors.append(f"exact_match,{name} fails its threshold or disagrees with raw samples") + return errors + + +def run(spec: EvalSpec, endpoint: str, artifact_root: Path) -> int: + endpoint = validate_endpoint(endpoint) + verify_runtime(spec.runtime, dataset_loader=None, source_pins={"lm-eval": LM_EVAL_REVISION}) + verify_snapshot_assets( + spec.runtime, "openai/gsm8k", expected_revision=None, only_snapshot=False + ) + _expected_task(spec) + verify_file(spec.document_identities) + artifact_root.mkdir(parents=True, exist_ok=True) + if (artifact_root / "harness").exists(): + raise ValueError("client artifact root already contains eval output") + argv = build_argv(spec, endpoint, artifact_root) + (artifact_root / "eval_command.txt").write_text(shlex.join(argv) + "\n") + status = run_child( + argv, + env=child_environment(spec.runtime), + cwd=artifact_root, + log=artifact_root / "eval.log", + timeout_seconds=spec.runtime.timeout_seconds, + terminate_grace_seconds=spec.runtime.terminate_grace_seconds, + ) + errors: list[str] = [] + try: + result_files, sample_files = stage_outputs(spec, artifact_root) + errors = validate_outputs(spec, endpoint, result_files, sample_files) + except (OSError, ValueError, TypeError, KeyError) as exc: + errors.append(f"eval artifact validation failed: {exc}") + failed = child_failed(status) or bool(errors) + write_json(artifact_root / "meta_env.json", eval_metadata(spec, complete=not failed)) + write_json( + artifact_root / "diagnostics" / "client-audit.json", + { + "schema_version": 1, + "client": "lm-eval", + "status": status, + "errors": errors, + "prepared_identity_sha256": spec.runtime.identity.sha256, + "task_sha256": spec.task.sha256, + "document_identities_sha256": spec.document_identities.sha256, + "endpoint": endpoint, + "expected_documents": spec.expected_documents, + "max_length": spec.max_length, + "max_tokens": spec.max_tokens, + }, + ) + return 1 if failed else 0 + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--spec", type=Path, required=True) + parser.add_argument("--endpoint", default=os.environ.get("SRT_ENDPOINT")) + parser.add_argument("--artifact-root", type=Path, required=True) + args = parser.parse_args() + if not args.endpoint: + parser.error("--endpoint or runtime-provided SRT_ENDPOINT is required") + return run(EvalSpec.model_validate(read_json(args.spec)), args.endpoint, args.artifact_root) + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/infx/benchmarks/identity.py b/infx/benchmarks/identity.py new file mode 100644 index 0000000000..2bf1b97df7 --- /dev/null +++ b/infx/benchmarks/identity.py @@ -0,0 +1,112 @@ +"""Capture and verify prepared installed-client identity, with no dependency installation.""" + +from __future__ import annotations + +import argparse +import json +import subprocess +from importlib.resources import as_file, files +from pathlib import Path +from typing import Any + +from .common import child_environment, read_json, verify_file, write_json +from .spec import RuntimeSpec + +AGENTX_REVISION = "754356e9a39acc6cc6afb242d123bb57c3fb6f75" +LM_EVAL_REVISION = "b315ef3b05176acc9732bb7fdec116abe1ecc476" + + +def require_source_revision(identity: dict[str, Any], name: str, revision: str) -> None: + distribution = identity.get("distributions", {}).get(name, {}) + source = distribution.get("direct_url") or {} + if source.get("dir_info", {}).get("editable"): + raise ValueError( + f"editable client installation is not an immutable prepared runtime: {name}" + ) + if source.get("vcs_info", {}).get("commit_id") != revision: + raise ValueError(f"{name} must be installed from immutable reviewed revision {revision}") + + +def validate_cache_manifests(runtime: RuntimeSpec, root: Path) -> None: + """Use the selected client's real schemas, including its nested dataset metadata.""" + manifests = sorted(root.glob("*/manifest.json")) + if not manifests: + raise ValueError("client produced no mmap manifest") + script = ( + "import sys; from pathlib import Path; " + "from aiperf.dataset.mmap_cache import CacheManifest, MANIFEST_VERSION; " + "from aiperf.common.models import DatasetMetadata; " + "\nfor value in sys.argv[1:]:\n" + " manifest = CacheManifest.model_validate_json(Path(value).read_text())\n" + " if manifest.version != MANIFEST_VERSION: raise ValueError('cache schema version differs')\n" + " DatasetMetadata.model_validate_json(manifest.dataset_metadata_json)\n" + ) + result = subprocess.run( + [runtime.python, "-I", "-c", script, *(str(path) for path in manifests)], + env=child_environment(runtime), + capture_output=True, + text=True, + check=False, + timeout=60, + ) + if result.returncode: + raise ValueError("installed client rejected mmap manifest or nested dataset metadata") + + +def capture_identity( + python: str, + distributions: list[str], + *, + dataset_loader: str | None, + env: dict[str, str] | None = None, +) -> dict[str, Any]: + resource = files("infx.benchmarks").joinpath("identity_probe.py") + with as_file(resource) as probe: + argv = [python, "-I", str(probe)] + for name in distributions: + argv.extend(("--distribution", name)) + if dataset_loader is not None: + argv.extend(("--dataset-loader", dataset_loader)) + result = subprocess.run( + argv, env=env, capture_output=True, check=True, text=True, timeout=600 + ) + return json.loads(result.stdout) + + +def verify_runtime( + runtime: RuntimeSpec, *, dataset_loader: str | None, source_pins: dict[str, str] | None = None +) -> dict[str, Any]: + expected = read_json(verify_file(runtime.identity)) + for asset in runtime.assets: + verify_file(asset) + actual = capture_identity( + runtime.python, + runtime.distributions, + dataset_loader=dataset_loader, + env=child_environment(runtime), + ) + if actual != expected: + raise ValueError( + "installed client/interpreter/plugin identity differs from prepared identity" + ) + for name, revision in (source_pins or {}).items(): + require_source_revision(actual, name, revision) + return actual + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--python", required=True) + parser.add_argument("--distribution", action="append", required=True) + parser.add_argument("--dataset-loader") + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + write_json( + args.output, + capture_identity(args.python, args.distribution, dataset_loader=args.dataset_loader), + ) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/infx/benchmarks/identity_probe.py b/infx/benchmarks/identity_probe.py new file mode 100644 index 0000000000..0f7afa9959 --- /dev/null +++ b/infx/benchmarks/identity_probe.py @@ -0,0 +1,103 @@ +"""Standalone Python 3.11 probe; deliberately does not import the Python 3.12 wrapper.""" + +from __future__ import annotations + +import argparse +import contextlib +import hashlib +import importlib.metadata +import json +import platform +import re +import sys +from pathlib import Path +from typing import Any +from urllib.parse import urlsplit + + +def _file_digest(path: Path) -> str: + with path.open("rb") as stream: + return hashlib.file_digest(stream, "sha256").hexdigest() + + +def inspect_runtime(distributions: list[str], dataset_loader: str | None) -> dict[str, Any]: + def normalized(name: str) -> str: + return re.sub(r"[-_.]+", "-", name).lower() + + installed = sorted( + (normalized(dist.metadata["Name"]), dist.version) + for dist in importlib.metadata.distributions() + ) + if len({name for name, _ in installed}) != len(installed): + raise ValueError("installed runtime has ambiguous duplicate distributions") + entries = sorted( + (entry.group, entry.name, entry.value, entry.dist.metadata["Name"]) + for entry in importlib.metadata.entry_points(group="aiperf.plugins") + ) + selected = ( + {normalized(name) for name in distributions} + | {normalized(entry[3]) for entry in entries} + | {name for name, _ in installed} + ) + distribution_files = {} + for name in sorted(selected): + dist = importlib.metadata.distribution(name) + if not dist.files: + raise ValueError(f"installed distribution has no file manifest: {name}") + files = {} + for entry in sorted(dist.files): + if entry.suffix == ".pyc" or "__pycache__" in entry.parts: + continue + path = Path(dist.locate_file(entry)) + if not path.is_file(): + raise ValueError(f"installed distribution file missing: {name}/{entry}") + files[str(entry)] = _file_digest(path) + raw_url = dist.read_text("direct_url.json") + direct_url = json.loads(raw_url) if raw_url else None + url = urlsplit(direct_url.get("url", "")) if direct_url else None + if url is not None and (url.username is not None or url.password is not None): + raise ValueError( + "installed source metadata contains credentials and cannot be published" + ) + distribution_files[name] = { + "version": dist.version, + "files": files, + "direct_url": direct_url, + } + resolution = None + if dataset_loader is not None: + # Import only inside the selected installed child, never from a checkout. + with contextlib.redirect_stdout(sys.stderr): + from aiperf.plugin import plugins + + entry = plugins.get_entry("public_dataset_loader", dataset_loader) + resolution = entry.model_dump(mode="json", exclude={"loaded_class"}) + return { + "schema_version": 1, + "python_version": platform.python_version(), + "python_build": list(platform.python_build()), + "python_sha256": _file_digest(Path(sys.executable).resolve()), + "python_paths": { + "executable": str(Path(sys.executable).absolute()), + "executable_resolved": str(Path(sys.executable).resolve()), + "prefix": str(Path(sys.prefix).absolute()), + "base_prefix": str(Path(sys.base_prefix).absolute()), + }, + "installed": installed, + "distributions": distribution_files, + "aiperf_plugin_entry_points": entries, + "dataset_resolution": resolution, + } + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--distribution", action="append", required=True) + parser.add_argument("--dataset-loader") + args = parser.parse_args() + print(json.dumps(inspect_runtime(args.distribution, args.dataset_loader), allow_nan=False)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/infx/benchmarks/prepare.py b/infx/benchmarks/prepare.py new file mode 100644 index 0000000000..988b1b175f --- /dev/null +++ b/infx/benchmarks/prepare.py @@ -0,0 +1,196 @@ +"""Freeze already materialized clients, model assets and evaluation resources before allocation.""" + +from __future__ import annotations + +import argparse +import os +import shutil +from importlib.resources import files +from pathlib import Path +from typing import Annotated, Any, Literal, Self + +from pydantic import Field, field_validator, model_validator + +from .common import read_json, sha256_file, verify_snapshot_assets, write_json +from .identity import ( + AGENTX_REVISION, + LM_EVAL_REVISION, + capture_identity, + require_source_revision, +) +from .spec import ( + PositiveInt, + PreparedFile, + RuntimeSpec, + StrictModel, + secret_environment_key, + validate_environment, +) + + +class ClientSite(StrictModel): + python: str + distributions: Annotated[list[str], Field(min_length=1)] + env: dict[str, str] + env_unset: list[str] + asset_roots: Annotated[list[str], Field(min_length=1)] + asset_files: list[str] + model_path: str + timeout_seconds: PositiveInt + terminate_grace_seconds: PositiveInt + + @field_validator("python", "model_path") + @classmethod + def absolute_path(cls, value: str) -> str: + return PreparedFile.absolute_path(value) + + @field_validator("asset_roots", "asset_files") + @classmethod + def absolute_paths(cls, values: list[str]) -> list[str]: + return [PreparedFile.absolute_path(value) for value in values] + + @model_validator(mode="after") + def environment_contract(self) -> Self: + validate_environment(self.env, self.env_unset) + return self + + +def bind_file(path: Path) -> PreparedFile: + """Hash the actual bytes, rejecting replacement/truncation while reading.""" + before = path.stat() + if not path.is_file(): + raise ValueError(f"required prepared asset is not a file: {path}") + digest = sha256_file(path) + after = path.stat() + if (before.st_dev, before.st_ino, before.st_size, before.st_mtime_ns) != ( + after.st_dev, + after.st_ino, + after.st_size, + after.st_mtime_ns, + ): + raise ValueError(f"asset changed during preparation: {path}") + return PreparedFile(path=str(path.absolute()), sha256=digest) + + +def collect_assets(site: ClientSite) -> list[PreparedFile]: + paths = {Path(path) for path in site.asset_files} + for directory in site.asset_roots: + root = Path(directory) + if not root.is_dir(): + raise ValueError(f"prepared asset root is unavailable: {root}") + children = list(root.rglob("*")) + if any(child.is_symlink() and not child.exists() for child in children): + raise ValueError(f"prepared asset root contains dangling blob links: {root}") + payloads = [child for child in children if child.is_file()] + if not payloads: + raise ValueError(f"prepared asset root is empty: {root}") + paths.update(payloads) + assets = [bind_file(path) for path in sorted(paths)] + bound = {Path(asset.path).resolve() for asset in assets} + model = Path(site.model_path) + config = model / "config.json" + indexes = list(model.glob("*.safetensors.index.json")) + list(model.glob("*.bin.index.json")) + if config.resolve() not in bound or not indexes: + raise ValueError("prepared model requires bound config.json and a weight shard index") + for index in indexes: + if index.resolve() not in bound: + raise ValueError(f"model shard index is not bound: {index}") + mapping = read_json(index).get("weight_map", {}) + if not mapping: + raise ValueError(f"model shard index contains no weights: {index}") + for name in set(mapping.values()): + relative = Path(name) + if relative.is_absolute() or ".." in relative.parts: + raise ValueError(f"model shard escapes model directory: {name}") + if (model / relative).resolve() not in bound or (model / relative).stat().st_size == 0: + raise ValueError(f"model shard is missing or unbound: {name}") + tokenizer_files = ("tokenizer.json", "tokenizer.model", "tokenizer.tiktoken") + if not any((model / name).resolve() in bound for name in tokenizer_files): + raise ValueError("prepared model tokenizer payload is missing or unbound") + return assets + + +def prepare(site: ClientSite, kind: Literal["agentx", "eval"], output: Path) -> dict[str, Any]: + """Prepare a new, exclusive directory; never mutate a previously prepared environment.""" + output = output.absolute() + assets = collect_assets(site) + distribution, revision, loader = ( + ("aiperf", AGENTX_REVISION, "semianalysis_cc_traces_weka_062126") + if kind == "agentx" + else ("lm-eval", LM_EVAL_REVISION, None) + ) + if distribution not in site.distributions: + raise ValueError(f"site must bind the {distribution} installed distribution") + env = dict(os.environ) + for key in (*site.env_unset, "PYTHONPATH", "PYTHONHOME"): + env.pop(key, None) + for key in list(env): + if key.startswith("AIPERF_") or secret_environment_key(key): + env.pop(key) + env.update(site.env) + identity = capture_identity(site.python, site.distributions, dataset_loader=loader, env=env) + require_source_revision(identity, distribution, revision) + resources: dict[str, Any] = {"schema_version": 1, "kind": kind, "model_path": site.model_path} + if kind == "agentx": + dataset = Path(site.env["HF_HUB_CACHE"]) / "datasets--semianalysisai--cc-traces-weka-062126" + reference = dataset / "refs/main" + resources["dataset_revision"] = reference.read_text().strip() + if ( + identity.get("dataset_resolution", {}).get("metadata", {}).get("hf_dataset_name") + != "semianalysisai/cc-traces-weka-062126" + ): + raise ValueError("installed plugin resolves a different dataset") + output.mkdir(parents=True, exist_ok=False) + try: + identity_path = output / "identity.json" + write_json(identity_path, identity) + runtime = RuntimeSpec( + python=site.python, + identity=bind_file(identity_path), + distributions=site.distributions, + env=site.env, + env_unset=site.env_unset, + assets=assets, + timeout_seconds=site.timeout_seconds, + terminate_grace_seconds=site.terminate_grace_seconds, + ) + if kind == "agentx": + # Reuse the same actual client cache boundary before allocating a server. + from .agentx import verify_prepared_corpus + + verify_prepared_corpus(runtime, resources["dataset_revision"]) + else: + resources["dataset_revision"] = verify_snapshot_assets( + runtime, "openai/gsm8k", expected_revision=None, only_snapshot=False + ) + for key, source in { + "task": files("infx.evals").joinpath("gsm8k.yaml"), + "document_identities": files("infx.benchmarks").joinpath( + "resources/gsm8k-test-doc-hashes.json" + ), + }.items(): + target = output / source.name + target.write_bytes(source.read_bytes()) + resources[key] = bind_file(target).model_dump(mode="json") + write_json(output / "runtime.json", runtime.model_dump(mode="json")) + write_json(output / "prepared-resources.json", resources) + for path in output.iterdir(): + path.chmod(0o444) + except BaseException: + shutil.rmtree(output) + raise + return resources + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--site", type=Path, required=True) + parser.add_argument("--kind", choices=("agentx", "eval"), required=True) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + prepare(ClientSite.model_validate(read_json(args.site)), args.kind, args.output) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/infx/benchmarks/resources/README.md b/infx/benchmarks/resources/README.md new file mode 100644 index 0000000000..51a4abee1f --- /dev/null +++ b/infx/benchmarks/resources/README.md @@ -0,0 +1,9 @@ +# Prepared evaluation resources + +**English** | [中文](./README_zh.md) + +`gsm8k-test-doc-hashes.json` maps the 1,319 GSM8K test document IDs to their independently captured `doc_hash` values. It contains hashes, not dataset text or model responses. The source is the H100 c28 real-verification evaluation from InferenceX run `35314892357`, artifact `10535461349`, measured with lm-evaluation-harness revision `b315ef3b05176acc9732bb7fdec116abe1ecc476`. + +The client validates both filters for every expected document, recomputes each hash from the emitted document bytes using the pinned harness serialization, and checks scores against the raw sample values. A high aggregate score or a self-reported sample count cannot replace those checks. Changing the test split requires an explicitly reviewed resource and workload identity change. + +The preparation command copies this file and the packaged `infx/evals/gsm8k.yaml` into the prepared directory and binds their SHA256 digests. Client execution works from an installed wheel outside the checkout and rejects changed resource bytes. diff --git a/infx/benchmarks/resources/README_zh.md b/infx/benchmarks/resources/README_zh.md new file mode 100644 index 0000000000..d279c6e92c --- /dev/null +++ b/infx/benchmarks/resources/README_zh.md @@ -0,0 +1,9 @@ +# 预先准备的评测资源 + +[English](./README.md) | **中文** + +`gsm8k-test-doc-hashes.json` 将 GSM8K 测试集的 1,319 个文档 ID 映射到独立获取的 `doc_hash` 值。文件只包含哈希,不包含数据集文本或模型响应。来源是 InferenceX run `35314892357`、artifact `10535461349` 中 H100 c28 的真实验证评测,使用的 lm-evaluation-harness 版本为 `b315ef3b05176acc9732bb7fdec116abe1ecc476`。 + +客户端会检查每个预期文档的两种过滤结果,按照固定 harness 版本的序列化方式重新计算输出文档的哈希,并将分数与原始样本值核对。较高的汇总分数或评测自身报告的样本数量都不能替代这些检查。更换测试集必须明确审查资源变更,并更新工作负载身份。 + +准备命令会将此文件以及包内的 `infx/evals/gsm8k.yaml` 复制到准备目录,并绑定它们的 SHA256 摘要。客户端通过已安装的 wheel 运行,不依赖当前目录或源码检出位置,并会拒绝字节内容已变化的资源。 diff --git a/infx/benchmarks/resources/gsm8k-test-doc-hashes.json b/infx/benchmarks/resources/gsm8k-test-doc-hashes.json new file mode 100644 index 0000000000..117749d3bd --- /dev/null +++ b/infx/benchmarks/resources/gsm8k-test-doc-hashes.json @@ -0,0 +1,1321 @@ +{ + "0": "986c19252f3c7f6eaeff2229e9a5777f96a85c0ceea931a92d2289084dfd518c", + "1": "00f836ab1a16e62131c59e2c6ce14552295991067d5b9f912f47449305e3b9e1", + "2": "de128ec686007fbcc408f283e02aca3b6ccbd40a0c6753d1e26c577174daecc8", + "3": "8823f438f491a49f230c43d5bc7c20dcf512b5dc15004a22a612f21ce2cdfcc4", + "4": "0c507c8d1d6245ffde3a097fa1a23c62ae2cd5ad98c9afeca1620903a9394a1d", + "5": "e580fc21f83b173523ba2f7d2617e3eebc959f167353efaa94ad3012e49b63ae", + "6": "c8c0e6bfa7261691391120318e967c7ae26fc12367e0997bdb7bfba47d40f110", + "7": "f0a02d2a0a80efb1fb716182e98a736ad470456cb405fcfe340c80ddf7d372b5", + "8": "1397c7e79cc5e1445dd7a1c22a2b75a6e0e8f563b62f107f971c63521c4da975", + "9": "b6c4831e84d0e3f16798974be6974bbcb3be3d504075c897a67229f03d2681d6", + "10": "1756b8e9ad639654faac7aa29d70707bfdec707f53649a0a3085b72dd5b198a0", + "11": "795a465691618faf27beb20bf2bea04a5b05d350cfb63047df69c82d963740c1", + "12": "d437de7aa5ea30e12e09dde3c9e3765ffe44b155a400a97fca1c85d98c783958", + "13": "244dfdfeb94bc5031b2a3f275ecebe0b3d0c9ce285c608d5bf2fa039ebfaa0f5", + "14": "e395b97afd4e2642f294903f1fd68149674f8907045936a57c22346910f40bd4", + "15": "4a40afe4b5d55b5aa59ca1c97e1ac1a808ea4dc591abd8f12a939f3cfff53347", + "16": "8d52662ed683404d1c5f225090293bd707c942fb370cf956572bef5e4057b3a6", + "17": "ad662c6b86448869ef350d4e32c0949d5373ca786126a665119d96fc35297d79", + "18": "be370d8da8f4f002296c5911387c0453734c07dde3f9306408cd3e5402f1e8f5", + "19": "1ed20b5d88e265ab77b9d36fa9384d3e6b15213fe61061fa1e3183c0752f9b19", + "20": "472a555c76575e7265e9ef58f715ea12c0dbf9577fda8df9580adad39fa34ff8", + "21": "4bea1d97ce1bfcc410e3510d44bcebf3ab162331c354bf4bfcdfd26c9a2e9f68", + "22": "9b6d47931068ec6fabcef2148777efab60c11dbce6cfe989dad4c2b8ead82c15", + "23": "dc2fc8059a325b5a717dd76cdd6c43fed23efaccdb44a3c7a7d4370c7a714a32", + "24": "45df19b1a75c387814959a7134eec55583a1c5c73ea71b3497ebfe4924d4b032", + "25": "e2ea3ea0dd7dece8ddeb1fcc4edcf51e8f4ecbcd27e410fe39ff9e2653a87eef", + "26": "9299dd077a50b9492329d305bb480283549b0fcafb6c9d49cf6dfee2794d0441", + "27": "c0202174f7400834d71c569f687941f5e9eaf92cd34fbcb170cad4835b08ec47", + "28": "ecbf43cce471de0ca7d3231aefcd16fc9affa99a983bd3f19c35ae2be20ac6ac", + "29": "8d548386d7996cdf13b6c831aed22503bc3045bf73fe3f67ced9694f735cbfda", + "30": "e5a8a991aea280807df761995084547f606ddc9f42d45d306a37e73f04782732", + "31": "43892bbae42374d0caa57850b36b02888d24c7c843e5464072c335fa28ef061b", + "32": "767bd5dce8bbd05b99d63d65764fb1c4d43d347b97d22d790d18f5abcec89676", + "33": "6d3c1b6dbec12da8b6c3d348d65dbc3245ed1fb3b19667cfc0163cd4aa159ad0", + "34": "d25995d404c23f5ec423dd307977ae5a93852b280cb0114b3f3a2951798e216d", + "35": "32d8833436e30634f789dd2d418d2a2cc380eddc33f5e4f46db9fea9f79fab09", + "36": "64917d99c339b9db62546499fa6e96da1aeac4c809017b8acb9202dd9a88c826", + "37": "8de6d7e2e70b616d608c932b64b38867711ee3dcea416d3694e91cc66a4b2b37", + "38": "2bd97becfd4621f00f26c3496597179570351ffb34032b781df2a2f50f34112c", + "39": "5e204b3f2fd739c69bb58e4b990a715ea3a9c1b2c614fe4a23466ba957c85d48", + "40": "5f8289fcab5456e2c6a1d26b34ba9fc78020621cea9888367083d0ef864b88ac", + "41": "a41984e20a80c9e385df17747e5837e8d0677303ddbbd2a85059c2a5e88180f6", + "42": "8513310526c6d75bbf661486b68c2032bdb3241a3f3a0b042cea44a89a76a1c6", + "43": "f673c9f4d57d672e39d834548c6ea8bf7146fb4a6c2ccd1af34ff5fa0404206b", + "44": "c09a3f39a9132c1fd765815d63d66d574c32b3a0820b54adf01cbd1ce1d0bc92", + "45": "ab51dd953cb7c32b4d02644aca9daff2bdb53a22da231e996f48d0043bef5718", + "46": "f709569baf740edaee619f0f6fefa426b84f3c14a0508f80ebf0e7d5c92e8c33", + "47": "35633adaabe4268fa38a01c10b222e3ed1e70b60a07f6e6b7a5d4eb2679743fe", + "48": "21d2bfa4e704b032da8ef22d1205ad830ceea3dc956cf1a40d1f01d6ab7f71e2", + "49": "cc090ff1f75aa926b2d49e8ecf4a62c97d587aa13d2ce9c26fd97c280166bf7a", + "50": "6661c601ea616636707e99444f5da73a48c7cccc2c70570d989a36b1211b352f", + "51": "2b86bbf2e59dccbfbee7eaa6921b2bf6fd78f323364d13da6a78190cdecf6211", + "52": "930339a81a29aaf23de920b455e44deae1a963daf65da90bab8629569ef25216", + "53": "9b404d0fc5dccb7c117955dc8c03bcb7969a2d924bdcdf8e4f06a574dde5eae0", + "54": "a6625f62f9e7d861c259a07dc92f01f0904e809548be49a29e7e67f8918df441", + "55": "9a91f0fa7e5ae2ed52f02a8eb90cbabf7c9143c566b91a1bd842c84a75343392", + "56": "f792459bd8441d0be3dda845d9aaa473064577cebaf75e85d1f6e23d9791ffbb", + "57": "bfb10198703bc183ef40d129c9d7e096a1c82ff904b2519758c579fde4f4b2ad", + "58": "ce742da6e2aa57a3650535dbc2d2502d2a73bfb61787e94b179b7fa1f3210006", + "59": "a4ddc94b8c4954383a027c5fae0c70d3b2031df185a3c74c398a384450b04ea6", + "60": "bc84df0db1916c6d135a5101248773f8b249e97631bb6590202b765491919851", + "61": "e5b5d6d95e02125c9151e31d792b5ec864fd599017e0de4e04b4ddb4e25a08d6", + "62": "4fd43c58cffda2c4b37525e2678420b6172c030eed16835308415da3ac0180ad", + "63": "c365d4ff3f44eb4580a6853c86cea69f0453152eaad3d59b2c3581480295adf4", + "64": "b72d86bccd64b88922050e8b58ef4742228c1b0d5d5859aa12927e2988cefbdc", + "65": "fbfd53d1493608489c182402cf56fc3f8704e456876ddb264552fb6cf6cf5dba", + "66": "1006d895748fadc2e735c765f9ec7e7c3b235f13605e9c4956311a6bbe88d14a", + "67": "c886da7473d0eec1a9a46995561f1272e329b5d5947870cacb4c6fa1a95b75c0", + "68": "36365b63703301716c02dff9776a73327b66a589c564680d4a9fa6ae756921f5", + "69": "7e7d07c6a2174cf5f143b9460d356078b59a1c2bfaad2816526e6e7e027a5fc5", + "70": "a968bedc07d4606aea97f3e5fb98c028fa483120411b2880ae4084717e6fa1d1", + "71": "aa7abfc5843a387b9a8db3b66f14197450663ceda3fb707d116adfb03fc66160", + "72": "7b4df6fe5dec7a17c9381af64254f8a0088ae044a3cda94f855850b0b4e377c9", + "73": "ee3573d12ebc5db22e013441308050fe25ddd67c2638c3dd0965584192bc8a54", + "74": "74c2be2490af387a0fa72d54e59be8d0e0612dd6697d60f36a187cefc7590f69", + "75": "451822b3c168c336ad47effc25339c80bd01732ab363534b805bac41ee2512fe", + "76": "d9425a826e3800c2a002cfa85af1ab0e7b04a9096ca0b143be9f271071a5b678", + "77": "c6a0ad6571b63fdc4d3813ce8c055adbacfa270146ba1a16d970f2588e521795", + "78": "55be8e640be772888ab05df4dc5d1448e1c86c3aa464a35d8949f7daae6b6c74", + "79": "be2c7ce9735b6108f6514df5d9b46066513322d466a702eead6e0e6d75195432", + "80": "108b4fe53641443aab680ea967b6d9033d1dba648ebce8539c13b90ab12d0660", + "81": "f8e337a98ba9016be6b037a811024d8d4e9b003ebbd1a469f5f0f13bfcf65a4b", + "82": "7397422c775f98bfd8adf397b9b208022be3679173ed7a7844aceeeb8eb706c7", + "83": "80bbce0a08c20370b680c2477626999d42b8b482133f395836461f24962e4ff8", + "84": "e9c53213de06ae41d89ea845e328e5a261b608beb39a2cb0ece7952cc5990006", + "85": "de4a5e5b8b6ac112a0be67fa0d14dfbabfe736a75db19743e6b65e76a60cc819", + "86": "aaacf3a2a41c8e7a82d0080a93750f3d411d749289762e90ab6d55dfd41dd5a6", + "87": "7dc9a37795a1c335af42533d46e91125e71cc659f49a8b8a205d6ee0665c420b", + "88": "d72a22a59d92694919329e3042a24a66adab7fa8ed6c3219cdf500f3bf152804", + "89": "6cc4e38fa46bece56f3c8879620e831a737bb046c411daa9620ce246e74e06e1", + "90": "3831e340f8777324b2736cc6992a0bf59cf6376a7c75db004c8cf57f813d5ea3", + "91": "66bfe825f762428c6a1d94250f1884f505bdf90db98159627361b8f11ed7eb1f", + "92": "90848c4e2a4bb7381a59a509c1a9bf7accbe0fe96b8f9a1decefea740a78a380", + "93": "d87227efc436f019987c99dc20944f687426abe57520537d2e20685cc4b6daa0", + "94": "fff989caf13b38fee421a173d56493824e545a9a832a29aee9d2d2e281395d46", + "95": "f9345f8fd901bddc7e96b6e4b06bc4ece72e595c5b5a7d322bb15623ce515b08", + "96": "8fac13d0eb5bd0d2a8384a3c1b96ffbebd4c39f1307d267ecb6d271f0696ccb3", + "97": "b0b2fa84bdf8e74116f8a930c56b3a577b2e734de0f6d0ba2d6cce555a6a67d5", + "98": "986b473854db95d4d4d2837f8b7224e1f5795ebb8164dc6311386ce793b45c0b", + "99": "40f5518622ec2ff9bb2be7e815d38e5e6a752cb17830901a742b8a2b58ac4bc7", + "100": "0bde10f02019769b061c22df7489e837d40592db7a5871fd5b627eb7fa5ca52e", + "101": "40c88d84417437363a88f31861822d5f8710587f454e272dffa8950888d087fc", + "102": "8b1c923208c6e9697b7cfdc93fbc8e31d8e4ce9b8ea2599f1e3b2007823dee46", + "103": "00b9dbef3728a69965d8afb02e1c18578dbeb3434202d2c36d4507e96b611df1", + "104": "da44e8b19b68a4236efac2bcd73231405aac4d5076d1b4a1bb19959b8b962043", + "105": "42069126a22639d4f3915db7cfa09e060346b26e29bdd30e97efa6d5ea74eed4", + "106": "49eab9c64a5adca38646eb1107555a12d5979db301e2beb99214d28bc0a7c4b9", + "107": "f7729add8f68998e4d3e6fa92a11a1125d91c940cc3043bdd11e13fc0b2a0e25", + "108": "1ee18b759a301be7c8d0d3d0084bab870dbca8742a23b2b9a5f30625d98575d0", + "109": "c62388da4a09a4ce7ab4a6b5518663c173b9459f3bc54a3433f22ab230162f03", + "110": "0e3b9b2c1054c0fcf921036099019d250c93b811e2dad583318eed2c4d774233", + "111": "4b822a54461726057f1909360d868da85fa617233f9c861aa67b281bcbdc1268", + "112": "46de6ddd950257c73f80571664eb9391c5671993469899afb62a3d36914b71d2", + "113": "7c567e5cf469b35c2760f5d469a856f6d02a5b632218ffbc192976d68d2b03d9", + "114": "f17ed39bff5dd67d2fadf41caf9b07ca442203151262dfd9562784fd1d87a6d7", + "115": "bea1bf740758fd8d4867cba754e43c1e7ea77c4174394cbbec4811762044e596", + "116": "551acc1d4ec14589354e463e5c03b63d8f5f1ac34a0d2b6525eb1024a717f934", + "117": "e50f5663bb549588ff0eb90e59b4e88b90c4a25024eaac8cdd07d51152962123", + "118": "908363fe532586f2edb297ba9cbec78d3f8c3f099a3e45ab263b4acbe5a797b5", + "119": "519c7e57d399346600e2a61d23aa061a40849c1a46645f0d158970e1251f8e83", + "120": "d4d7271c33ddddd71d0ea2abe19aa3be4f900aab74406bc99f8aac66d2358ad0", + "121": "aae73fde0e5278c5eb91d45c5f3131424a5131a544779ea4c93a2e75d41be8ea", + "122": "f0f1f90dd4fc6e03e686c46a75cf955eff1bf09479bb633a0626591696c2b810", + "123": "3878b31a0e4251e376dd1a691a757ad2013126910f529e39ad655370044ad966", + "124": "52e0621651326bc457ed6310512b059a255e828f631638e33d2ce1a7fccc6de3", + "125": "d2012d174e01672a51e8987dd1e94f4e250a6ed36dcc543b9837f8ed4fddf109", + "126": "3528481f1559837e4834d6d90136a296e811d9ee5c39ae326617f60cea41f4e2", + "127": "014fadcafed441c9b7e5bf104166472ef9d0af89300109ab3f36d851ee15ad12", + "128": "62b79fde18509e4037e82de67e813b510e5269abe45c119c35ea612324e3513f", + "129": "8fb194ef8564c540db41f3f28be01fa3e913305499914bb4b0881280e626eddc", + "130": "aecec1ecffd31e06214ba06dcd4b127108ae1290577934f7dc682c17b712e85a", + "131": "54e6c3dfd71b2d5275f4ae24697a58ba435a569dcf6e523903a4134d09348205", + "132": "14523997e9fb53f7f4f930351da324b847bef3856aaa36b7cbad129bd920ab52", + "133": "2a8794858bdfc8a22717eac6a33cecdc452b6999a5b5d3668b0f64eff6fa6f7d", + "134": "bbd5c84977bc666d4184b50215edc1dea03fe0dbaeb11eec1e4dd1cbc0e5e8c0", + "135": "9c40e605904ee532500003e4e67d3934280ed087dd862c2c9adaf3d23833c25d", + "136": "a1d732b7b85d79812eb19a51e215d098cbb5c6919dfc8d51744258a8181bcc3b", + "137": "764d7f94d5d0663fb073a13bd7de749c7bb3d75294ea254fa8292a2fcd30ff05", + "138": "9dd61ed6903db6bea9e620ecd5abd14f04acb4f15bfe1868a6cf6658b71cc32f", + "139": "d0feb548240390fda51b533fcd158cf804930b90067a2370689f2a29af6c3b59", + "140": "b62811809551465c0a56aa63eb2e035865cda21f0a3e3c7c88dde9407489d69d", + "141": "60c9f735f96758be99b92f9d2a878c8d5a91f9bb4b5322c7d58690c8e5df8038", + "142": "75f60bbfea15a78cbb38fbc370fd82e00809c71bc604d1f6e75b3bdb51961aba", + "143": "0b4d8b4cf4011bd985890265a371de6cf1c5a502f1e1226b0c6b9df113ab62f1", + "144": "cc1bfd692966f42589e5706d8133346bda3b732d501f26c1ff3dd867cc83dbcb", + "145": "14dce3d80ba7e8d809872283379dd5dbf36f98b325a8844e243cfc82add445ac", + "146": "025490f1a29c0e12016cbe47cbc793b60f1777fc6bf66f19cc5de560cd8f241a", + "147": "325f0571066018da806179c095918853244846ca10a4f2f1ee4b716738856c8d", + "148": "fd0c5e9ab83d070ffb448912dcf9204e49bc381d523dd5364da2b5d5d869b54a", + "149": "f0a9d4e3d76d088d4aefa5a9775fe71e87ca9449d8895db4448edcbd324f7dde", + "150": "2e37a1ec71441d35e7c23cafe78831fa3d4aa13cf3072a9ef3a4883a570d6c0b", + "151": "c3f3a8cbf0f275c49c906a2f4b8667cce3aa36969668443453c7fab5e33e2ab3", + "152": "b822203e55dd6c44099b62eb8406d5cc353660342a3f972b52f4512bcc41ef00", + "153": "bb47e91e0e9d824a462539909789800c55cb6efbee300ea450bc218210e5a7ab", + "154": "791cc14c36efd43ee45c40ca48807e9c8fadff6eb208ac64edce2c81b27f60d8", + "155": "6012ce5dd6970759e783825054d0ab397bff2cfe1ef769735824e95d4db207c5", + "156": "eea22bb91865edc6e576d89c8113ad8cfcef6c6c2356e5006f08a47f4b5221b7", + "157": "614865796ffafe730b9ce899b89473f5a2a166e8e7b0a0a9b725187892a017e1", + "158": "90b4a8201111b6478212f9a446f499be7b714b0a68ec7cd6c42923dc4cbb5fbc", + "159": "3fbe7c0d023b62fc75eb133cb2bdf314688802c343ef0901f9fa544d7a61a13f", + "160": "162047b4f5dafcf15cb35b9cb946ecd8700416eb875390296bfd4abb6f928016", + "161": "56eb8bc82777b2a2097d25686dadec52fb4f02001b2425d9b6eb9e457de4d937", + "162": "e77bc83cd4881b55c072e6d99726221c73ad8246d0a476f3c8fd29d8d5c42370", + "163": "a1da3d6d2b062d09ce4dc3f5c82da888332734cf9045dc0157f01bc888bf4a89", + "164": "c18abe1fe7bbc4d3066f6c9ed9d2971619a10c49bf93f4986f4ae182ed06e834", + "165": "05a5ffb78b1ec8ff591bdfe0a863b6b46c9c3735bc5e389dd20293761cea7118", + "166": "81e8319ca08c1b75eacd79b6c625a8a3e46027d9bd7d76ea2e17763b9863912d", + "167": "f160b7b7dca6ea77ff522c845f6989a752b630544b4340095304022f8d2077b5", + "168": "1e4d327af3a660d5ee600a8b99644329ebd4318307e991aaeeb4c5481b121392", + "169": "ea49be5f3d80ae41fea70183485ec344aa416e2b8781a0f836c85916158726e5", + "170": "ef20e371d42d432159ad73115990a7678979f857f85aeb9d58c10f5f47386dae", + "171": "0b3d001626287771eaf96ad493333c99e5abd3bcaeae45ac7e578efd765d65d2", + "172": "b084a76278b73bb6b108f23411c82f56305899185db469eca423fee3c05bab71", + "173": "453d822f09abca3e5d8aa29dd5a0157e0aced4eb32bc1d1c01376b95af5870ce", + "174": "13bbb290b069b2e5fe231d72fa6555b80717f5ef59b3d4eae881a5248a8018d7", + "175": "9aea8ba5660e133236af8d6a1369e7c1b963b26d5f473b4e3659df2aa52a4ae5", + "176": "2b6afc54bfb01367f97bed45029244817401838da1113387cd300badd1adb998", + "177": "b409a32f407b7209a7a5e9534cd7652127d270633a0ecf4b47503580d6d8f1f3", + "178": "234edc53c2921759d742f95cd2f0420aee8b037da76dd0a1a7c5139be1760f7b", + "179": "2dacbf26785eeb4d09eb8498c20a4d32451d8a210d40c08dc0f4d0a7db30690d", + "180": "f4ad33d656c23fced6110541e41a821f37cf1b87ef9b094abecad4240a9598bc", + "181": "88e9f34ec98605f372e6c36a5a3d64bf86a10f57ad9de744b007a1bbb8e3d7f2", + "182": "56cd99e67ebccb55c178bc5f1a483e73518dc964d4fa113fcf0e85b402ab93cb", + "183": "23050711d7f297b108c1c3f3500c5395e78ec773cc01f638c7395107b5f456aa", + "184": "f0b388bfcef455a538ddbe03039a1c41bc0bc47125849632ddb9c6fa475fb5f0", + "185": "aca4fea116191ba48ae16dc9961f92a033a37c3d688aba86b0217d8ea5abf384", + "186": "d939efe2a19c20cd3276f64f9c21d79d35bc0f703925b5a7c70c079865a9daf3", + "187": "6fc0a044a20dfc02de84aa8e79fcb73feb834a6f115353aaff9f8479f9457ca4", + "188": "3cc2a34c20bc73826407fa01d6ab22841f28233fe83a1ac5a86753e8af7d1a8a", + "189": "8ce399f3cf8ecc78be743c0ebc8279e997e92d1dcf4798caa1266ae6f38e16a1", + "190": "c1c8a293824b06b0715b2df6812cc62d65085f9bc0ae0c9e4d033537feac7885", + "191": "b1a3eff25841525e205f23b857d8708ea890f371a8c4356d21b2b73aaa5cf67a", + "192": "717a870c4081c501168567674bd0bdbcd0d4929b2d2f1d16e5c604d75095c834", + "193": "9a423e66e3330ba56b690e40f5e12b9b1b01500e1a81f028141d4a0c7e36c028", + "194": "3b7d8f5145e25a6d4d91412237cf03d67f60f6a389d421f48746ce520670435e", + "195": "87e80d6e28b0ef70c4ab91029183e7bfa6718016a662ce097da2cc57a6d803af", + "196": "c8c089f5c9df3375c7fc67ed8162cd6f37392f7f0cf9b53bedbb7981fc7db90b", + "197": "6773f01a5b222ff81c9e0305efd0fec820e4525051c3cd40de8a8027075da82a", + "198": "e352919f9c554b7e8674ac843aca703119ddac4090f99115b356e606ff5878f7", + "199": "a6bf516e8820134d9a99c4abb5f671b316dabf832c1e031e5055e4af4dbf7688", + "200": "ed827f9f37da4426b17958830cb7a79dcd4c208254ab91596fd57ef7db33316e", + "201": "3b9b49579a1f6af53813a0a49f15fa21d68b88245b9f525794dbad73b9730ee8", + "202": "32036b42b9fdbe3c4db38b08208126142c7d7c1295d7c9846b9a0883c724f932", + "203": "a4f686d028810afaca574368082c3e7528900f343ff2844831500255c93a965c", + "204": "37aa78cb44eb5df0eecb1422755b60eca39302e21a55c4fb296107578d6ac20a", + "205": "d24bb13c9efa0d057693ca82d86ff852a41210ff024e89b77f611388284db02c", + "206": "cd68c8c9239692639c7f6b857a9341f93e17742c4d5f8dce377e1c1fff1726b7", + "207": "fe7450cc494a95f8cdc0336a6a38003d6c28b165c9bdf2dfdc94c9508045b6b3", + "208": "53acef13a43b3d798506d1c4e078ea06e045aa99c3fe56641827fa68ff0130df", + "209": "d6051cdae9b8933ca6f898fa78c0c1a9edef4c7cabf5030fb5fe0a94baceced9", + "210": "8e6ab3541c62973a47206d8527b07abf30ec4abf41e15905381488f4aa9ed3aa", + "211": "c59b8fa0d1d23b3fe0c7b49d247a6ffad1972aa62929611cf7600236e69a9328", + "212": "ac9f2b4152b173f2d9749b514537b93bfe4c9bff6deb73c77af00f1f7205dddc", + "213": "849918befc4d05840f46305f1be2f3188ed63032c6a96cd39d6f9dab607f5271", + "214": "7774ba549b628c86bb51444b8b534a7955ac5b962bf2810ef9b6d690e79f1d9d", + "215": "5cceae2e58190766ad007420d80265ea5b903cc279212fcc88c45c9cd4a155bf", + "216": "e30cbf99979a82e258e10e55640a1d775ccc86386cf2e9ff8da548d3ed49108b", + "217": "e7da17bdf9fb1e0affd78aae2ecd740e46e69e0509b058fdc5ab10cb0a33b3e0", + "218": "5171b9ba73c901468f7dcc83e8ccc7f44614e21058fc9e77b0304aa34e28289d", + "219": "d4e459909dc4151404e7bc502dd3520d78029f3f02687e5333d28849aaaf1c52", + "220": "1e49a09081e2a5b76c549d03a42efc02b5c3cb3b6dda32188de43bda8ea0ec20", + "221": "8cf4a56601e905961ce43b47f5f7b73064cc7af4fffbe591a8d32966004e28ed", + "222": "29023155375769042619dafd6e316020db416de6bbc22908792659a4a68ecb8e", + "223": "28e2c016373caa8cf5b20c79db4693b235924c569488d18a156883ba11df8e28", + "224": "1659fd9f742ab3cea2e07b95b69268d04547cb3c41dd77d21525e4a60ac9eea5", + "225": "80602d8f8774a0f3176259504bd0bbc4eac46346eb2199195e8c58cc89320bb8", + "226": "ef92e2403039d3996f9ac167243976203d50c28a691a8a5a22375f77f78f3f51", + "227": "d1fabc9e037e0c8b020ed6c53b274eff4848c680eda65611ee15ab21f22f3ea4", + "228": "7b2419ccc450b879848fbff4a9009d1cc2b764e124b1a325ae6c707c323ed25f", + "229": "e7ae1c24b2ad2d23d82d2c3a09df8448b713d94e026dff30c86362358b20bfce", + "230": "67f771ead03bd7d08caaf5e02c2b560d24f37e4711f685450c00bf2fc5f278c0", + "231": "740bdc7dc00ebd04d723177757ebcf513a70b89195f3a2fd977c7e67eabce0bb", + "232": "9d1502b7419ad1622a075228e619610fe29e1fb4554df7e82ef587fde1f696d2", + "233": "72b5e73247f4c98c7961bbe84d8b592b77d22440efa4eb83abbd9a46ec7affa8", + "234": "f7719508c09add0e2de95b1f90b4576af79fb31e0b4474ebedd76b67eed4a1f3", + "235": "c3590c0941c1dc1f56b40af75520e16db68fc22ba7ac81b7366ebae0bae87f25", + "236": "b5b74378af634fd4784960d7f77ffa02d4b0b1e2af268af9f9fecc33ab101b50", + "237": "c389e01f7ab49d9895f67be52c2745b4a76d331c04e9910ddcfd13f4643da666", + "238": "ad5387be97d8c3bde7a5a8b762b5d50adf75cc8f09465620b234375e2b15f698", + "239": "d4133e9104f78b872d5eab6a550b53de32c1964be92eb0e28aeca7c934b5a3dc", + "240": "73984f7399fda87116808eb209fb04ee6c1f98f093018c20ecf626979e4de4fe", + "241": "f09b50c809b82f8db69eb82d3e5702c8bd1be4c41378ea0c15652eb005e34952", + "242": "a7b88fb7ac600bd71eef11b547420c596f79cc21958a495c8addaf12dd9da937", + "243": "0a240733d8e934286c223f48bca9887d92e43412274a9e543336f2c31f0cf92d", + "244": "fbc5c6cd4de0d72a09e40fcb87d2c3eb2fd4e4e2b8171ff8d6989450b390c890", + "245": "c94291ada787b4a84840261e9eff6e54a5f2f465b1f06f166c05f55f3822391d", + "246": "6ed6614dff0c2a12e993bb1a84c0dbd729e6214af27e11ffbebdca2d311e5b8d", + "247": "5525c1a846c847602d97ec296f877907c4ccf136b9ca697c63d9ccb4d4ceb402", + "248": "4dc1257fdbd1c619c73356dc97f87568d27aee76d2c355a56d62ed3b0e281e7a", + "249": "6584886f4f64b14797ffdc8cea5372eb7d0bc55f2910faf81d04830e592c42f8", + "250": "ff858e6774adfee0d6b89c5ddb92d91bbb4ed83415f620f6692d0b4578ea531a", + "251": "30e3eea5603f80b91bb5c6e6f22dba763354fd80ab7f2da377bd8bf0f588e443", + "252": "c4675339d7004c607c2d2b6cdc08a2150d3dbe88b56fe8e5c49c8d8818392ed0", + "253": "dc0b3e8cbc34c1c22b6f9e961ea4044b44b84936721da5363af7c5e191327716", + "254": "9ef1124449ea740c0d3fa1d00358d8a2cd80ceb960abd3f1d87c71b2eb2ab50c", + "255": "81c4896aee2823c8b81deb2fa92cedc5275f07bdee3bba38a58a655dedd9e732", + "256": "f8feafc15b41f4997c6e1558a4de99c766ee300e36de441162173ff25a46b626", + "257": "a69d818d639e46e1feef1499904f3a8ee7e0d43e497fc7303c7c61b2aa3cb004", + "258": "5668541a0810aaeebe5b51142c8e5cd7d7aacf67b3223fea9d9d8168f1754dcc", + "259": "c3b77fee20e5f6c41bd3ae13ee38242faa8dc46a6ba96ec3073349ca84d3b6ab", + "260": "7b79ecc97cb9a3f6ef345f7723d328039efee8287c82a3371868e7a038970522", + "261": "9ed7c569d5878d0eed13fb2eb3f00f9b11e4e2af069665359fc5c91e3ca954e1", + "262": "c59694c4d7ae6cebcc1e79a88bb2deec50ef119f8a7020df1da80fd13e0be5d1", + "263": "ad23fb8e31a4b13164f31c7d3d5ae24295a408295c5d5682df552d1d4f290781", + "264": "7ee1eabfd56a31accfcbdd9e942db78eab4a0170a18b197b28523a12698a104d", + "265": "47bf338d6c19a13e8fc90b379ada0a1b11bd6041e91e666242ae3b44897bc3bd", + "266": "3faa7fb1ec7a4322c3a71bb609869583ff5d4db88235c337540d7aca2511997d", + "267": "f8b315b365c3cc52d8a28c6ad3a09b9cdf60994bfe3c757ef572c5e50fd4bd9b", + "268": "c48c70f12fc5893500534e91f66b2dbba622be7135274d2fee55a0a9a62caed6", + "269": "4014d683bc9d4490f6c072819ec9d64dc4b72c65e180f98590e439b433752363", + "270": "62fbf2f938a29a34b89f8dd1feb912857a078a7cc47079c06f5845d1d34d2fb5", + "271": "11b8d9e42b7a4f4200d398fd780f993e01bc0b494700f9ed47a442b16900c85e", + "272": "93cdaa3d686cd52dcdd7d3de16987c11d7bd6e7f700b4caee7c3e8a11cda1852", + "273": "18d56ab7eb43dfb182dcac1d3393214958bc6bab591d83d9b619227da52acbf1", + "274": "84f31d26525b8c22d33774e2b18447dbe4d1b7a9e7f122a78dc7cad30b6cd222", + "275": "2e0619f4ebb9246bf1142e96f878d7ac070905e57e51241be3bba177812183d5", + "276": "71d3ab4886014b7a899c6f7cf693e062f169164b33314d3ddc95a9434a614d45", + "277": "04e6900f691305463ba8708c01aeecde3359739801a61a2e1953107b96ff7c1a", + "278": "0989d2a116dae74aae9d336d91b96e22f1886736b3e175ed094b6769b8bd3fd2", + "279": "ad7922d08cd4dd01ece9d536a6d60439d9decadd8325adffadc48784d3fa199d", + "280": "7655c4c69212c99c25e12aeb83adbc480a1a29ea58805547a5d879631ee81d11", + "281": "7e59cedc7870dcaf0c264ab676844ec1c0552a925e33e6047da2d9a121eee290", + "282": "0f60d8b5c8b26dc30c52703bc485c2c18d3dea1d07eed8030f1bbf17dad8ff4b", + "283": "f1851fee060c6c4bae0630f2ff01433acf4979f14112083c53123da5d637ff13", + "284": "1a65326c58b52cbe9b25ec20ae273be56e052f2ace2ac8f536212a9c0a851a67", + "285": "e765aa5608bce37cded3539a3867b1a00d9064029d4cfe1072b6a3dec73e95c8", + "286": "746c07f1abafc6eb855de201ef452fe82d5a849ac65f3151e6ed385f4fd2a0b0", + "287": "cae3254f6da8d967527fec84467706ded9ac5fbd15cd2268e1e442292e8f386d", + "288": "bcfff4da9300196ce23578b38140e823d6efe069e6acb75d0cf25d793d07bf16", + "289": "20029eca7caad88ff72227f9b6ca0910d6221b2dd26b6092f006fca2c7fc5f8a", + "290": "bb00e2ac20bffdfa25210a4a15bb8864b5094199e1ba95933b8337a0e335750a", + "291": "06459d97130b75bd2b14725ee1c3eee5aa5bb878b71037794b29e95adbe9e65c", + "292": "bdc378689bef5893573128ccf4866906c816a08ef0333572935a711ef2453e84", + "293": "5cfab9968f109f8825e59965287475a5d45a49b9a46ca6fdef4dc87aa2aac826", + "294": "455a7ef5bf85ba59588cd734229e958a1a922b35b6316f476a386787cf6dd517", + "295": "3aef8f1573ea615dea36ff114274013deb62eb35a1ea9951afedf78794ed35e4", + "296": "01c45ea2bf8fbd730d8c10ee7661c52be362bbf0be0581d4639e302573d5b04d", + "297": "d5bb6b49b1cdb77cf2f83444d340fdba39090b8674b87ea801fb9784e78e0c1c", + "298": "68aae62991691e927874ba754d01e5a86ddf164e56804bbad725a20e0285f688", + "299": "2efc54f0431caa3123f8f9476fab3c092f974f921fcb1603a9373ca8bf202624", + "300": "24fa7ebc70ea1482a564a6a7064e2e40ce1931cd47ebcc73bf917b103ef05e9c", + "301": "3cce4dc1de18c8815858fdfe0767c512c75c6db2612390a98f3d11b10c9aca86", + "302": "e461e50d471036ade6391805ba3663c331d64dc2db3398119f29936e15112da0", + "303": "f84d13c96f9771d483d7702b54e91eb89ebfb714fe06cfc201e34be639b0f453", + "304": "eba4fede8c52e035d0e22794f727bea4a66d70a137bd1e5337324e6f21982964", + "305": "c747013d0ee26476e950ef0f7891dfbcbf02ee688df0c65ade7d1f3ba49b8157", + "306": "f500d24324250182d6e6828717c3affb1c767e05e93b513cb61c97c90fedb7d9", + "307": "c59bb0a65878b6863bf2a8130f8d95d289c6b29856df1a9125df7c623c21495d", + "308": "296866deaa33e450dfad522875c27982615e6fd27003d63f89296a71db1337b2", + "309": "9e3cfa9e03cfd6431b266d02b2ebd7f363f75bda6a24f9a4172291c44216abba", + "310": "5528090a1ed691d15753967bd44901d8ea621e38e99edcdc0cbe2fa5dc042604", + "311": "fcfd5a757cfa92096b7f7282c2ddd44a37af231076ccbcc7ee6b6a49e29c7f4d", + "312": "47d61696592db62aaaa4e104b34c4f455b7caa46b594afda38284a14acff5737", + "313": "36a484c390993df545fe114f743d90723a3b049e9b0c70d50da71c22fc4dd13f", + "314": "614ce4767dcc31452d54c5cce5e09e727ac38e381bcd49131c77f728900d8d53", + "315": "dd6363fbd03b6577ebaabebd394f4a077e26b6b17427c42445133d0e17951cc2", + "316": "62493029b55a7ba85a2533e3d1c1574351e841bd8d5561e60bec4e19f72094fa", + "317": "a30a544444bae9c46600cb3592c826e49d0abf48a038b08ce7fb44cccb2e60a9", + "318": "43240bbf104ed0c19a0dd8bb4e57e01b31fa67d2553d16b572c77ca548a00b3e", + "319": "a710dba6a1d7a5588c02ef2beae1fbed6a93b77ad18f4c78c0cd4f97a7fa58e5", + "320": "2f1052d96df847f0b1a8037d66cbae0e6e8d2d0fe0793da3af3f5fd2c726172f", + "321": "a94a3cd2d5f765f7456cb252c75c3310240091522d209cd1f2898327f3136fcc", + "322": "6a8800d665df0c6fb50eb4d5e9d0d39c0d4403687fab3b1f5ee901aaffc87463", + "323": "7db8d4dd565ec46bda136dfa3ddd2a088c7ff7ad24019ffa3e4c4529830c7e2f", + "324": "fc7d6a7dcf547a130314c6280b1753c440f2f2f669b014f2c5400d535c84aad6", + "325": "36fa0aa7e966dd2589beda5d52aea9fb177cd437b2d07eaafc9bc9f08a1d245e", + "326": "789a876b2edbb96f5ae6d5a7c0cb22676e6bed8e580995d616a88bfcc603f2a6", + "327": "c093f633c63526706b95d029a53465676819f13a2ed0beaf921c20453d61bbf6", + "328": "1705d1bb0c4bcd00c219c1a59a56613cc57f453174152ee3ee713c2004559513", + "329": "c59580eafbe5b36041471c8dda078d503657855a0f692da46572848abcd9fa0e", + "330": "400b05bf1f2c27afba0259179b59baeb7e62bc6f198d6038156ea35d59e677fe", + "331": "4b3255e262d50d5a1a9e7db4002baec9d44239485e5d2849e040cffbf503a5b5", + "332": "06fa8bd3e195182eee495a9e3dd9e1a96fd36173b9efa883d7d2bd97016b3aae", + "333": "89e7705ab532a6aa0a650343dbb607a3b8410ff748c6f261046fa36ecc638325", + "334": "18e2bbeb5654290354d890d8e6a06d1c431619cfbb291f814df41c3c69df8cb7", + "335": "ec3320c1efc54a1c33510edc542e9daa302d6151c6b4ea4f86cfc73472cb6957", + "336": "ac2ab37a72126fb0dc5407f89c6c723070a30f2730597151f145f0b80071942d", + "337": "71cacafccaf27181f2aab001cafa2fcd78c26d840dde3d78af5b78a00c15e4ca", + "338": "71c4452968d7c7a797cb9e6f0b4710d302d33b26d24848b1790933c910c22898", + "339": "df8eeaf2621aafd14eea870854170ef0d48232a6fe9de399f7b50ca2dd1a13f8", + "340": "fddbe7fa339995ab313b3040c117b61958fa9d512c0a192502e87c94e88c8a3c", + "341": "e4dbd9457c13948692c9219ab0bedf98c8c3077ed6e5fbb262cd7ba96cf892f2", + "342": "62c2244f52b843e9286e8c13163c528910f95ff27eee2631e76b99273a4c62d1", + "343": "92bb1e940a7534332744115611cbacb5f48f7f13cd0b639d69d582e943f7090a", + "344": "25968d43d4f105fa7d0aab34f5841185a7dde4b688b6bf24d616f1e6ef7614fb", + "345": "61d15fcdbd2971e8455f4f934d276a6d29e79d94916a0a0d82aacae7c20f39f5", + "346": "036afac1ad65203a885b66c33d569ca010a5e9c8d601dfe98062226b35d11784", + "347": "5f8a264fb49d2bd9fed4e0369f8945ce9e1d0675abd2e5d29ed420ebc96a0426", + "348": "1343e6b302741a3848402d64db7541e496f75f9aab0cb80dcb5fd9436d9a5005", + "349": "6a8e2ddfda0e8ccb350f964e67f830c23aaf00aee793c03c4d9c67de5e189f7d", + "350": "a05bb5d743900a1d2d25edc778c19500ef5b03bc2d4f8b6c779aee5849e2136c", + "351": "f8b5b47d746d7ec9108d53167e3425ca87e0813e431c88acd1e96d7a32347381", + "352": "12aba5dcf48b074607a54c0945a8e69b57628561134f83c92824456a954113a6", + "353": "ad346d70d9e3ed4d191171176ff72a8e6e41642013691fb4e6f931d60b7f14c8", + "354": "08a7a687a104c35c4cba89c69d86fb8bff91195edd5e46da3a44283a53a0e12c", + "355": "719694ae81f14b5da387dadae1c69630b37bc6e5d4dfb6af1a5f0cdbd48d33e4", + "356": "8c5df136426880e4659b9242f1b652cee8379e82211dec27079e6e28360087b6", + "357": "2e5815b3a8e4eb0f2a06a9c33a8db4c5c73ab7f522afc6865cb222bfc46e1e27", + "358": "0ee1dbc94b18022a5e54ba987763602d1a7f8098b42c590335b34c19eb46bc05", + "359": "1377164ee568fe8eaa342fcd3419ebece68612492baa6924f69af2c2c9043ded", + "360": "6fab3bf520164670f958c267e4cba26ee922673978e033e75aefbe40d5fd6d50", + "361": "4d76ee9a0bc95ee8feff8980ff568003463f8687d1659acee7134833b7cf965b", + "362": "16add087f7a1388bd92dbc3b253b1e35afa3cfb288cd6d334852ef1a6fb9e364", + "363": "f46acae63d1686055d615aff171ac8d4b034777bba73890dc3f001c1927d1e3c", + "364": "f94a5503462a21dec1202325c3acbd7962fdba6336526283591678fcd0667c2b", + "365": "0d69e3aae1851f16024fe6e6256310f5197a4a57a683e968c4b2e390a87f5ed2", + "366": "60ec9e16534efdc33c414d5ad21dcbe250240276a545c06a9104730893c522cf", + "367": "2f8bcc6745e287a990587136dbcea6b8c2d7101536926eea2fb9ff02521408d8", + "368": "081a900825b2981a97750669b49b7ce396e45061de9d377ec9b997b5992944de", + "369": "4a03c548acfc34697817aec8b5cda77adeb9d8645acba71e48e19d4c5e984198", + "370": "2159631d426db8a8b562e9028b7e92862eb66771750d0b3ba5aba03130a12f5d", + "371": "a4f4c51427274d4359245be7d5f231e7c5d33ef98675f2e66be1f669a5eca662", + "372": "d025ed55b09229a96dbd49a88ea73dde5dec17f911cadf0939dceb44db846d06", + "373": "5ec683a4988aa487a9a078a45c578743a86805c5bca0e82500a3bee52e71caf7", + "374": "17f618a86256065b1e9111576c9d8602a01e50d75f04c98b971156111e2f230d", + "375": "7a616089586d0a2fbb5a8a196705e2594973acf5f2b1c96992eabda383d93175", + "376": "8b5de716ede9809e5190fff6ad72e085620eb1b4c7307de276bb85c0ad22ead3", + "377": "7d6bae91f814fa61d8ff60df367861436d56067aeceb7e9fb7fd3b924cde2ec6", + "378": "8732f4e4139bde6ca577c532b4ac5d732865e71be3a8b579be6d144ea1e34741", + "379": "856c9bcc65d285f02c771cf46a9308db6bd2bf6dcf745d6811be5ad92b48a45f", + "380": "7232a1ced844f2339747c44571ba702e042eb85f7d34cea06ee87d95b1eae084", + "381": "5420ed93ff4bd4f26d7ccfb7108564279f3d7ffe922ed22c0e244908dfdaa4e9", + "382": "c197badf612df9546aa0e830ff98d2ce95691fe9bd2ad47e425f83954a2afee9", + "383": "c7c1b190883bb6eec2131eeda7e402130e30d509704a57eae05971fab90e589d", + "384": "0e52ca6f37cbb76ae4f21de91db507fab67169dad55d74bee6efbcd442376f80", + "385": "b4f06a565aa91d4dc54ecacae1c06fbc22897dce65a884ed4f067bd55dbd0973", + "386": "cc5e7297389d59ac069f020232c69304ddf11099032afc7382446248a0b35419", + "387": "2904a1370488994341f28bacb7fcccd43f671d0982f6c82e64646bc2831844fe", + "388": "a7714e98739943228b1d468e8b7381cfc71da9c52215294299dd4c4543455c79", + "389": "1aa859abb42035b1f73d32dcc31ff9d4190dee8c502a404ee6a2cbace80ae068", + "390": "922c4426c755fa3c744606ebe687ce9ab876e7d56a9c9bb8d98eb80fad5b01ec", + "391": "072b7d6a48ba5f26634859c19cb7b0e3c1beb03bcedd55d3e98b66539dc64424", + "392": "889319e9e5d4cce289f7758dd6c151113515b4af3442b21217440c037e377e7f", + "393": "72964ff274996331371e854b965b695380b8f0d55fe25c45b7b81a2c41ab8f35", + "394": "0f2427301a17cf35d8590e585523ad1dc280cc6cc0879ee8b014de0a298172a4", + "395": "c20b11b7f5fa46e857c8a1adaf3b6eb6a033f4351b581814d663c7fb636ff5e1", + "396": "27f0853c2ae75f4c1fc6407f157b81998b4bd61e72bed756e0296051a523dbb1", + "397": "fe9352ea1835a01ce5681107274ae81dbebbac2e50a455b137bea38584cc5132", + "398": "31a3229790f5d5beba05470fdee96e05d3bc4feccfeb4cd137d54a2c69bed24e", + "399": "87a40982b7f82b7206613e15900896d892a19c2758c9c53ed2c278377fa9d95c", + "400": "aa6863e121d23251c286444c9038b83a7aa6308800e194cada0585eb241bff52", + "401": "7027e90e538f9b4ca4f57740422ac59fd692fcf3f07ea983ab3b27403c1940d1", + "402": "a14530a1c34e32805c9293c96ef52872aea0b632705620367ab2bec35a221ee1", + "403": "723ffbb44bf9e76b630d8fca639cf0cd156e0e08dff4c3af1addc98f120ee2a4", + "404": "df8998ff773be14d58531c8aa73e41dbac521e943c75654fb20592a4bfd1efcc", + "405": "b20c1966357838b0402601dfb9f3108b99224ea9aebf3634a50f92c7d23c2367", + "406": "17d9048c1f45bf517a9812ef55f2cb7a8072f760729d3b09045abd21a4e63d89", + "407": "6c0996568e6408af6e9fae7cbbe1afc9d29cde1edce7815badedcc7e790b2d5b", + "408": "9566f442afc5dc3cc6ab86b0f171a635906c91614b9258d9536609bcf4909816", + "409": "e7121c48933c33093a044c3718d261f8c0defdc17d5cf9900d02c740af7945bd", + "410": "91b13575e122f6bae0e6565d7efb06518721e4d8c0a9449b625e6c3487722808", + "411": "5eeeac58fc807e0db4e69e387f0d1202bdc6cff7f5471f74ee0c228214f06b42", + "412": "6b43837efb06e7bfc8aaedee80461c9cc529e61cb2dc9d9cbf4604b3d020b1a0", + "413": "17537184593b7993712bbb3159d2e36183022f3e4a97fe30b5ac21bdc69b47ca", + "414": "5e1aa16cb8519720277a8e680bb84595238577f86b75e5eb6b29ade41960b276", + "415": "7d2fc2825799c8a1ca9916435cac8a264b06ea76dbbda5e4508760a84c1ecfb4", + "416": "7f1129cae7a582e92f8e026191f7e3d75f3ef67d01057fe7d11fdfc2ca3725f8", + "417": "05a2e7805d043e04f31b3266ac0929d2249a5cd0d3aa8bdaa7adcf63a04e79db", + "418": "3a8bd1894bb0135c03c0c48b440a6d7885e2c0bc9bc180e2390d34708b451c5f", + "419": "3268b74de5121d131019600b27e3fc4551db037ebc34cd7a4926eee4218b740d", + "420": "e15a42654831c932b17a15ce40cacb3764a5c250fb3cbb5089898d4d0a986628", + "421": "dee9b1bb930cb35a34a79e541366e60ac44c0782e56ac08b41f9158494f4ce77", + "422": "7de5e07af859d70b67136a9c384eccb0475e5add6161257b253fe3e9dbeb8b46", + "423": "a7bec0dc2036544177c152176bc9e83e2a277dc4ae8b3d89505fe81a1c1ed86b", + "424": "822429ca5063240be14fab78baddeff2c1f5d5eda510f2ca7d6c2006d7de6cd7", + "425": "5b1cad04de36b11fde8ab84e4d39728c6eb933906ceb021faf8a713a71b10c72", + "426": "a7c1b757d1e0f83329f97406af463ad7736f3b7d10e491b66029f562260439d4", + "427": "54a89391a9e93e8049dd4014d403986f209bfc006f2e5fd8830f960315ca7ec5", + "428": "612ad7ac0c7432fdc607020e91283abbca343178035b03887a0e543f6f00219d", + "429": "52b06ed6026d2569560060cf95b1a7e1afb36bc2a5077839ba59162a7357dc65", + "430": "c1fa5a8a902f3504bec3f75fbc88ca8067fea87a35f1739a9590dcb865eac9d4", + "431": "84cd0a94410459462aac85ecf3c75d5963179350777338d4d6103dfb619103cc", + "432": "39c238e97fb5bd5940a1ecb1735a65145ec033570c6529fbba3a621cc5e398f5", + "433": "32ee952c2a01bea5d617cb50ec0b5ebcdd160e0c79badba03a3f999f3034f6aa", + "434": "2dedd3fbb6c8406d69c55eba0ac09afc4ead98a1628a47aafd967d299dc1c8f0", + "435": "81c39f716cdeb76773094b023332d31d4012ab251bcb45d0ef96ec82729823f0", + "436": "6774b61b05c03db60664ee8aacd12123449eef1f1b29e0b226e2408b786c5964", + "437": "1f8b7f00f625ba9552a57d8f7cedf0af4dee289ebf7c65a077e40cba6dfef31e", + "438": "8f35199bc8345d01e8f0ea184ee1f522e538c4c15d76d513490ca7ec2e704dc4", + "439": "5fdda4d3f8d5d24045abb38f5543590cca9e54f5416dfda58508acfe868c7078", + "440": "c620ec3c835e0ec7d8bbd38f2f2868971cff950f94c163f6a70c4577445f4fe7", + "441": "0ff2264d38a2f1b22ee246f604f14308c0394d12b50c80312c9fb179aaa469f3", + "442": "c097510789ce487e721c05461cb0c28dfa45ca38956d357a6475b88d37613c27", + "443": "41509bbfe80986dc357bc3fd396f2ea99179a07db9989a157acf63adfcccaec5", + "444": "9134f8a0d873e488789405e2de011b841562a3b53e19d2220e05d0c13b239090", + "445": "3ceae8ab68dd859c4d0816c8b9db93d598b675316bd92e4cb279f4e91c7d1cb6", + "446": "0437226b2968ec92b03e991dfc6f7fe4c95ec8dddc1ee6959c4f8e6485aef400", + "447": "35dbaaed4f8d21f6176aadcca47c1d3329ecb64ebe7c52e445147b6bdbc17c22", + "448": "0d429a221b73e30c36c7a40af91f88e92644bcc5b7ce1fb7107af63751f58fcb", + "449": "10f553b0285a0e5060d7449742d02436aac8e1099751e5acc16a5dfb888880d1", + "450": "1bfdd81e0f7c324c50ef359d292d047c569f3c9d7cac6171143d8c2974d07c1f", + "451": "78ce86931a36b71ca4637bdb5ff00514e1b812274265f75228bfadfd1ba1858d", + "452": "93b56ff9115ced70a8fc0f477e35562149753e21fcb9fb8919f139642bb7f424", + "453": "f4b18570a8c8e817ea8df86b8bd27fe8d36b4620402abf16174c938f52e5af1f", + "454": "2e9074d67241257f09bf0b48109912a9bff48143ca093eb82723b081275bda77", + "455": "70f871ed1a3b153cfa0ecddaf5add03d2ab53cc05cedaf4db2c6ce986c82a0b3", + "456": "bf0b1631bdb24695897c6f49783ed1ad23530d39bbb843d8b38cd9503a183ed0", + "457": "e424f4b4e0b721a77f2a693d1bb520a5dcc7ce258ad7836b21c1c9f52a84cebe", + "458": "4c01e96c63956098b4380bc2c351a04962e22a6cf4b97eeb01dcfd81d24fb18e", + "459": "5bb4558776437c9bf288039caf00372b89c2fb9e6bee7e56d33769365b0e1674", + "460": "0d001e9ed40e596768cf8d4fb2dd9746948ae90292864ed8063a1952d8ae96db", + "461": "f2e86adb29b6cabaefdfc971863f5604b236c0d38bc55233c29fb9267134f386", + "462": "afa1f77bf9c3fbc90b4fac479e5aa2bbff5a26df73e090006de2c14a9f066c76", + "463": "36792e040e8c942febf2074adf81e943491f2800813ba3b2cf450799cf44efeb", + "464": "d44468e5860438b5902c69cbb72a87bc4f2b9b2f882008f297dbc08d43613854", + "465": "5f0615b72134beb107372e975b4f287615f092cd102b3fbb22f8c329ce77bf2a", + "466": "00c0a0b7c61deba1ea3556793181f753be620f684997de921c54054eb543dec1", + "467": "9f3070414195ee692fb561d9f806809665176988c4590740df24edb857c24c54", + "468": "51809d22b5300f520767f380eda88430d0e6bd669cadbbfe8dcae53e897120f9", + "469": "a6c01adeb700cdcd69a7dfbc7fdf6ebbb1e4fb74a857080668f84613c608d89a", + "470": "f5770b88f11c1bdc4399eb36587ddb88c8c35d59f2eba47fcb415395fca93e14", + "471": "75c01f574e8e5f39ec8e259119ad531daf669fa4d6606cee0aa9fbd00013cf3f", + "472": "3e4882157a99d7d532100d90f796cef87786cc5d33b9b07ee2fd10e8a13b9fa5", + "473": "b2a8c67aca898a1a09fde5608c38b0508e065104cb08d9fb31ee03c6d1b98cb4", + "474": "0f8ba4f641ffe2e6b5282417fae9f815a85c632cfe04a4902a645bbbf680120a", + "475": "29ce580f874c5f1a71fc64c43ede0350310114ece838cdb145beb2da5b024a2d", + "476": "fad18a2cff471fe7a3e6f7ba8d9684c02c66c616fd781073fae82b7441f183a6", + "477": "601122ffc2cbbcd72136499c947261c9c4d6184d1a81eee16de8aef0ab2804fd", + "478": "585bbb306db1f7ef0ebc2981bd9d605c1ac4ba919a746bcd9feb92e123213478", + "479": "250e29b61d390b8f6b75bb2b7b19bf2bdf573154846821abed8a02b93a953d78", + "480": "81e86c76e118b1e8914aa578b4baca9c8aeeb398e4f243842ec79cf7505f187d", + "481": "3d5c89f3dea8786a3659ebe64ff4971c0ef067a350bf08b6daf4b64fbc3a8410", + "482": "dbca490e9a8de1878aa4bb975fff612b1324e56c3bd108895ea0509fa43d3ad6", + "483": "65a68d1412cad92b2dedf48efbd0c0d54517f24e8ddeeafe073e057c4f431c02", + "484": "f27e9796899213bf5978a43b163aa5d511d9c9916f222ac9942c4a697a51518e", + "485": "7059aa4f0298fcbc1898e7cc27239845065108688998736cb6b534ca3578b17b", + "486": "0a949d1c3289de383988cec99ada79157a70cddd00a504a53990aad889510a42", + "487": "2e8292d163c6ddea1ddd49846ed0f678acb4bbd80d77226e84a1e1d1e6c130ae", + "488": "d5a820183e72823475fa7c104880d577014bbdba1fd1823dc03725349c80a287", + "489": "73048bafab0037696dfeff14a2125506416dad6296857c05d064d6acf2458bd3", + "490": "aa80219becb4735b77dc42d75d3e950287ba560274d9f3a45a1d5c67eb5eb49c", + "491": "b54669fbed622d1318b5a0b990ecd55b91388c5d30c7fb2abeb693ce0385cc5b", + "492": "4a0259efebcd4665ce3d6c3a586a62264c4ea85e119d1c86639ba42381b11972", + "493": "974996ac1ae5c09265155218e02e56e4cb6aee4c320f72c8d3668f34d8cb69a5", + "494": "89a82bb6bdd119a7ec073df7e8585b1dbfbef2f88c87a1e004d0d896abac9045", + "495": "a6cd1b839149018ea853af76aa564ea136c9a8634b048e6225754bc49bba83bb", + "496": "01eaf723ce00908ccfa1abb052dafadca34da0f3b5c92b62888a0f114a074e86", + "497": "e811f1cd21a4a5a37be4149343282a7a8f38e99de7536baa4bdf22307d182d7c", + "498": "92962f52b00de799aac998782dd9bd924df830f0d49f4712683faa8a84a976fa", + "499": "cb653339fb185b9485282c13f330f52624a32f9c1a05b80925e1e1630d2db9ca", + "500": "165f9b6e0ea37f09a1750238661e6eff10811dd5534bdd217411de6e206d556f", + "501": "f81d17af982badb5bfc1e98fe22edde99835172b22965e8f11f4a0a5b36a40c5", + "502": "acd15715cf874c3cbc55ee91d4f3eae6374b41410fc6da0c53383fcc812edad5", + "503": "315247ea910d3346616c0021e74f768327cf9f68cf9c8c2648da46bc6722f6d9", + "504": "60d2c395b201d7738eeeb027958b25b8f2574340fae4b6129b32531b59f441d6", + "505": "67a0a70bc43f8e388c98a221759d08e4d4458c1b25d98ee9474e7f24ace16f6b", + "506": "532a076d3647a95af534db93b457e6530772fc06086b9b374af8471ab8e73a1b", + "507": "681241543d6d2604445f2efa23988c5dbf098fc28aef271588f7e57b68d8c556", + "508": "fd564bc5463b31c11aaaefcb06b35683ba675c1ff9477479f235b41c9b952f3b", + "509": "212b6504a1ce4172f2b2a60e8164c358d678d158f8ea441b0195ba235cf16f6d", + "510": "dd9a3893c8545d38ddadfa63ed1412949fdf7802e8dec79893d67bc7f48ada5f", + "511": "f1dc8da569dedf70782f735cc8111e5c60d8d3553b5e61735508ebed69b40463", + "512": "929fdda89390b1e2ad7cb1671b24784bc1090f5c86a41f5ff0129767946ddd0d", + "513": "52891aacccdcd2125a90fe1037bcf67fddb30eef8943319298a5aea61ad61cd2", + "514": "f2552bd3b9ff646852395a95947d2d4d52fe7af0ee36a5a154f47396745d1ec4", + "515": "1f700b4f5d51851534e9a466a9bfd923a4b4430d54a52facfc780024bfff3acf", + "516": "b6671746b574fd884e85d65255c22cb88bc01ea4ae2f5355b9a2e7903fbf04a0", + "517": "49ede8ca570e9f5d8d46d69ca859586e7d044143e2922ee8277be328d830dc45", + "518": "617279e21a5ec317075d624f26f4f3542d5a24ef0658406972a63d7a4fe93830", + "519": "a36337646c143e459917a4591c137723b6e7cd651e732c99ae2b27ca7c51e87d", + "520": "44b050a6e54edc1ea91eac77b3fa964d03f4b3312e50191d815fe91c8778f87d", + "521": "e2ebd55a465b27f7909f45f1742877d0f321b84166e34060e72711aa21e1ed72", + "522": "7516ec938884a56cc0f73292a1164f0ab56ce74b7c3d4b0764b420a7e9147c44", + "523": "f886ea41212c37c2baadcd4862e554642a43a77e1a21d84235841da492d3ab4f", + "524": "5feaf1f1244d0b7fbe7c6c777d427039199899351dcb85c5bbca4bf058a81062", + "525": "c8f2b2a515dd63fa2e19e74f69760af0ab267c5fccb9d149e9a7f61be4d538c3", + "526": "8f2b3776a5f3dfc459610e4ebad6f9f396194f9ae10aadd1072ca6aaaa1c8636", + "527": "f0c2ee8f04ea04757f21a45d807ce8b5fc68b667234df30d2ff0c53793ae0554", + "528": "0f5e5ba8991cb725ad19210126b69cb2ff12e9ca7ae643e86c264b196d698f5e", + "529": "e853a81bbdf04ce8eefe2acdff165654d741732439b49913a2afa94639e4c2dc", + "530": "1fc655532af9dc06c25ed1cd37cd50701914ae2e1b5de17c1b62a48a6122867f", + "531": "436f943f97a08a19419271db279e60b8e83cea2de5aa5442506a97de3368f42a", + "532": "b0101ce89eac55f7a92d25cf4c5146c0d651532a34022da930268174e22ba182", + "533": "00e58a86088a74914b413479003e01745527ffe3902c00812842758bdcd20bd8", + "534": "a62dd13c834254422ed19f96620aaa0ad1aa5eaa17f7aba126ef23749c2c596c", + "535": "38dcacc4715b2d9c3e1795437a8afa5b03e92bc8d50ddf6c9001a1c57ade22f2", + "536": "c1f9707df8e718f41a782d70bc7d2e05d3c9256cd5fd306fec37169e729614af", + "537": "4542b46bbe0192fa5bb9d240efb80b79369724e69c19a02667717ba520528c2e", + "538": "395dbd9b0c89d635eedd116308316907a6545e697ad270ff9daebadb761ff74a", + "539": "63e7305fee6d224c16ba27a14a517d35e791c66e6a792238f5cdb8e39270eef0", + "540": "15edf14762156986efca744b00a703502ce63e5dd5b85740004cb70c87acee82", + "541": "7113604aca06b1baa1802f97ade0710ec678c01d39231367fa1107e2563a9d11", + "542": "e2149c759419588c2541849b83ed5162a85a9de900928e76b6d86ea72eb2db68", + "543": "360d12d30343f1b476c3dc11767c68624ec91f9624d9e4c06dc23abd13e4b5c4", + "544": "9da44ae0589326bbcb8fba63895e068969e01ac4256358a2610b9522793f7412", + "545": "dc9818105fdf53cf2f600a661bed0499adaa595b12aae0063f07ecc8cfaf53c3", + "546": "5b01a4926c0561474ea1ced495c0fa576d4c5060b43fdbdd169b970d068edf3f", + "547": "07d79101b484c2aebd5823375a9b2059f4bc77b0431e82be39c4dd07aad638be", + "548": "c2a52746510e3e1a238128539fec54b5a3eecac7499cbc1adb53b789239af03e", + "549": "56cb02a4fc0294deaf4b20e980707a5131d2ab1744e3704726493a8b62cb09ac", + "550": "c0685ab4d55db2025769113122390cc546bd72d983f53a6aa20ca4a5960d3b12", + "551": "cc92575f2094c74028ac166b3293f5c310fd1b0c59f423b3b8b595e7af60f9ff", + "552": "bf744fce15ffd2cf6a0a6c4fa5b57bad7987d170b07e3ad25eeed137729da754", + "553": "0b23b91977960cf620efee7b9553eaa23e8a68a7e0c090f6c5d3c8f38e013d58", + "554": "25a8a801d001627f5e147570ad15a8edcb2b1f1258bbb9a3eb30e68dcd985899", + "555": "f9808880064fc77fe82995cfe30e2a893a5c6c3d77dfd2e99f60c2d259453259", + "556": "a2a557bfb1ce31b4c8fb4622d825f886c744debec4187df5609836df2ee42c58", + "557": "dbbd09cd5dfffe86850b53474040a24fb8a6a1fa809b50fe32c4e3be776e1c2f", + "558": "8e8485bdba9825657d1ecd6a6d1b65e523ffa4a185d7d724023043eda00dd70d", + "559": "62c29a02ec2d95836a83f3feac2ed0bdc40fb95e5edb723470503c9b69d81eb5", + "560": "41efa6725b74e8d0cd6388ab09e777552f47d9c37842a721a931f9b3e367b80a", + "561": "9f2ea38f92850576374df3bf868bc77bdf1b602d0458c1d8b64d52902d9f361b", + "562": "02a5a730dbc5335c7d5aafe8a7fbe6281594f6fbcb16fd2c3229f04bf1296334", + "563": "f0948ef48f9cdb8f752b837034bfafd913c011862b966112a3b7460e1647b73d", + "564": "909ca62c93d93b78a48547b7e18117daa0dc2d74cb0b4fd6a2aff96a9292e7c5", + "565": "d2777b7162a4ea92fd4f90ab37f3c725a1c9b2f6296fe470bc6d34cc56edee29", + "566": "2744ba1d1de91efd91626e50cc3cd7aac8a0df767aff1a71f2a7ed7283dc1023", + "567": "6a12cd71222893f2cdd7dba237ce82916a9196facc4e8d4198f04f1e2e1b240f", + "568": "0c9acca1e2e60dc17ddae72f6a0842d0c29cc468c374828712fd6e16c1866f9e", + "569": "d596300a8e5b62c970c195084aff14d6f7e6bee932aa0bea56c65940dea4b1e0", + "570": "6363ad7f218b503bf922fb46f7cafb0d9dfc6bfd0f8fef528e5ff08c2aec06c8", + "571": "566d934c64ca4f55fd2ad8ab4ffc2f5ab90b0d8db50cd3f7826adc78958e56d8", + "572": "76382cb508de21882c87023ffc7c7963867cd7fa080b930e05232960ea433795", + "573": "2bb51139902c7d18553a3ef4e60b75b69fa92ff1ddfe023acd6449a862e4bc53", + "574": "95442ab7145afd0454faee555f940f654175add071792bf022dbaa020237e5f9", + "575": "5d24ae4154e16d33cc04c109f695322e6e0e995133cd0173c4027aeb05d7f591", + "576": "076650a5a9fc5c6e5625d422e1923d8edf39bce12f7082dfbcbb4da829609eb3", + "577": "e39f89c8caf1a32e630be7f6a7a121886fffe7c0fa06ff439b9da3e2366e9e65", + "578": "700caa261eb5178cb78d40a562a368e521a5e2f5851defca5ad2ce31c678269e", + "579": "fd44bd06ccc5cb30332a99d709b02861f859abb863c0493f83d2f91ff3f38d8b", + "580": "f67a4b04112d033979677739574de16ee45fc9f21be619d3262a508eab569085", + "581": "750387cf84a35d2fe016e9990adc0f1ed3a13279d9ee7948d925ec0949b4d517", + "582": "5c94333f45ca9465e6234a16085c1d36ad85934e4a9952d105695446c0d0d334", + "583": "5bdbb4b7b5627350cc3505161481fb53a4a27b18063963b5c62fafc754fe24ae", + "584": "cc81ae8693c1d99ba02e9ee7f37736b19136c97dcef978f4376a609cffa36c5d", + "585": "769dd02577c0a42dc97b3876c878d1833275145fd2f620f38cafcd333076cb68", + "586": "a80f2d222ece6b502394f34fbd11041cea6e770a29af2ecbd89db3a0064ec1fe", + "587": "af4d07565a4b5fb643fae2866cef62ad2e5a0820d1a938454fda4498adc1a821", + "588": "6485e15d5b7fb79d16eccfb8f08722d3824b2755016f858cc4f0203c2ef0497e", + "589": "0900e706a27a7d13d087b65aba0295a405388d61d26352818d946f3f1c6694ae", + "590": "c265f66d73b92a40a32cf7e2eb78262d9f1586abb42e68474feb015c7fa19f31", + "591": "91eee1266448cf548d6be59817ec71b32dc7668db7050e1642ec166e8e40da07", + "592": "57eab5ea4aedde1531573bbb5ff41361a9be0f790d4cd86a1ecb73bdbf62661f", + "593": "27defd682b234c6787b5046d0b05a42ad5da02693b69e2dc3c29967c6d01d44a", + "594": "92091b96a13e66f6d22098b4d1d8b77cf168672fd5c7e623dcc4493924ebbcaf", + "595": "ec584438203ca1c3750b08988030f658ed814f97a792375ed0eb11e9427e8f00", + "596": "d0b7b78559f410a25420b4ccfaa9d5be0683b532ca00b58abb94a8e01122b6b8", + "597": "f83778315360c64e3c00488d1714b0ff0411dfda3f763169430726ff8a4e4679", + "598": "1c710d2d8f2723cf5e76d5d9888935a52572fe2b947b736db5d0766121d6874f", + "599": "a4102f9c1ab0e4ea9cd61b55d66d93f8dd0dbb5ea7b6e503b7f50c6a42811047", + "600": "43000cd416248e9062b76d9593bc86475835566831e357419026626a19c2d1dd", + "601": "5adace2fa4d57e40e7396b2e53763ba03e413fd91ab66722065f42ea4339258f", + "602": "613593451f8ee9312768fbd6bec7acfa7ac110bfddc23e01cb3d214f1adb5af0", + "603": "946052c9b7dc6c6fa2ce9c21c64621bff21da0a67eeb634add209e5bb604d560", + "604": "cd846294dc6f0ef40106477dcc731d9614880b48e7f27beb7e6efa2ab6620e89", + "605": "1f1a3d39eb0dc9b5f766e67e49adb7a50d6ed78c9ccf6dedaef994e81bf4bc54", + "606": "04e2cb6b696383adfe85644f97e765254d8667cb182b3308f702d0270f4b9c8c", + "607": "268f459f16ef9441c644005fbf1f0fbde08b6930ddfede83d85a85d1b87c6184", + "608": "0095233a0f4b8cf1b9c902df3f1e879fe61e5a09edd495add8e5c01c1186662c", + "609": "dde0b468c14c343bbd69fdb20858ef0ea57c2d2c5952550bb49f70cb13e48ac7", + "610": "a3dadb44b416325efdadfc01151a71b7b90ae872555d90b614b199894edcd7d4", + "611": "0e877e596f5bd8b6d3ce9828adfeb27dfae6e3806d68dc2abebd5ba9f65de281", + "612": "87aa27c47a9852d832eed53b290593c17561c629d8d7e8b926ccec239046edf0", + "613": "ef39065b61d64d893fcd47a42cd6594e047ee9828b7690e959b06d2728e2a3c1", + "614": "6e1439d2017fc4052329df3928f0253a2c6e5aeb1614bcc97ce7db9e041d5bc8", + "615": "650d6dfb67e1b7f683d73b5c7e57a8cf725f8410e82af948d451f368ca3a7401", + "616": "82a9e9cae617af4783cfc6edc82041e5c7b714f92de07a99673c44f1a38a607d", + "617": "098b7b4654a58ebac07070116b6ddb7ac22203d89bb0e4e76cc4402958abb766", + "618": "7a7a630b0fee8d46988857194169d13ccd04bd95e19f0d4fe95178a3b0750df1", + "619": "f4c1af0a89ae9f075b0f415ba23d510408aae5da9ba61e58d456db418bcacb7e", + "620": "29a02a40402c9a926e4d90f44665f0c287dc665af90fbb216310c8f0361b5eed", + "621": "0cc8d04579b2572440c0ef98a7803f5b9278e7eb7dfe71c5d98ef4676fd96a7f", + "622": "155e1b449d0c77b7c945dd21516b6970eafc25ab1c721e1cd4235edad4a12a30", + "623": "bfad5c35341bf649f9a4bb155e65c8943e14a59a33ae8f551c5f9626e2082731", + "624": "f155373f430beb06325c7c494becc1cce751a2200737e4c5aab31031c25b465d", + "625": "0cbafd84fcf6964ae836326a9df4c28301f5a828e25d8e27ec23eb0bd0f4a571", + "626": "442c36235501168e66a2a00b08558232bca8b656fd3fa785ddc31de44b311446", + "627": "d1755442f4de0c3c71b6e6ef6efec890f240ee5ea8477abbaf4f42326e05b043", + "628": "b364acc972a8d33429a9441b0043f6f4adf05ca96804e18e4c0d3a0357824506", + "629": "23dc7d2b353dae4c054ad137a50d2514e5e5e0599bc5f24e410b8bb9666e0ec3", + "630": "e8169ac8d76cc22872eda4899b9d26cf364ccc812e7a6864cc523103de8af835", + "631": "e85781ba0d56f668eb6a262cb6095ff61ef2ca75d9b4b48a1f97cb3955ed9c5d", + "632": "148b0eb057807fe2759f859373fb0dce303ef5499ca1d3e460fc750f6ab3b0fb", + "633": "0736b8badc2b39981228ef2171c9a947b633e7303d5a1869df7b31c7609f155d", + "634": "8c1468fb3412511184fa59f87eff1acb2c642dab617a3877aa2ba602a62ec1a4", + "635": "42a7437ba7987b4e0e9a4a99122aedb6f083f2b591ff6175db1b49ed13d41dc0", + "636": "b3aa73a22a9a8df6d5c3b8968529a36add33d9507ebe5213cca387b058efe7b1", + "637": "27e069d0b00c9046e9704f1f61aac9e27b72de089296d528f1c3e55aa02eb55c", + "638": "59fdcef4d53b689e72471abc175b705f7aabeb2057a64a0117eaf85d171c6bd8", + "639": "13f329789c133a6d9f3c4985c325d7d9baacba39346c33956d97c078fa07fd97", + "640": "30f4644bc8739ab35768e369eaed49415a6e5abd663cef811e82061b138b2158", + "641": "a0ea0c0d147afdf96be67b12f87e408476cbd3b6100aa8f689f7461c15a82edd", + "642": "b752d3286082355899a0f3e5b075db9c45c5fba113c474db9d67d4b919d9cc94", + "643": "636890340bf4e49283221f32170e6dfc5b85469cd4117cdd298666cfc79e5ca4", + "644": "508176c0d01820299e0f486256c71e2ffea13fb35791cc69f2c4a90687e0d933", + "645": "cb3524f5fcc0db9c4dcbda905542e0e19661780c7c46ac2cde7cc1c56a78019b", + "646": "88668237f00615ca37524dc041ccad2c13f8ede5292f740661752d450e6274ed", + "647": "be798c56d932767cb0c9f6e9bbd2dc859f2608d374a3ba2da505c3fb24ccf8b1", + "648": "4e4a65b4ad8f65d69fe8e93f8126c14004ddb1c97cc015544c3cc0326cbb09e3", + "649": "a984bd3c65845bf475f26041b53200d65923e01dd6fe2fe63cc9c16e52f8a184", + "650": "f04aa17301b9cfb62c786755b6c5e1be072299e8fc36c0f71acb12efdef2def7", + "651": "230d0fcd94b95d3ae2f5ff53d9b0579327ea97d64b2c722f57ff044f0b763895", + "652": "8092af52b458ebfa28814ed07efcc51cdc69c28752729cad535600af9c8de0a7", + "653": "81989e36239e55dbe1970c44a16d4bf3a691c4827a300d70c74fece6993df94b", + "654": "087a3d7a6bb9e05268f4ecf71571fbb6573f637bdfbc11c24e3bb3879522c4c8", + "655": "1355e29d5adfbc87fa32180e6bce80bf93678769d25d7aa671b691c226367221", + "656": "5fa788edb4cb3e66476f02a129aabfd8105d033b626069c3c256c2dc2192a148", + "657": "589eb2d62bd7b384d840c59e4b8e1c083ccb1ee44cde41e96adc09076bf29409", + "658": "26c4f83287e8b1bb9eef8019fb3de85a739ffc945d18210529514ec36f1534c3", + "659": "183ad92031ae827685eb6141e2333db39b63096a476b03004a4eb0a1e425b388", + "660": "d3e7bda328cedea4649c229c61aa6d59a2d3101d9c8018da11701ebb956757a6", + "661": "0881ce3d86a45640a1f6ac3351ecc11904396abbb1e01d59569de302af9a11a8", + "662": "b76019a6e0b81449e893cc4beb70fb4032987933154627222640759643790b18", + "663": "da43a8f972e2ac9c67202554270d3872ebf9cd70373e9cc38d976d8981d5a6fd", + "664": "8cfc0d7b7c5493dd90c6512d3d5c140e93dae47ce02bb5f37dd22eab631e1048", + "665": "c50502aacf61d5cfde1d9caa4c30c25c95f69acb683b828af8547a0fd17a262a", + "666": "2969a4c65fd1a6c63caca4920c47f3bd8e07a83a85095cca220f100fa4f7bcdb", + "667": "9e1f8ea01059c89d56dcf39ba1f48165c42e96e907a2d15cd245d48f83eb2aef", + "668": "ecd5cd0087afd11492655efdce87941258f07d204fc01951de7db67accb9aa70", + "669": "26dd01e557de2e992b08290065a0d838e22f4045e61ffb8ed1a772a549da4c28", + "670": "03159699d87919f6f08a5aaaa4010ddc31cd18b0f6f686619c16d415c85a6771", + "671": "2afb94259b7475e2b0b84030561b839d64015cb05b0fef59d9e7a8c800f13f52", + "672": "95ba11bc7a446a89660511742358d108954475358fef19aa63278b2272a50967", + "673": "0aaa8172e3d46ca138b61036c16e9fe30fc76a013d0082cfc47b8ba6485b2484", + "674": "2f4b402d7378e21bd116885728cea0981451593c93e429091151814ce627ceb4", + "675": "33cbaf2ed8b086df2a5214c7b80aa13db091a40fbf65dad4537d814edb7976e5", + "676": "83d438d7d6aef7dd25a3bbb846e693ad6977dff625e75f57c8dee940ac860303", + "677": "5bb2e859df651d023e73d535cce6dbab7b54bc4df2b74a2b7cf0df2c4f7b9163", + "678": "702023538337c45d844c6126ee6bb52b2356951be9e422545432b63f4f0f7e51", + "679": "76d7598f14c509ebe1a39035f86d188aaa72208903381013f8a8f7f34def555f", + "680": "f18253d7b2b258e26d23a630a0c5e0c1fd82d933c02fb9e053e1cfa15fb1fabf", + "681": "eedd3c99bcb74a111b5e0452596b5c51307eb2759244e24acadc0c0ef44817b8", + "682": "dfcb87e88fb59f8d02bf1f157868d601541e3621ff8cfcac847e39595062d9b7", + "683": "3b82f6d785410213c1f11e882c182f85566aa97d748be67c4b2557956661f5f9", + "684": "c34c2cd166739b0be37116feaecbd59d7c8dc4e51d8b8169b4aeace14705557a", + "685": "2db4ce3ebe184487418ca9db4e27b8e6733ef94a5b2934f703933d7cf84f6510", + "686": "17ca0f65995ce829fd95013f84e6ccda8909c1296cca17f8b5a5bc8728ab55e9", + "687": "c9b4f380331c2e050a3be96f7da4da9c4969eb3c01b3f557c6b1f40923512ab4", + "688": "4f2bef49d3392c957d8ba9149090669507a73c34a31863222f7a3453d27ab6b9", + "689": "3bef09f224df557ade5bee66d64d3ce38f06ec7fa79f2f4e67b7d0071496fbf7", + "690": "ae27f3ac1acb726794643ebc8bac1572241d26d2d9f8053f991ecfb249ff6bf4", + "691": "214bc1cf92275cbe3b3e14c8c21a5c1be51fa4fec5a9bedce423f082b831bf04", + "692": "efe9cd2e416e668dcd354b7d1ade5fe619c8b732cdc784fc2060a31d83b1feae", + "693": "60a4cdf9c76a69a099c6a7e6dad4747a64bfe228b9bcd2650959d3fd412e2ac1", + "694": "cc1e3a87836d5e673b7f06e0679bce638278b17955e0f589658eda0351b92a39", + "695": "fb81c90f3e6658a30f7e1e501e9ea1bf63da6e5010df7a55ed457849a35d5640", + "696": "92ae9bc9b04cbe07b776347b335a262b648e5e516dffa6cbd7c00240066b84cb", + "697": "f06cedafdbce08280152ccddb51c385eaae6c2ac193eca3e8cbe182e42a9d082", + "698": "a1f77f5ac117a72340bcd7d1d61bac73f279c61282d0fcbaab3633dbefe11b5b", + "699": "d47ad2c508add4d41c6bdd7070c090adffcd35ececb7ba95abb0cdd39faef088", + "700": "6241094aa425a5787aba16eb7db9ab3258b813002f108396f0de6787e5066206", + "701": "5b432b1675019b7182caf6def8e0f9be81e96685b5fc6b0968edcb68e2108bdf", + "702": "27163c853db0ac3c611907c6c81381f8b25310b78214db9759e0051432e9cc56", + "703": "2bb1064d9f8d069c4ed820dfba7bd7b1aa9c2f715a9378a49e5b430727e709b7", + "704": "a02bff723a1e47c6bc24158f6b8cb54953edb9514a4465a63970d766ecf7abb0", + "705": "dc34ef7b817acdd8618416a6a53bce8a312a6858bc7c531549dd257c0a7137e5", + "706": "5d0d17d4ca867ab792e18d4a8cc618c9a712052c55e09ee03f1023319b5d350a", + "707": "0c00818dd0dafb473d65d2a8a4ff5771eb5b7ae47cd71eeb52ada8c0181a012e", + "708": "2996f7bc3cb0c5bd866c16126494b91d72e85baf65b8fb0f3c4df3e63d10db9c", + "709": "6f25fd21e775324321976b31ed9d32b7bfb3fb9bbf30e24092836636a3c77c61", + "710": "c2e5a8ee7169cc59f1b64816aa6fc7b5e8d379d316c9faa6ded5e16dcaaa71b7", + "711": "e3d47c20925c5d890a255e1f7f533d59c4ed528e1bb3d1f27f17c088fcfff71b", + "712": "1947bf6da1f5b7cfd95132c97bf73192da28fa2f1469d6cec0fb8d3adf285310", + "713": "6c816f8296f381c3baeac741a976038fb62ccedf40d81746f0c2573d7f24cdda", + "714": "5e63828b976bbc68ef596d444991ee3ded58da560761bc1ae558649adf4196d7", + "715": "253849a54a72177c28f18aa2a9bc89b8cdb9daca9198eec21afaf0f843e679ee", + "716": "8f0ab551adf59b8f26a1030e48105764214d06a456edf4fe721575efb5eadbbb", + "717": "fcf8ba403f1b8dd5e0a376ac8abb5300c264390983046bf4bfa2aaec2923ff50", + "718": "77efbf90ad7eb6c6b8461b88e91f6413068fc25f9f4ccbc40813cb5c687841cd", + "719": "04621ec297594942db24e9136c2a714ea7e30c8c7b727836005f066888bc17ef", + "720": "a3ffdf0543a3fef77f0cc710fa9f7e5ba426ea76a4e8adfd326067b3cd963603", + "721": "a7c9ef9dee748d78f715f1dd8617b1e163b7ba39415544a9decefdb204079426", + "722": "8279ba81f3d90a0e8f80ab6a3a79d05d38f81606264a87b65349911bf8686242", + "723": "10c4b54d8f76970f5ae576e04f6f024f8a4dfdbc921d5270320cb1cdff7e5f92", + "724": "be4830582397aa825d6dbae4986a1817fc1b8dbc851b48bd09f4cf3c88d37187", + "725": "83f316f3605ed65f7a1aaf3dde2866bc05c74b847305326bbdad22a8dbc3f9b1", + "726": "963244a099736bb1ccf4b1c39632c7abfb65e66454ff71721b11406a74bf1895", + "727": "06109d03a4e685e77a40b68e8ff22f770acb8bbd240c4a8ec73da7364e89548b", + "728": "b5156b904a1ba715b86c08399f0f3773af2f48c210220ff20ff68bdf3947b3e9", + "729": "f825731f0e1003f5d7a2988c1ecd12bbb1b0fa785dd3baba7e45ab39d3ea84c9", + "730": "54956e7cdf282440ed26ab4facb173b2b8dacf21ac4d852ae0e373be38434529", + "731": "933b86589601ef9f5e02b952006f23fc6794b37ccd27033066c828be2a28239c", + "732": "91b19e92c9724c0831c69bd97a6c0f12bffe72607c6105a1adee24e4f00ac390", + "733": "aaa91eb3689dfa642c485361968a03a50d46189ac75eaf281e23bf5a437ca455", + "734": "17a6ac2bd844d6a7d5bfdbc37a83660e35c7e6626fa480f714eabd27b800a0d1", + "735": "80e184933fa67cd850a04df80080ddc946dac82b07ea53fea5ce7c5283d07449", + "736": "9bf3aceb025e65318635d16f11b60af1a6d71761192d8f721a16a4de82aee77d", + "737": "dea9f756bd7f006c0c324fa146fa7f782a673f9584dd90a4fb746c9e74a7baa7", + "738": "f95b6235307072a93b38fbc233cfa3064aabaff4ae157432321ef881955e0099", + "739": "481d13704821f5dd17ca18876816faafffffce45a3e6e55d08ab1b7329544699", + "740": "e554d2e8dee40362f2575e1738859036a4f7e44ac09ec359bc897dcdef0b06a6", + "741": "1f51e884bd769fe3da07334c625eed5eef83887fd7668531c0126155368c1c51", + "742": "382f50f25d25c607280b4ed85867f583520c55434660fa1640a8fe92ad73d3f5", + "743": "224e7145ca1f2accc880115eb46e05d2d5eed390cd560459f2ed3c0c6033b278", + "744": "9bcbe100b431eaaccc4370848b9fa06281ee4e64073223d584911a80420f9269", + "745": "487e95999493d6d1f79252e017458ed791e6d3ae6f5b8f11ae07bdedd17e9a9d", + "746": "cf7000831a6fba876ba66627c9cd1de8f5ac8d96135a1f6d592781f9d6c12dd1", + "747": "011a31576b7a7559d1e0316a72e83d090caec03890497bb6595325b0d6ea6301", + "748": "0c27b2ceb22788f906d89b330f056bbb071993b20fb2b027e28f31dc61a92c53", + "749": "477df253ca4d779d5bb4b75ffff92df8f8e6d8a4cf3148791ef48af4237ffe0b", + "750": "807060fd901f50edff3b30f17455a8e219166e755ca0389bf757bd25788049f7", + "751": "55e9cdb8e9346c7680877c8e5d4e402a2661fd8332cba20db310df33f15c5ae0", + "752": "9dd9abb1a7c16726655b4b5365f61c1619bd391a3f2c0636d68d328e615c4aa8", + "753": "38202ffb4110b0f4146a1b99a979ed16b2ecec540347e1494f7085e15fb2f739", + "754": "c127079a8ba383ea7680c6ca6ea3c1f0d193375aa2ed344b4d5e61373116de78", + "755": "bbbda42bdbb9b778a3a52e64fe41e576f93178e704b0438232177434021c5be6", + "756": "22180f1e242a1af51ee63c8ca8d3a7a1be527f44efd92c5c86b85ee6fbfdeabd", + "757": "a3ca09678530b6783f218cfb9a51e0fa668cc7c4995d85b9628461149080a518", + "758": "26e3a5106cc4fae9638e9adaa7d3bdf9ff9ef5330b5559f688343d6ea7567175", + "759": "d96309d32a827d020a8ad20f6ca47acb0d3fc317ce4084fe34b767bf75226de7", + "760": "8893d802de4db6f58ade3c0f0d884bf93a77bfabd80588b64258c0d72ea4ef27", + "761": "c4bf9bfbb8573bc8e6a262923e85713a5762014335cc78a18ce1b300216aed4a", + "762": "e6bc45c6c8c825883efb85a74d98c2424967edaddecf0f9d1a75d36a334349a5", + "763": "64e8cde4c4e773b00f024f544d91e6712fa64fe18d37120798dc3b1dbb060cbe", + "764": "c8f12cb2e4f9b5b8b94c52aa06a647010d1d9f8272d71bda54487132fa135c8f", + "765": "5a16757d233b4f194df187395f7b06bc36ba21509febcb9f7dfe73c470bc0f8c", + "766": "c00164c549464a9a167e1cda750767085e17e15f38cf2e69486c69046437f307", + "767": "72c1f2e1ba2c5e94147bd46252fc4396b6e18f7e1ce8c8a3f7c807737635700d", + "768": "b7570c61ed6cc24df1d83a3535014e8ec14e68215857b55ae8818e2c76ddd546", + "769": "210126d510424364549002a6f097cd8d2cf4cb9d8dbdcb9adb425a410a142ecc", + "770": "3c988178595f68c6aa634ddf40f5fd369c6f47b5989a3dc64ee80b5d491dbfc9", + "771": "476c792c9e143bbb16c13a61d108982b9f28ba626defefeee20af3c8003e6f65", + "772": "dabe18314c79ab64caf609323f05e98c8654450aac62cd1265036cf22d1d9148", + "773": "c360622bf04349156bf9dc0378c608da1babb9056c1d6859ceaafa353b547fb0", + "774": "4a88528885a64d5646e96a3ed758094fb6fa55f9b6beb262b28aa0f5ef022fed", + "775": "720a22a5809ba5882ffbb9c7ba83c84c907c9abe95ea43fa0b9506ca2cbb5a0b", + "776": "a1ab08820f5a697bfec5d263ebde00f247e49090e293247fe3295a02d29c3d35", + "777": "88ced6f861f5caa4baefaa8b5499d7ad57514da99544324144563b22cf95f500", + "778": "4b7c1b015f2a40f1ba9465e5c332ca4dba6e707591348ded896bf61ea3f35e0a", + "779": "73f2e35b5df39f53589c6f55a6a539e923214d7a1803248d5597613845bbc1e0", + "780": "165fcca7a5931b599e6586492fe16e9f09450fcbce7bb564544123176cce350f", + "781": "7b2d258f43bef4241c8ff7ff02d1ab077f102b0a9de47a45ff29dc6e88f2f713", + "782": "7cd53ee212d0ae2afbcfd9634000ff6c9df2d4c49252883755a5df5c45c0abc2", + "783": "97e41d95735f2f56ed709988b13d31b13b409b9547e04a2f559876e45a19e00a", + "784": "0ef6832750545edc09519c5fdd333f711c24c18a30312e83092741a50f5be045", + "785": "c941bda270f687a685210412a9c29d36bf04e5e7d07e9e4e66cc7040fe56c8c9", + "786": "bf4bde69583f1855bb80be53e7661d3a237941abd5c497260954ca035805cb7b", + "787": "754058bbb35d7e8f1a687706bf4a83ada3b054923f425f9089e4978841552ad5", + "788": "d332a3bf15eeb15d684b352aabc2f0643d2124292f768ae10b47ccc4a7160229", + "789": "023450dd23ff9935b00a987a8735aedfa782b0c160ff2bf0a77f8b3c47aba80a", + "790": "38f4de0c79c6f596ecef327bbb9525df0007692bc7d2258f57863a8a721b09b0", + "791": "34ec80be097a5825800d3f94a131cdc8abddfa3d48fe71248a1b96f2d94f17c7", + "792": "acfcbc5c78c7a951337ede5b842901071fd759474009ef139a52b9973063e96a", + "793": "b7dfc1b8a61c5b8276ef77b41697a7a8a9bc9bf39e2de0bc7900338417f749ae", + "794": "6dfcdf16d19ebc022bdc9d5f04630a956c0be84e8d675473c6270717c5ccba24", + "795": "d0de1587a273d3340281bd6a2e6d634c612c975f901a4c24bb89cfeaed14b926", + "796": "511f25525f569d405a87c1dd8716de423baf48c07c7bc55b5cb0ac046248ba00", + "797": "cf1c9b1cb7dce39d39b98eb7341cf2e2ae6508d3eef2397e73ee27e916590904", + "798": "43339815152376dc1c60d7ba438383a703b3e0779e08c52b5c6ce519866d92a2", + "799": "d2ba3e72aa2dfbf12e146d323140331faea06a980d52fbf968f2a6f3846f8213", + "800": "5fdf8988b2a560c82a2da42beaaf8b093bc8130b40e39ff85ec9037f7f708432", + "801": "aebb7fc48c13fe80eabfd81bfa7b3c5cada441e71f936f4505e44793f77da997", + "802": "1e5cdc83273304b99867660658b2a32782040f51b7752d67c43fefbc6bbdfc5c", + "803": "bb9b07e5336e7a865f2e325789e19793c9f3d958f0446838be5f1788763727f8", + "804": "16e4f6e2f1f32bcc434f2093ee7dd2caf7fcae08d199d6e94574ae1d0fd9ea94", + "805": "79ffbf2a9fe88ae25cb95bba670422c2f73fc3ee6afa38c708f04fbe9661e084", + "806": "236bc7069ac692312868aa9ea552afa5c79db82576a80fa3402a6ade916b6946", + "807": "daf10df8343a373791fc553d66182235bbb2e26ec8a0a45e428c7f8c81f02166", + "808": "b6cea784e8ad741813f28d67aad97ea8c66c9c1eb5abc1ba379277e5479b570a", + "809": "571b8b0a8afcdb0c55d9783d3aee49693a517eecbad5a7ca282f82ef112609df", + "810": "db1b48f46bf35bc2595fb2f5c4d928eea89ed48bb4b65d68145ddd986ca57ebb", + "811": "30d08a940a9b3e17198388029e8b222c88a13a62adbcb249fd13c4ddc7ded9c1", + "812": "7a2c45e4ee6d257159ddad492dd29b430f2979a67f7a1807786fb2f1f627d297", + "813": "fd66b4857101d217b2522b778feea4b08de7bf875e37a92b55a9ad29d3c1f3a8", + "814": "ee8c1451c7ecd736ec8f264cad8bef456f3368128ae096d2b92e357a17ad184a", + "815": "5c8cd0ea239858f5e22fbda45b3a8668ebb6e8fa1901ffc796e91ffc353a9447", + "816": "770f305b31252705b08537e542f132500fca667981d632340756373a8e7c6c23", + "817": "db2538417fcf17b3a9c29b109c17ef3e2c1e3a51ce9fadb43294b4a389de3b9e", + "818": "e6f3eb890b20b06103edb41cb9125734da73ad03babf28745f624b01b1a369d1", + "819": "90efc6d724882288b9083333be6c0cf3b4eaad862a7cfc9251c574dfa70da5c6", + "820": "93ddb363bc39d0f870d55ad198556a9708a3415eefeb58c919bb869d7edbda40", + "821": "290e67361e738103929f253ecb00091c13c47ed39f115a66bf72dcfdc930e8a6", + "822": "174f4d0e185c15cfe735439e704db0426eaeefb037c6aea8881b7b28d7bed2cc", + "823": "54a8b614b1b1656d5570b8486de3d3376bb9ba623d535bff0377c5d9881d64e3", + "824": "67091268f99eb247b37e0f2b5f5d48c8f3507fcfe61eb4aabdfd89a4bfe7fac2", + "825": "26ffb6447c88b382a3144ecb34e120930068edb85a3adc014a96e02cdfeb23ba", + "826": "c6bb21cf457a6f0730bd89ef62c56aba62d1ea749baa7372acf58bae424b5c90", + "827": "9b9574cd066fef477fda63d99735199378ecc07cd19d8b7f99707ab2051185e6", + "828": "cb870e90e15665003be7b2010f0b07e5fff2c0377b09264c4084640acf46f3c8", + "829": "be0ba351cfb8583fc158ebd8624574715734d60480b5798bdb866ea4adccbe1e", + "830": "dc982345368e8ccd4912714709bfa527715fa52e30c92b7fa373a1442f34dd6d", + "831": "096af07fc004477f85ae20c1aa87c7477368cc4e75642e7039261b4f2fddf3c8", + "832": "1f599a4c0bf837fcf03cd3936ababa7397455e258720561f27df4eed0a8be1e5", + "833": "b81da735bf761d93cba084ecd96b2fc917996aa4163a5a736aa93388865df205", + "834": "3ce18c90ab87ca2ed5f21343a177f0871b780b519da576f09ab0be91cf5dd3a9", + "835": "1837695a5dea461e66e584bf15ae7f28dd95e06e93ce4854b301a10056e6483b", + "836": "28fcb31fce8a4c36b2d75a17e3504e05a69c0483fa2cff6aa2e8e8095180ca3e", + "837": "ddd317d0d3b9789204e1fbefc30afb1360075be99910a013107221d70e5a286e", + "838": "31cd81b41b77dd72f24f89aece5a1abd4122f0e010ccccf176d576855ec510ee", + "839": "d5a687093f56a013a63b50ad79de9790f96176666565191948cd78777a85ae37", + "840": "3231060b866e162febcc08fd3bdc65979bc7e3a1a772a9ce366d704754412198", + "841": "8a681dbdac39fef25dd5e140378ed92e228ea353c371e39619037436391daef1", + "842": "692cb368a9a68a92ec5d3a22e230abd2331c38a481fa19e8d46b6f192d899479", + "843": "a65b042356ae8bb3895a1fc23010148e3a70acc7d7a4b0b20aaea77a8e8fbb23", + "844": "0b59a39bb59f89b03b4c087bf0befe6d1641d5a9681ad0a2ddb64c938517c40a", + "845": "bdf7505c71d7d2764993810ebd59b7262ea506def1c70b8d9fefa809c2273f79", + "846": "27ba05e389ffbe77f482b07ab23608a55df2c6ad7156eb9869936828c453c55c", + "847": "3f9642f5a440323c968000a8f6c9b13c8435f88d227432f3ecfff1ef9693257b", + "848": "c0f69a32aa3df437df8bddb65f39ac7b9b134e9160aea5be8fc00852287391b9", + "849": "53d4d84d6e41011ced6303e1eb7392ce9f08ab679841aa754c1f56d3b99cb36d", + "850": "86d2610a306feb16dfb22444ead7de4850c3cc63d1b824e2e92bffe3f7f832ee", + "851": "a500285dc532efc7e9a2c4a97420afc17174ba329e89de27e7f0639042cb44a6", + "852": "141a52670327f42348afcd916b99de415dc5582313fedf80c78b70f270d5a39c", + "853": "7f89e483cf564d659c1c792ee0cbcc3deab337a690aff2592dca9f191388838b", + "854": "0307fbb9efcb5e7eef37fdf55393efe9cf68782054c02bb2dbbbbd391be0f83d", + "855": "a7dc4a60b32176df968581aea7ac8166c3d0488bd653e290383e0974437b6772", + "856": "e30866f89afd41de88b6d6a0525555aadd68d857f91df24964806b5736470866", + "857": "951ff5a8c2a9671441f29a660a389ff92b1b58a8562b2f2f6bc52bdc4c35faf5", + "858": "407b59526b5e7db5b4f86012a5f314959e239a9bf957f0aa8d7b8ab4ddfe5cfa", + "859": "b781171df71960e6cf94313872c80a7c8d83885a4fa37f0b7796a43d01efcc34", + "860": "7b07835fc10881cd12a5c5f6d3e5aa69ceb5c181b9a8b553cc4cb26001888756", + "861": "4ba49a75009814e657935b4b7fb67f7b6968176bc4bd4e40f141871bb05cfd8b", + "862": "b21254fcc6d8318348bfc6378056be9155f9307c1f1eaa4ffcaeb1c8770a29e9", + "863": "57a840f4db81d468841125f7bbe4bbb9cabe7970c4b01b129100cf85cc3b08f3", + "864": "d374a0d8f94bcfbe882d651199ce6f1f1f2877a3b8ab1d0e6f80ca01bd858d5b", + "865": "91171221897c7c880e785f60a65a0e35d3d9e8b6e76a7fdfd57324177be5dc43", + "866": "0600fc97d1060ebeb6fad2ba601298a26dda70434697ef0b297196ac964a2360", + "867": "f6c86b12cde4cb274de1e7ae06b4ba0d106493223df4c11cb7d4e862ac619600", + "868": "a2286c5ae2cdf8a37ee2264ce3cf1f7c0f479b94c862c996ac76bfa68bc49664", + "869": "6675c0c227a28a9dc6bf45daeb77019cfb690a3aa6e25a8e1ddfb1fab4f4c3dd", + "870": "ff17fbe03de978d554881bbc7c5b640350384e707b4943a05e0b3002527b10b5", + "871": "6db8cbc3cd2928e777c4b3a7aeffe4cff750b873a8f1c6e475dc77efbd748da6", + "872": "20fc8c9f5cdd570c6d836968dd408315159991d594542e5f0ba00b2d6e98083b", + "873": "36361490baa4da13ca9d47d2c5389dbf3395a163ef862bdf980d946f703e4a0c", + "874": "0d91bddf70f4471f227f125013620bf0b4da3be0638fd6b6e24faffaaa3e0a4a", + "875": "2b0f9186a54c6abfad5e243343fec3cc0a9146457c254c54a35463c657bb6b60", + "876": "55350739eb85ec6ed6ef466533d0edee83419ca5866d57f8da552a31de1b9f5b", + "877": "45e0e08c76ca6636f7031fff780d1f9d314f1d29c611e492079412b9eaaed60d", + "878": "355e588e3051e6ac4ba9e629af1f319826cfd2e1bde0b0400f0a1e42b74caa06", + "879": "0641d0ea66e0759f7e931b137c09da0296fc501f57f099f0568a74b1f3b615cb", + "880": "2e503b82648e40ee06600da19a500b71841fcb480cc7ab96387da370178b1b36", + "881": "9446ad93cf8db001a55e488f0e8f721d0e93a97a911c319f267e94cb3f712ef8", + "882": "c8f048a571eec954ad80d54d88fc4b600db3ab7358120cb8b52b34d5a7a71e3a", + "883": "c737e8926c9b5afc2bf94ebb7f5fabcd78dda5d3bdf29cb66fa5021c91c8d361", + "884": "a364235562ff8b49876f6018fd1b3c7b396dd130421c3bb5bdea3fe54055553b", + "885": "2482d9b9b31fdb1ad68a3fefc28aae53251542175e200493ec050613f9fc1b29", + "886": "ecd44b3855f0713633a2c092d83c8b7a0608064588c77b240c19ae53e47079e8", + "887": "ef1eeb4ab9e653345b4cce5103132bde4afd7738a867b422c6ed0748c1c91a2d", + "888": "6c136205dab6e3727943fc75288e287ec1b6c0ab6ae2aeea6fc63d30a84b7707", + "889": "f482793d61bc3b3795dda9851f94446235c3b4e4a1ce7025b65d3011d0aeb65f", + "890": "8e12945b64ec36b70b2ff56ca21372f0759ef2a4dcf9c3c47bbf0f4a36f169e3", + "891": "8b936c982060a059ea42e0656640dbaaa9cb045abdc73f9e2bacde29dfff9e77", + "892": "93332178b9e5c9eb64b283c8c8bf5a73dbef52a9a13668eb37736d52861bdd85", + "893": "dff5ae0f7964762e7c1581c650056d09a40d3acdbe00c7521fc5e16c959450eb", + "894": "fa19097bd0328ed81a33170aceeb83a9bc2e15d4595e483d22c22e4c0d358968", + "895": "14cc63cd258cc66c94868740d2611e01b8815cdc098d86b5911e65daac5c679b", + "896": "0f306b78cbd5bc65eeaf386426f94c91ca243af0aff3ac9c200535c8a8cbab30", + "897": "b8d2e6e0f2d6543d6ef38e9986db5136479c1e070d029f49f813acbae933205b", + "898": "093481033d5da63087b90360d14a3b874444a471ed2b9d1a0a96a3cf49e29323", + "899": "23a98e18dbfa6231da4cd01bcb28947f7d491119e75c37b949bfb5ccad31e0f7", + "900": "918964912ef6585889f387633e6d56f9d3805fdd6ffe23c4e3e713aec269dfdd", + "901": "bffd4f639108ebe0196e0531a222d98483833f5ded8f70e20dfc5f3f7d505978", + "902": "542792318774e7533623a9bca81538a15055388e7626492df7e713d4bf60f0c1", + "903": "cd4e77cb24cc280e51c816ca968e625e29c5ad54dc35179e296373f216ba72d1", + "904": "55dae5e576047e024f5e73656b234b3915f2a31a03c98edc95e3833524aab4bc", + "905": "f908ef56541bc73efe6ba8ab8f56f5a081bbfb3ed89a62cef598f2a8fe38f190", + "906": "f79b734804cf71c40eab36c0cee091ae4077ce230639fc37b81eb8668f851825", + "907": "38b4fdef9d710a38ce684da7b41330fc3a830fb14d6cc0cc4bb12db2dc35718f", + "908": "a37cca58689c94945e99b799a5edf468f5e7aba39aff04be468edb15795095a2", + "909": "da05b647ad86600077661db9d82377770ddf3dd7088446006ee932a4867b519b", + "910": "361851e8b2ffee70cade0b9c81fed1daf9fbadb8811e5e165952c78e82eebcfe", + "911": "437e542fb7bda7267f2481ca7d167afacaeaf1bd4178c751e9845f42d7d5c996", + "912": "19bf0f05a5a630c9a515fe306bd9dd895166da38b7735a395c6e3f080c8e7cbf", + "913": "535121e61ca225da482e9ce093960d254ffa983d77a0523280461db3ab729557", + "914": "96596a125990a5cd235aca1c57e5cb1ccc0e992b158d265530038c9490f628eb", + "915": "26b2106455306f26fbd46454c9b375dbc893e2358e86fbb43e78081ad4b45119", + "916": "b856f30cc8a179a4361268bacbec896d1a050f525e43ce97738e816479b2c89d", + "917": "e9035bac0d4ee1a5aa4a6358c234abc1cbb6e97ba22657bce45ece2c7aa34313", + "918": "34d037e24b0933d3a870c8663246d717163b5a02852676343572031ee84f5e0b", + "919": "873ceab0054600dff15059663fd8110d75170d404dc8eb7b2f91565ddc38a367", + "920": "6f9c7bc56f7dafcc885b9f1518cd1b333e24a68376257f8b2e6901af7c090835", + "921": "9c0fec35f0b4610cc5fe4d08f47aea4d7c0f4e0c5aa5ec5bc7223fc66eefde91", + "922": "b6f05b0bd9927b98cb816a9eaae41290385d79da687b40fd5c71afa65861f3a1", + "923": "4ebab7ea0fe58947bb2e0786d2e565e865d3ba5c341d2dd4978775268af80295", + "924": "b1463f18a7e7d99f478193c7d991252a953f1fb396d517e0acc918afa875f210", + "925": "d702fb0651f827370cc716d40927d947746696c4f7d1cf8e19627993c871047c", + "926": "a53565a4e62f2c126a6796b649ec29d3138178cae80328aa103ef6220e525ca0", + "927": "1304c67a30f2bcd12a9283db51bd85e349017a7fb506d639126107d258ce1fd2", + "928": "9b762c890d74e8fdbce753db6b329b96ae3e12c691d6de3067587c17792bc23d", + "929": "ed31e06c2e19073d9e3ee373e6080f4f8e4dab9ba46de44aecc5e7190a44f2bb", + "930": "48277b5231484ab6abaf25f34e44a18a867bde601a7e930a7f2920bf4bf8b531", + "931": "750b8a1b4e412d99afbb530f1f2b00114ed4ce520fe3a9153766e8b09a963c63", + "932": "482ab5feaeded83eaab361581bcebdfb013efccfe46f00c2cf9b240fc3e505dd", + "933": "0dde32e290c4595e77ebe7dca7f9df88ce5213b0b2ecc0c11bee473b4489db4d", + "934": "38c9853947beba38bc41cf16ee99f6314eb0c543229c9cd6accbdd1d1dec28e4", + "935": "6352968d6cc87ac2647f5beee1a7557067cb3770b84d9600938513ad0701fcde", + "936": "36e73b9497970293e77cd2254bb50524406791f2b35b4731b1c603ce90faec1f", + "937": "7046108cdab3d2e15ce43f6e0af05df357c11c747392f117fa3a723c73923e67", + "938": "a59418745e9809e4e1d38291cfaee4eaa79a082106ba13b652c80d5158617264", + "939": "d39d8b1f31847456215c5e5f4554314214066a172e3ff72df67a92051e5447e9", + "940": "892b1ded5105b45c8fbd74df4cbda7ac9fc3a8407ea44f31ed84f24f269dd663", + "941": "3de539265e564b7a77510b9e95f4cd9597c8fa36d2e7a0b75ebb7a4b7f52c3d5", + "942": "027fe4e0b88c15e465f43065dee0bb177c68136ed55be730b67a725166a94f69", + "943": "97c796f642f1762af50d2d123daf637fdf1ce43cf12a0c112081ec4777a95838", + "944": "91d9cbc637f36349e2a73e102cbcdef082cebbeb403ed0857fa7bd3e5c714183", + "945": "4a2366806eab7cef4e570a463fc05e9647b9729e231ea0030304083c78f702da", + "946": "a5a5d58878cd73ebab396ceca2f2614cff4871610361b4115a8548f3c7f1ec1a", + "947": "fa466f95f7981f6d995939a1b3f2a8a8dc66dd3443be879d55ff164a9566d44a", + "948": "e5bec5e9ada32c65b734fac7b912b72c1d4a44bc40d88bf4eeb09d8fd1fbb602", + "949": "25d5ab25ce90b6215a98ad5cc814f6126a889a3eaa188ee56d44d9f1b20ff3b0", + "950": "1c68ad8633c77b3b7600157a9421a71028f7cb5c0c91f6e8fe045fc089c07be8", + "951": "5acc164681d7f812a85f0f5b0feb7ac8dceaa176fd5b47a5bf353c7728d1c4bd", + "952": "4f6581625c8e279a5b5ca7ecc3731b1ba17b387f32e4b6f420c6d71ea0da6219", + "953": "f9f5f19b68ffb7538cbd63afabde7b914795d8cd52f06cebe7db8fda3f6cf781", + "954": "1a9948c2a78d258c4052b43458f7dc228b1f7d20d354299c65f67399fe7af398", + "955": "3217a888d8c5c7ea0e0968cf9414aca1a2924eba1c8612af6ade4745c4589c66", + "956": "1a81f6d172e7114275b3e9de39bb1bb9d73ee9a13184756e1effb5f5b1e802d1", + "957": "8cf06abe4d8e281112a8af7101182f9b232690d4d535ec4767b0f1b01736ad8a", + "958": "03826182b26461c77ec3bdb2054c475f2b864d0ccfeffd0660cf7d52ccb027f5", + "959": "20bc47365fb175c4dc1ae198ed2547127f8675220c4f913b9f4d75d634b948f0", + "960": "2aaae6e97c57ec730bc921ed6a2c4fa61a486e7ace22b8431edfbdb6028674bb", + "961": "4febb0279feed890f3b756fec7d3a3c5c356affadc23179a3d5a48945c677d3d", + "962": "6461b98a0dc9bee36ba256aaa0f56b563fd363218e2b2822912090e6eba83241", + "963": "c9bbfee0e07165cbea43561e74d1a1e793c3f4de95dc9ac06986a7c94fe10ec0", + "964": "20c35c4b0ca3dfc9fe6b0de08afa784f8bf0c18a537da11e9a5dacf3a362b136", + "965": "f4f47bfdfca5b6c57270ddcc1328d56c37ad6cccce511cfa6c5fb53e4015bd8f", + "966": "ce7a33fc308837f2b1722c60b94bd17c5416cb10eb606a6c3de0cc29a6e17ea3", + "967": "9ce6b3b9a8420d096cc35d4eb43ee1049469c9ef1f5e8d53271dc1c1d6eac1fb", + "968": "5d177839029c27b05c8b7992a3ffff450af65efbaf5cea32cde82614024ce3ca", + "969": "819651bf9fa27b296b16c28b0ec647fdd5b14e6563e54f09ead12ed463d82ac6", + "970": "f379038d652ed6c88c3e410651eebeab6726bfb9e8be3261cee1989addba9ea0", + "971": "a2ed6b95fd66f4b1c206507b2d046744b58f779b2814f6e4a9da949388f12106", + "972": "3133027d93ba4141382dc86e30e79b1a03fa0474dbb66edd3e7bdc9a5805d2cd", + "973": "b315b80e2119207baa93a106136c5846857c553dedaf14d6a10a2e1ae3c827a2", + "974": "7557605ad0f8267a48ead835a7260e8efa8beb75c4ca2275c0a22edf5bc406ad", + "975": "51619f85fd5371f17f5115637cf24258a2f71c87a51c41be1377b54fa221aa47", + "976": "7794c2395c952dfd3abe40c36bfc778faee9c02a07f5ba86d76909c462620e89", + "977": "610e2f655b63f91161f0afca50382b4526eb8f92827e2d0a1c5d37dadbdfacf1", + "978": "361ce55b3d718be4583d8f337ee86d4abd87645d602cdca8a8186b11d56422f8", + "979": "5dec0b07fb1181f19e1c0b45089b88ffeb890cfdb4332bce0d841a6e951ba590", + "980": "67ce4920de47d6d25940a9f7c0a40f5b767b81461a3bf8a5ba58403324aae030", + "981": "4e941456d4487eb4a335375f7d474926428c7dabafb4cfa256842faf0a76431c", + "982": "575609e148d44fe6b3e66017a0ac58427370204269911a2ed1dd2f694397c8c8", + "983": "64a51a2f10060cda5e0c300a7e04a58f07b1a82611d2f84ec067bcb68a110648", + "984": "8ea10152c4508480bf490071f3323dfc56f5ffab7799fc4406969e856c2b5830", + "985": "fa1492c0245ec62c39184a723bfa0c338e5ba9d30e7190584f19063588632bc8", + "986": "8f7f9ad039edf656918b60c2cc5699daf8300dc992b05220cf3a4d2d2bc07bc2", + "987": "80f738eb99d46c27cb7d6c21a5586654cda7ea9bbe716a8e3efd7aad9e03e021", + "988": "40a1c80594dc4fc5c3f18c0e2312b3261686ed2905e7a9a8ba8b7ce07ea91c96", + "989": "5765d655c2446d951aa7f6242d521d9c5949b9f06834520c3441fd6b9af5784c", + "990": "d403339e39f92564e2804ac16d91dfcdc337d6d739b7393c5cfe1a288ea8d936", + "991": "a4d48d5fd43cdcec5a21f6b762d7e44074613e1d8192ac353818784b5443003d", + "992": "bf7f477f7e9fc8d5849651231313c80cda655269b01ed81e79a10000659d2a51", + "993": "b3c2f478bd7821cd450e6690c9ca661429c7de6e6025ec16e2d3d7a5e9b7fb31", + "994": "f089ad856c6da3fea5bd317a52f4dea95ee159a13213666ff4179d4f6747e4fd", + "995": "359da0ed1178db2bb78004b48e6ca344dc62663c4cbc306bc3048de9273c745d", + "996": "9860a2fa55de5092b202705ba84774f5ee301739c07361e6c21360ccd684638f", + "997": "a5a105323fbaacc1acdff82ce306f698552644a90e242356e6072a449180d76c", + "998": "19535f697d3ede3ead7bc86f29c12cb3430e7658d530f48617f6ac5bf384e30f", + "999": "4aca5337294efdefbe319cb3f47ff63ceef63af8d619cf77f94e25b016f73e62", + "1000": "ece4f09670b2dda58e110176e3e8685ad57f69d233c3569ed3b6cd0339d6e161", + "1001": "9eb75b96e2a622a204be19e720c44729855d31291a46ea97c1ab2c5717264be8", + "1002": "75ba8ebf096e834e0a2a173f86857879706db317cd9d3b13d0f7e0b6b72a57f4", + "1003": "defd066107d6cc6c0226c9d1f4af3bd9265c01eb13ec08a70fbc068296634fc2", + "1004": "9605b718a8618500c465c0f34e610107a84424f10f8467cd6d23fcbe566474d9", + "1005": "285fc96408c9874d8345a205d0976719bb09528a58cfcbda2014097e9b018a50", + "1006": "5cfdf984a8766ff99b0809d8e0ad236a03a1dbcecd098fce57d46d947062ccb4", + "1007": "dbfc97d2351ddbf01247b9f8c66a7c298b01c5dc7c48d960bd71d2d8d82e0cea", + "1008": "aa18e90efdc8d665cc9e4454622b7ef63c8818e315bcfb0f6e89b74dea43c425", + "1009": "0dcfff139ddf938dc5e2d3f11564ba684485919b11aa98b5d809c3c21734204c", + "1010": "3fcc25261f62ba8ae0c9f412724542190562d310556115606088d1eaf2d862a4", + "1011": "f426868415e13c2b80943aa8982d99fc0ca62775fd5a92c44d2f85b7047a0bf8", + "1012": "be3f051570460205a578292820457615bb8eb51ef61bac055497d5ff3af16dbd", + "1013": "7f681841dd30929e2c69e0c7d12a42f72459f7a6720be26dd6e606c57c769530", + "1014": "4ba12b80276d728111ee5cd6d524843f1c84831a7b48aef80e93165449829c84", + "1015": "13f3cf6282f3f05eb21a339d55cdd38863af3b85703124c14bb580bc16838bf2", + "1016": "3b236bd2b36b951eab0674e6959091075b9b834a3c960e8a91ef0726ea66d9dd", + "1017": "84b72ff56523f7c335e0551b1b8843396ae6cdba774aa51679aebf440e7ada44", + "1018": "2cea15c21afcc1e4064b49dfedd6bae65395bac19007f4f436dab91d0a87069a", + "1019": "ba7147d98fe12063f19ccec50379f6028df64f738fb401cad072ff90589563b1", + "1020": "3e34b5a0d604377132ebd93be5304c63a4746e347eb819911c35e96586a0453d", + "1021": "93a54349897944a9285eaffd37a6fbe30197bc843875ef2d92f448f292a03033", + "1022": "73232279d195130a3b91119dfe4f26b30b88213b7633adb057582a21d25b39a3", + "1023": "fe3a30c55dcc0bdaf6a5d75cbf50636af9de2b8b1ffe7d69d24a45730886d3b1", + "1024": "76346221d6bb74a66302d30de3cc5522a8425b299e25b99355576d5de6366106", + "1025": "ecabfd5a7c08404e21a0d82d410d83f89dfccb4dd0201494f72b9f79c1447625", + "1026": "557da21144e502020f23bcab020f33e776a6aaff48c2e12065b6902d9cc6665c", + "1027": "9f6ab9610df9f3511f61f33a7dfd715a39686f9e266045cf95157ed4d5267c4c", + "1028": "956dd57f7d137d686b7c8c234e857d76ed5274be06564a461a4de69bc2e2c918", + "1029": "1d36fffd8735d9d904e6424ccc636b1a30dc9149683bd9e76b1344020e324ba7", + "1030": "7e074a94373ae358dec1e6b503d36e544b5d1f6578c76e0fb11070b56bb376e4", + "1031": "fd4fe37dddb69dabb6312ccdc248cecf62b81c230e845c9993a45ac01731a93f", + "1032": "a9db8b9ecb93ff4c73b1af5146498fdaadd4adc5730d2f5f116d64121c27974a", + "1033": "da1d9028913805dbb365a3d8f59f8f639206c6f181c8da43a79c7e01d82c4636", + "1034": "00c374fad28dbd3532b6236b406aebbdffddffbb197393e73db3c3781fd07475", + "1035": "dcdbf0c11c74b6b9572da50a144f79e1c7b9d4210a1dffe15fae8eb66b06e973", + "1036": "98504766724d24b15f5e6994c106afc7d00d01dff39f5df356d00db3b7ae1955", + "1037": "bc3d838a4398a2eb883f7215c8f596e17c93fad9865fb4de219c9595582aafc6", + "1038": "87b938648de71bdc71e3e24aaf1406f491cd47ead8a4b7c9b9d2dfb8618adf6b", + "1039": "c7f07a8e71fcf31d3a3248c5ff17683f9fb3b949c7633fb40ce9bb36c69e7714", + "1040": "8f8f6d7c5c184e3968e31d3d37d1b3b8f54c98316ee1d1cbb9269bb9b23447ff", + "1041": "b317c327573e0894fe2c47e4be99e4b1837e018edb918c0b62d86b73d2eccdcb", + "1042": "21a08db458371a56d3cbe9415bcac17d1881c842df15b341228fb56c52e84738", + "1043": "84c4a88f589fdea29073bbf37e1850e9abfc5bd8fb20591519c25e6ef3028a48", + "1044": "6fa68f1982bdb6952223804dbcf98a61c068cdab18344c7143fb1eb54e17a9e3", + "1045": "ddbc9c2a49c184849e9796265a82c0a4728c8e302777ecb27bc54c7dbbd37ff8", + "1046": "224f88db8e75225c3a5c84016943655f2a9e60555aac9a19eac8fab5ed7343f0", + "1047": "708d88f4a4c52c3fcd729fc20129929e8968349bddd1d9d801c95aa6844eeb91", + "1048": "e821932f4c21cecf8ba3acdb0c907b3eac56f4543281fecf4c23076badc417dd", + "1049": "329bda07e73b81adfaf89eac9537b953fd9b7e4a279f4c4e4e0395e23910c5f0", + "1050": "e498fecf773f62290bae28ebefa5f6f4708e418624ccbf6c1bf0b2e376ccdc7b", + "1051": "7a05c5e1850d01af8d3bc0e6c0a3e2a53e297c8869ef99f9eff412d712d9055e", + "1052": "a2e3f3777a455be1816ddf40169e9258faa843602961044381be533e469d3fc7", + "1053": "fd87c7f2f96b778e6038a29a2de2389ccf185828a771b2aa53347990ea82ff2d", + "1054": "b26f161710804bacdf1b975e5135bf65cf4f3dc67f1288d6ce3c70b3036fc36e", + "1055": "b71161f58484e39de01ed63013224856a20146c8519f58bc58addbd8a6fd4a72", + "1056": "fc50de62547fe5f26cb01204da7175b43fa160449c34cbf1412b7a3b67edf825", + "1057": "466b123c8995ae47694287a297efad4208bd150511bfceca9c34bb9cd0245981", + "1058": "d69b3f915fc2a3ae05f9bee18f68fd9d3cff56c195b6b63e125f9a337d2fb199", + "1059": "d2743922eb5d9033cf10176e5553a41bbf35e674b05e7acaf9d59a75cb4d1fbe", + "1060": "39d1db4cd320f5ed0faf65d8280eb5bf3f80fa7e0948677c8535964fbe659f16", + "1061": "a9ffb69cf8d55ea6ee105f8b536462450f121d57c91f97c14a463638a710de54", + "1062": "ccd187510d0cf98ce5b54728b68246eefaf47e41e3e537984b92c6f1d9a78410", + "1063": "9d3147537ffaf339880d0782c4a6caea7844b6ed2faadb4dd1cb53d7c859d353", + "1064": "9f44dbdf688f7121b7d1a9e6527563475b086e4226db555f95fb74813a532d96", + "1065": "e7685b907fbb72721d3c668e2346f67f24a2ba531fa63e1ad5a8378ddceabea4", + "1066": "13f367d4461e8f73c1debf559d34a3c059d57bd4ed8b68f1af0d3216099898f3", + "1067": "5139c2845bb95845ec5dd93f13184d3bab740a274c87249402a2517f3dd1ea15", + "1068": "b789e3bfc584405aa0d27ced17e99f75cd4c31b2716943c9be055fa9c517e1ab", + "1069": "bb85069c403348edccd8e6ecb9adf27da4c00d99d0767d5e0f0166d3cd53ec1d", + "1070": "7fbf52e6df58beb4b819bd9ad43fb4bf0abfadadaaea0c79fa6dd7ebe63e65e5", + "1071": "163b572ad8b0154cafc92db9e0471239442a50e3198686086ca4036155b04058", + "1072": "ccf554cb408f8da202eb9ea6fc2d6daa136fabd0c8094103da0883f6bf5c113f", + "1073": "f2a8232a67e27215e089ce43b40d3485c69c196d33ad46adf49b582fad0605df", + "1074": "47a083bd1b212730edf63ec5fee716ce48dbde79546f79712a9a63bba018ef70", + "1075": "286309904b74776345b0a4adaa255072f9615d2c59c091bf068328b24cd5aa0a", + "1076": "4ede1f566a2da63af1f64c7d9a13fe2668de193cce577fe8d47ff4441586b98c", + "1077": "c68e02af24155f7bdd8a3b76a455ff6ca219f5f014967cdd7f33adba9fdb1578", + "1078": "5b6f9172a680a517c9bab88093d1ef0f6df2203c6b67e6374746791032ce1fee", + "1079": "1d4cda8c630536613c8a164bfa561d6d8a417a25ce63cba55788577c8aba273f", + "1080": "e816dc358bcd015a575e656ca36b1426ea9ab3d0714abdcca858571d13fe4bce", + "1081": "7d9f4ea6b1abdb7bf73cec4ef57f8274b1b9b0c2f3cf1116c0ed17959ea276bd", + "1082": "6d9c46a7d11522059088613cc42088984913950d23d6557843e334efa946119c", + "1083": "03c191ca15b3aeead4bb55eff191c04fcd293c6f802d93a10c51a790249d2f1e", + "1084": "3793302c93d45cc443e0fac28e20f32c929e22cbbbc1ca2e8e16fb8149278997", + "1085": "11dc87516b33ca40c02d77278be0d7184c6c0f71881676c49697859823fd79f8", + "1086": "1d62460556614e6a3a7c1463f2eb91e8626c0db462df828feead726b5ae17e5e", + "1087": "c243418ef9b2f56641552c52a4912c7479cbac2e34a32325f3716f4c398226b7", + "1088": "15baaaaf6bf270376f72b080f8ff171dab2fe5a567818207dd50a3a3aba9ea83", + "1089": "2154ab29cae5d2655605fe1a18d37b8829617aa309325c0757829afec5c2717c", + "1090": "ac372809d327df3a8f8d29a83fdfcd67579215386c3514eb5483d581a506a69b", + "1091": "2c8104292440889781c9cb45556b91aabc0f8b8b29fff17f466f21e5d00a878c", + "1092": "df9d6d209df70476bc52031905539dd3c394eb76842a3f4ab6e9a13df2989f60", + "1093": "dc1b3c55e49b437180eccf859c911df1ed8201e22ce1768b2f8752daa5ce98ed", + "1094": "65bd95ffb59f3173bf0ba7c08665cd9b958ec42949168384a8d523480ed4106e", + "1095": "f3e6676b8cc88cce626ef370e978684615435ef6cdcb80ea4374d86774345daa", + "1096": "5070093077ba5652f5913123ab4dda4fda0d4ebf0e448587d79c6596a29088b9", + "1097": "79fa2e5b4e0816708abb05a05be304743e7e2ef094be26b139f1cfa753aea485", + "1098": "315bd57746fa7c051daf0dc80c461a1df09df98d8d43ba88e47d001af79e8b66", + "1099": "3ac997419c11fa57db2383015bd8ebbd2a54dd5b656965d2a2cf6117bdca0525", + "1100": "be6e2afe7b9ed50307ecffc6c29c087d5e665142359a600bd4d875fc2faf87b9", + "1101": "3d5bd7e9ecd34282c5bcbe0b08a59df05ed568f799bf8ef2f9e983601cb51836", + "1102": "4f76db6fc16e66a2133c61a6a522d32afb5bd1435000ab0d1c4b617ad4f5a13a", + "1103": "12de450e09f901feb1384252a2be42d93297baa03f9615a81fd8ff1e8a816ba8", + "1104": "d916a30218ec3a2eff79c9621e28e3d23b745e0dd132a33a02f587035e9a766f", + "1105": "02b16cfdd581b4018dabad8f4a4d0dc6a23d0ee64b65e84e0bd2b58a8ac2f04b", + "1106": "de3724e80bdec0b9aa8f8909a4f46c33158a854203aaf2a6131674525ca6982d", + "1107": "c891bb307b7b5919d3d19addcae7f4789f9962550e58162918ab6dc4502b63f7", + "1108": "9691d54d45e6074c59f4e0403b5cc1e6dd696378429701f685d442284bb17412", + "1109": "5204cca3a696fd9e947c949aec737e7ae35194bced04d116a1d028287554f9a2", + "1110": "5c54d55f12010e1f25216625a8179d5c85264793562b72435b81cefd201d1909", + "1111": "ccc06c4cc9e2b0e1949aee893881b3e3d7a15e36b87b41086a9c43887d880b5d", + "1112": "0668c6508c6aa5fb837c7c384a4fd729633bad20572c723e1462da9c691e3c40", + "1113": "7d228d029d9e0a9d99d244f9ab24341ad5df3825d46dcc8e21b4b293636a08aa", + "1114": "75432cbf329f75a6f6cf9c899c3c485e96da5e6270011d2f97500e3edb349bdf", + "1115": "65e8a8dbc1f738482acba0b28a3db288641944d43c591cf5d20ae2454ecb3156", + "1116": "1d6ff9569a5407c816bc388d470d4615807b33b1e9dac241da450e0f83bf7ffe", + "1117": "29b804a79c81ce20a8cb25b5e95f7e28d855ee091fb6563afbca3f7aa79335b0", + "1118": "464739d9c69ac7376a14a33e771eabc5d5aa981ee1359ba24ae3c9c5fb733820", + "1119": "2f36c675a723d7de412dfee66e0a5662fa39537d1b60d5799dbaccb3a97c888f", + "1120": "87b1b647080603ccc23d9ac8b2a8d7f51c7fe54aab7f0f593e5e0b591e15850c", + "1121": "35805ff22bdf798ae3a31702fa8468e50874ab6b614c07457c1abd5dc5e8e978", + "1122": "56c0dfc93bcf5c2418f5ccc6cc902eef31396a8ccb7bda71411c472e825f7650", + "1123": "41de7a0d2227509fee9ec03f2293b9e80a1752e3e62e77fb814c620e91f227fb", + "1124": "435a4c8ec3a6478ee5dd8346f369e1cc32e032452a04032e71fd099c6364756f", + "1125": "cee0c4b134fa9236f668bb4d6f7cbf04e2fcb7429c8695044ef556745d4576c0", + "1126": "03a67eaf0b57da93696e930df6ab8d8122e2b7a17a33db9fdc09dbd7cbbcfe86", + "1127": "c1521db65c3b46801b68f16bcbe78d9aaad50740dd60e86622089c6269481930", + "1128": "6c5fba3048fb2e45ac2beb2e2ffa8fbac5878d9c2d7c5daf127262ba7babbb3b", + "1129": "88abbbda8afd6e525fe1ecd8134e0d242c8bdca972c7df567d9d3c54c1ec6466", + "1130": "85d671d006b2677897398bbd748cac4e5a6ed6c6bc27bf35ac57de8a7666c23c", + "1131": "f008c7935acc8b02871bb008d7a737bbe6e5d6454570210692967b20eaa7aa07", + "1132": "f1391680040d23b10c64b44909f09ff6a40d3b9e6469131fc04d1dcab7cce27c", + "1133": "4a587f2b9da68d4b12db4dfa4e3bc4532b22b76fce5308c49702653fbab19ba8", + "1134": "07bbc624dd83c89644e8195e626ed2c1f2d9d94c5d854d95cb6598c3478f3369", + "1135": "b202200f6b25a13759902fb2c75677364039d338e055fef68fd482b8c6513d6f", + "1136": "71a61e99047fad6cf78fc500a78a27a397a0ba735ec3f2850b00141cd4349726", + "1137": "b6014a139d33a0432920a317c918d9590db29380095bb00989e03e6287d93965", + "1138": "c667267bdad5d3ac819e8fafd5741349d9a702fa792c3f87e1f5105ba007cf64", + "1139": "a9dabef8cee07da978dc39cfa79376ed6033ed2d28e763d356a22cab79ce55e7", + "1140": "7f6c9805529fe854b4e254a83d90c5d23655b109f8a83eac52dc3ac3d5feff15", + "1141": "75575f542d44b59afb9bdb09760a1604eadb0830c02a81a86d976be9cf2c5f20", + "1142": "bdd1cda18660dff5f21174e1abfdc8f3a4451fd16524fcbd0c8e2f99c54fb8f5", + "1143": "ac40556f2eece42ff8a8e25807fb4ac561540bd051a25b47615d3412f981b9cf", + "1144": "12456bad34ec50467a59aad0b25a84da28eb3ac841c9547b2722fe7374636daf", + "1145": "1e7d434dbdbda1e0b882bb077d1ca48a984fdccfb00def9f017492fff2c2019d", + "1146": "50037a6cd78f4dbdf0c3b4feb432c3379bf6c40b75c64029d821243d3796cd58", + "1147": "d3913df37b78ca448d1ee097c40a7ac6ca5b74f1162c283c01b02db7ad48ef55", + "1148": "a28d4fc744f919bac9f46dfefddc4476f7944e9be82a585d355f4650fe71df9c", + "1149": "efc517eee80c3baa4ee85fc0b713cd39f5a18aef54e1624e5d2b34e6a268c8a4", + "1150": "6989cf5a24c701cb9dff5fabbbd0b2b45077d63c7f07db23f6ac1445045e2a52", + "1151": "1b8dcb9396eab466b1f64d1eff772bce53dc600f52bca7972a3eada7d073d3d0", + "1152": "7fdd79d501a1b0ab03d269b6211f923ead226a7b742e875bc8e69d7254cf406e", + "1153": "78a858aeed0c2829d443e636c622a0662b6d089dfc15645b2b2792821b271a4c", + "1154": "25a802ac057ed0cbbc2a7534f77fc982801b9b1de23706ca0ec925bd39e0cacf", + "1155": "22cedc15a67d5e3a722a84d8e59688a580884cb0ecc689eca1a135a8c97976ca", + "1156": "b5cae0cf19aa97e88fd73999169477f07ed46f0c1f9374fa192839859373aec0", + "1157": "bfd4e4fa9e185f44c31b0c6131783a7473d3b391fec5adc25079d84c43818aa9", + "1158": "98897a35cd59726ff0fbbaf1307785e997588f6fcdd74300bd61f8ae8acb2e44", + "1159": "4a4d7ed669f6be8f2cc01e40199c6b38e87503cd0bc17307cfb3c83a1b2394de", + "1160": "97a03fa109aa53dce3fd36d7f2b28678b765718f9c9142c418ff04555b491d20", + "1161": "9391fbb49ac16f9a89cf1f40b21ef2058c29b80dac5b3a87c8a99259e3c9efbe", + "1162": "c02a649e94c7bc8eacdd74a7cb2a66e311535b85b25ccffb112ccbe986888fd2", + "1163": "c0143327332db02a6e76273651bfdb8530f39cf3a8ca641f227dfb373c662840", + "1164": "ceff433ec66e55657124f8f7c49f3826cd000a9691a1109eeb18e6fd3d2d769d", + "1165": "9349790196acf11c5f8c0443ef550089ac5a3716922227a27ed14f55ae58aacd", + "1166": "f3e52ff6bfb601acd206060240e6825e8667ebe0671cb2863cf792d3e221fafe", + "1167": "5f9ae302ac00e8bb455831c937e6179234da4e2529cc4f164d5658b4ed62b1e4", + "1168": "41726ea720a7c32ba2dfc3b58c8c97787ffc4fb02c6322c8567de7385081f487", + "1169": "49594596a07e3b4723af86d750eaa54d6ba8d54a4f44179e7cad577cc329cb86", + "1170": "58f2ec15f633cb719de5bac5c32a2e05c191b9b635d5e69fd7c36cd5c7d10f99", + "1171": "a9615d72f9fa31c03eecbec996d7dbfd2effd6b70f4f54fdfc397c375773a029", + "1172": "c8e8f70d429ef4d3ee3db9068e2c0139e23e51b2624f7cb26f0268b5c4c5b54a", + "1173": "1cd956cbae9f9d759fbaf5163b0a5ee388b4b1a0b508ec5be5d91893e8b8e2b0", + "1174": "a76638946a12c489a5492bf9e2956d9be7f3dc92ac12221a2ac756e1bff6cf93", + "1175": "df8f3b4b89c807bfcbdf166fd95ebecd46cbb3e0013bb6ba0db190f53f6909ae", + "1176": "363df612fd47b9f574183b3d3bb165094df81a4e84104cb987f21806da49fe15", + "1177": "7acacbe422d6601613bffe09a96ed39e0924e00813ec289315bc8652ffc6373f", + "1178": "68611f520d4bdbad02b0867a1fa03d2f47ff5e5caca78a1d910c446d0fa901f9", + "1179": "85813edab1fa8052f0ff362f3188e6141f82a14bacfea6f445cd5a2b987b3bfe", + "1180": "da1bd9dbbf054bd1890d5aebc1ed2f1083e607f6aba07aa571ac8e29ab4c958f", + "1181": "857594d715edc91a39b0c2a9dbd5bed7d856f41b589ff970bad72ca692919d7e", + "1182": "dc3b6f1543dabf99f28894dbb4fa60ae5f2cd6fde50d76bfe8b91d7f417ce3c4", + "1183": "189462457a4dc0095f54ab7af9d770dc9e044928b2a75d2efa12b73e593ec757", + "1184": "357ca6c79f9681a2dec1392f17d896840b0d09e535136f84141e5d2ffbdda1f6", + "1185": "b239c762f5094e7b1c9785fd3590658aac6f0a03775f7f1c9423672bba07bbd6", + "1186": "f5e971ef68c1a3e654cf18da48d88a2e64631bcbbfd49429fd21eb9b6b604b88", + "1187": "8d8fa1433faafdd0ae3eecd3ae231d79657d4f49accf5c0ce3b977560cdc3b0e", + "1188": "71ed7333f9e37dee89075d0f3cd45a49d46defc0bf6a2cfd6a3a5fa61d071d67", + "1189": "e1c9b735135a0335983e9018d58972ef0e8f1f450c506448b66624968128dac7", + "1190": "eca0adc20d0ddf77ef2cce8ad764f6e1f674c55931035622d6ac24a960e75564", + "1191": "a0e55dd2f37a0d0d8da96a3f9891d023ef85a9e85d63d3f131f52fc4a6065b90", + "1192": "449eab22e58c19e258406410d55d509eaa00c730ad63728d9856fc933af4913c", + "1193": "544ad49255dbf301e493a07f5c57a5d5ff3fdafbc7a11e3ef666cea7a1b3286b", + "1194": "0d93500c98e965cff37d4ebaf52b81c532383f45c89c5727bb92507067ffb364", + "1195": "f43d663ac98b9c1774c0ee5ade40cc76e99181a0e8aa74f61ae93de0f14e616d", + "1196": "5ab3bcfcd3483dd90435703e9d426f20ece485a22ab6c310f693ecaf0ec4f03b", + "1197": "92c494e76348fad2e56094245a1353db92c2e22141454930156ec361c479785c", + "1198": "a346b568646297aef86bc2f4d7f672711b6dbec48d46cbc3a3917ca4649b606b", + "1199": "6ceb7a613958090fd04ba5ea2407ce1aa86aeb6235e0c0dcf6ba5dd49b7d4c4f", + "1200": "0e436bb625dd2e5c1095c99d1c2ec9d0b7892782d4697064de3eb893207a0508", + "1201": "b183d26723fcbeca81ff16a86f8bf4f379e48ded54988e69794ebe5c8163ccf9", + "1202": "fbe26b5692a12a983ce1437b92da3402002e30e57bc44e06d273bca7bc1d577e", + "1203": "c081c9cdb84ecb4d89bb554f07c36225bc9e73988dbe688feaaaea08eeabeea6", + "1204": "985d1323f9fabdb572691cb7caf62a65866e9502b0042bbbafcf1a9f6ec35bc6", + "1205": "15188bb75edadb6db155ba06b3a2e72a50d275f4e55ac7014b1a781e409d3c08", + "1206": "9a023ccf54b5d9d8f6f1852874d54a0d6fcc88ea36c3a7e3c68d57d0616b97a9", + "1207": "1dd897d409a22ccd60e8cbe06245dc4ea4586d4378b92300754d9119c7f29a45", + "1208": "7c9128d62b6316ab352bfb07c1011c82ad5088c6accca86c90f39eba40fb2421", + "1209": "468ff24c32d4bdeec4776c91c0a9be0f7a4b500d95eeb8654cda168b98c9fe31", + "1210": "f38a929153bf158ee96b944331051251423b4e83a469227d38ac923fa2c2b667", + "1211": "9fe83c7fb519469a59e3697fadab261ab4b690125ba5e7155d7fe122592f21ed", + "1212": "8726612ce412f0c45967f7174a653f9269802dafc020442041700132cc828ddb", + "1213": "2fe040999076fb17ff080ebb8d6163ef3aaa8df47ccc1004bd84e4fb664404df", + "1214": "2cda1466766db3a7ad76861f522dc0908d5af9e4a119750f3e1171d85a4e05e3", + "1215": "d56adfe5858ddfc53b6b394667d839e06cac489a61384b32b66864ce79408c5a", + "1216": "2b1a1be4ee74f268edac0d09002f002bdbc4d6231cc94e9a6e866fb165bcb4c3", + "1217": "155ba62402891588755e7106d1d900fbd01fad7ba9764245f89878e14492833a", + "1218": "fb9dab594086c7de1e66a06a989d53c4ab07b1ec245968f3eea507757a2f89e8", + "1219": "6467251e2de32fa5b9fba1d2ee35509ea625c4a0a9450071cdafe29b4084dd55", + "1220": "80632d6b8a237efa4a1274e3495212f129bcc5c45b7652d21652aeef4bb10415", + "1221": "096fcf02189e34607c58a7775aeeb2e2f3632e8004cb4ac7a7782cf851729714", + "1222": "89141a0d06c7ba61513ce5d08702765516ed3cd530e8bc9a462462c71c505f92", + "1223": "f72caaba8b06f106e9fb469829e6454f029f47decebd965420e9309b696bbfa4", + "1224": "477f013db3b2c4175a7bedeb050b3eb36885d779c85e489b093d63b60d63f323", + "1225": "3ff0a4ac97f4813675880a8e33d86f3454975f442de82ac458da88ed51795954", + "1226": "d2f2a21af72b4ac9fc97de150266a3ff60d0bfea535fb67e7a5cbe56bd4ff338", + "1227": "1cedbba99c8dfb55d0c11845a6a286b2b6cab1c0284c39f537f25ae13c50062c", + "1228": "a20d50dbfa9d2ba54cbe9f990a263185ae94428e1632c49dd9cfdef992615819", + "1229": "2e344e859b39b51774aca40ab8762bc9a60f59c6dd5acabb7ff1c61c62102554", + "1230": "d000595932d749acb8796a3b1d7986386fa81113c5fddc9420a4fc06686e854a", + "1231": "c46331b348785148c361ec204376c6bfec816c53737aafbf8820d80aca7be3ec", + "1232": "eb2955e48be0446bb91ea956a3619257717c1608daa837c79e1d88501a96ab2f", + "1233": "b0d66ab4e7835d71f12756875957ca638bf6065ea103bb9f4749c2c1e102ec34", + "1234": "d68957376eb6ae69d174ce259669bb622baaf04a8c76c0123e292391511547e0", + "1235": "3cc27f0d9d213e22426738771d6c74014eec9ca3033c86e412ecde340919f42f", + "1236": "12e16899dbcc2e91ca2c2c912555f3b9eb8b4cdfff5c21c2478df6195faba681", + "1237": "56a7e5a0d57b9cc79f2fba2d246fa24944c3ebb25f3462ec1f3a36bdf59c5c50", + "1238": "bdf841d71415597ea4cdcb84eefd5ba4f991b40414d2eb953faa8d15006af52b", + "1239": "b0c775e83dd1a632c05cf70680654263d4d796c752c63acebbd3f22511a48db9", + "1240": "5d6e87c3f008220c9c01d0f441600639eb28aec45d7afb81c0aa070e29cccd56", + "1241": "e65f9ef2175c9bbf3ca91708890f224aee4c518b0c14c901c2d1426034da2ca2", + "1242": "c5731db1f8fb73407f6691154ca4565438cb00e968b9578938269edfc2aa1949", + "1243": "dd171e519f6740fb3f673197ae84d05e7ab812fa1b4ebf3cc9c29f47186a44b9", + "1244": "dd50b8d147d4027a2f2f919d800a3ac6ce7dcde06487edf552cc32b1ecb10191", + "1245": "3c46fc9a7ccab1d491b0d1251ed2f3afc6d5ee2e705d9b331436ee2bacdee15c", + "1246": "64dd597c48c00421fef9481b54f635c090639e516a11824b8a5df2c0b893bc61", + "1247": "7d62f107a38a7d3d790257351fc33fbcfdd7088d0d1ae060354084458c091b67", + "1248": "2a5dde30179e2dcb790646bcba3bceb11eb352f1d81b3c740de240a7eb5d86c5", + "1249": "8b87fd21287a42877f0a6d19d4667ef113736cc78301a81cb875ea80cf53be24", + "1250": "172e7ce7889e6fcc9f73d65f0c8807354272b9463b5291be40edd417fb1aea50", + "1251": "1cb2643f722f322545904c12498e16377846e7918df9a36ca4c2cec668287c89", + "1252": "37fececc86dcfb7e9e068458396709cb2be208c2a9ea85f3bfb0f143ff17ff00", + "1253": "c4b9c51158342c618e8e8355f9a8d586f0c5e76ef532c6f46fad13d12ef5f848", + "1254": "c1f549d26106ec20ca03610b00e2584dd324e41a8a7bc3afaf0301953906b046", + "1255": "a5e29009a2497d68465bbc45e702ca27cf86bfdc8b45fabb143744b7f69a2705", + "1256": "cfaa9640be8099d8d5a181cf40bec36080be715b40809e815a50c6cf501890fb", + "1257": "9787f80ef25f14d30e71a6432f4f25af780f1fbec93c94d8e2ab630ecc404102", + "1258": "2689be85eab3bb656db29b0fabd917a9f5430db8156c7ed1b2744ac1872ecb98", + "1259": "ec5177835c9de85bf8e9cbdf1280743d8fce2e99ff75d6c8e2d24d4c12386f31", + "1260": "73882f719d7d868df6e1bdf7a527d22e065d8f0b5f903ac87096ca3687b5c1e1", + "1261": "45007eef8f36f66d240f4a4305dc2a2b700707a5d7186d9ba420a3ff05046089", + "1262": "59e42cd314755b771790cd0dfaa69b6e6ed7543c29e54865bd055e36cbf3fea9", + "1263": "ad43107ff63244e14ad8d45fee994ee03582dd56ce3c3f8a91dd73e5daa2dd3a", + "1264": "8812113cbdd53746b09a4822f8f56709368bb2e5332d1bcef0c77919879d4a71", + "1265": "4e3762c826ccfb49cac9beccd78796039d731b6b73def2a651490c9541c36fc2", + "1266": "c847f8c49a701169ac24cdfef4cf28664b54160dfb9296f31d7f63eb2131d6a1", + "1267": "eac6a8e0ad64b1c048afb4e964a7ed032bffa7c2a70bd90989eb0f4bbc5fcb86", + "1268": "433d6233d569f4728521a8b9f1a05782f67800786851869a85c6b76b9ebfe8d9", + "1269": "074ed485141d1e6eed97614ec80c5c48e1164be130f488a22d9ade25247047f4", + "1270": "e89dd179a83e2daa18ec4db5af5d1d10832fb3f5c2d0f9512b9ffefffa9de4a3", + "1271": "f9f7f9adbbd167175d708bec75248d2006e14f98abecc29a8edc76cdc512893f", + "1272": "7d6e71be80b576d01fa7bc5a0c71fd0c4635c9c75754731d65090a283b6c2a3a", + "1273": "762e76a191231daaf31c11d01eb2c90c7e339b0bdc5b041b8bf3abd0c885824a", + "1274": "247e5920c31498cf19f75ccc451fdc956143c6c7f000f5b316a34652acf898a0", + "1275": "a6debf46c0c3cf9562924edd3a36b4a90a44e84521f9963dd6b1909ab161c576", + "1276": "2e8e3bc1fbd407f8b5f1ba910e150768f39014bdbdbf492dc11f068faf9421dd", + "1277": "8162b79340e8c73409b45e9fc55db708143a56f0f9424c8c85caed446855257c", + "1278": "e298520b90e580518ad5006f2cd38861a3eb9f87c08ed88a6b703fa813287f48", + "1279": "a1e98602731a61e1209c857c7c082eddc4eda56c4ff01480fdabcd96693eb75c", + "1280": "e36bb634af630cae70015700897f55c3139a824a4916a433a04fc3c3049008f6", + "1281": "9f6fbce0b82a39030151b1358da58824e9d733a9cf6cdf06acc4b400aef78bff", + "1282": "1cb20bbe90cd827fd051d65a45d79c68929691ae35cf5bdaf098d925fc3df737", + "1283": "157fc156afb8b5c53b3572b87c5ea0986186711ec0c07728777854c14dea5dfb", + "1284": "46d17e18094908e3944803164ea524cb49ed086cfb984934d53e1e29a01df2e0", + "1285": "3c241ec693d47f086582c37b1c0dbb10f31d30b39961ce2ce2ea3ed40a569d5d", + "1286": "5f2f539543180d5761ff6b3abcb16f4779fa75af6641488a54882b61ad3eb313", + "1287": "31e87372ca2e1defa6d1ec25a5d51e3cee47a8b9450ac988aff8c3e4a513eabb", + "1288": "d8e510bef78f57256731cbd2ca55f0f5b33084364fb073edd76ce4e3fa3a5e8b", + "1289": "046590b336fa67c576f0cc0df11cbc9a699a4b62a6404438f4be3ac5fac000db", + "1290": "cac1be485f95bb4be21c14c5603eb0730bca548e341eb4f7a5cf47c93ffc477b", + "1291": "9a34be0e9fbece8c797a75fff2c7b6d6d84481cc40f74c790e148b71d35f5a53", + "1292": "86a0a29149f429338e51ca3e384d7deb5207da75126a091230532f1d6c5b786a", + "1293": "2ed928437e5dadb7267fbe04d51adad4bef04ea57e977fdd4ce28af8c141c0fd", + "1294": "fab3fb3c46ce23d37a51cbd72f9d5396005afc4c2a80a67f2445909fa9473af1", + "1295": "33ba546db9141259831582ee1628e9c20cf3e25f12697196b3c3bb793d9171b2", + "1296": "d7ad7ea62c989045a2ece11c41db1784ea10b16fa70028304a1835a3667a96fa", + "1297": "17760a34c707177a376bea5f22445a5cb25ca0b3236ea90cbd729f80ef5dfd20", + "1298": "4b1f094eb875e38d0e9928a53305d34354cc2773eae8e475f19dde301402a2f7", + "1299": "cc7a4a98e29d4ce53e9b21c815e86465538de5a7f8ed2f07e5cff57d7618b10f", + "1300": "0a026f60216096a318e86a43649da9c1b16422e918883d9ea59ef290355b974b", + "1301": "7b7144e48ff4dbda7f5e3f7aea5da7f29aa12ad3467081d0e66c1bb355eb12d1", + "1302": "045a444c7c6541b415dcdf256f7f8067cb05cc54d91809cedf952ff21b2ee75d", + "1303": "b42245c9f4e269984a52e567d7a43862a93b9b503eafcdb92e8b7f2ae854c1cf", + "1304": "105069b7c884e7bbbd6c27ac30fb1280fdde53df05cbeede69862b0fb3be4346", + "1305": "3fafdf403de797f561dd68095ef570c71f945abbde2c6a05abf882a59e7e30a2", + "1306": "58d0f21ef67f537231991622a50edcaccc29b28301e071134c629a72373496cf", + "1307": "c6240fa64bf7be5c50844716cc9bd2174e1a823f22a4943c626f78e328a92730", + "1308": "09f68770c8fe3f19b20cdd69f6bd53ee12952bec31201b4e7aef900d40e6feac", + "1309": "f1329aedf0e32f35aa7241f6bb74bc897b65b3ce3d47082c1da680cdc9a14cb3", + "1310": "6e721d4a5113f3067be0cf9dbe635ea54274a280fbf3c73d31457b4dac8c75ed", + "1311": "3c8fc5f3b401bed92b3cf3ba8a515bc4705ac46f2988525ee305f76295c10de6", + "1312": "17cf9f5f4d93b6a900d9e836950ae49be49543ad75e9cd683a0ae213ce3751e7", + "1313": "b8a6f0e6bd515113ea4719febca26d0c14770147db567ec30dc58499b086c824", + "1314": "4b648c1af65b2852f4c9ba2eef9eb26e663074aafd43c2ac2f05695ddf136001", + "1315": "e52bd3a00ab772638c1171dc5270566c6c277d5ec3a27bec43e0b0d3e83e5279", + "1316": "910e1acaba2f0d1e314a5969aff2d993a38ed1f2b4cb788e23f02268a0e6ce78", + "1317": "13b5d072ad663c6f26400383aded8bbd6be8e9115078b2639c826e14b20ad6dc", + "1318": "3bbcbd5151a8911eff9fc12332776eb85b59eed44d6a131550b2f0e5cada8d67" +} diff --git a/infx/benchmarks/spec.py b/infx/benchmarks/spec.py new file mode 100644 index 0000000000..9783e1b2f9 --- /dev/null +++ b/infx/benchmarks/spec.py @@ -0,0 +1,181 @@ +"""Versioned, explicit contracts for the first aggregate client lane.""" + +from __future__ import annotations + +import re +from pathlib import Path +from typing import Annotated, Literal, Self + +from pydantic import BaseModel, ConfigDict, Field, field_validator, model_validator + +PositiveInt = Annotated[int, Field(gt=0)] +Digest = Annotated[str, Field(pattern=r"^[0-9a-f]{64}$")] + + +def secret_environment_key(key: str) -> bool: + upper = key.upper() + return ( + any( + part in upper + for part in ( + "SECRET", + "PASSWORD", + "CREDENTIAL", + "PRIVATE_KEY", + "API_KEY", + "AUTHORIZATION", + ) + ) + or upper.endswith("_TOKEN") + or upper in {"SSH_AUTH_SOCK", "SSH_AGENT_PID"} + ) + + +def validate_environment(env: dict[str, str], unset: list[str]) -> None: + if set(env) & set(unset): + raise ValueError("environment set/unset keys must not overlap") + if any(not re.fullmatch(r"[A-Za-z_][A-Za-z_0-9]*", key) for key in [*env, *unset]): + raise ValueError("invalid environment variable name") + if any(secret_environment_key(key) for key in env): + raise ValueError("offline prepared environments may not contain secret-bearing variables") + if any("\0" in value for value in env.values()): + raise ValueError("environment values may not contain NUL") + if env.get("HF_HUB_OFFLINE") != "1" or env.get("HF_DATASETS_OFFLINE") != "1": + raise ValueError("prepared clients require offline Hugging Face operation") + for key in ("HF_HUB_CACHE", "HF_DATASETS_CACHE"): + if not env.get(key) or not Path(env[key]).is_absolute(): + raise ValueError("explicit absolute Hugging Face cache paths are required") + + +class StrictModel(BaseModel): + model_config = ConfigDict(extra="forbid", strict=True, frozen=True) + + +class PreparedFile(StrictModel): + path: str + sha256: Digest + + @field_validator("path") + @classmethod + def absolute_path(cls, value: str) -> str: + if not Path(value).is_absolute(): + raise ValueError("prepared paths must be absolute") + return value + + +class RuntimeSpec(StrictModel): + """An already installed child environment; execution never resolves packages.""" + + python: str + identity: PreparedFile + distributions: Annotated[list[str], Field(min_length=1)] + env: dict[str, str] + env_unset: list[str] + assets: Annotated[list[PreparedFile], Field(min_length=1)] + timeout_seconds: PositiveInt + terminate_grace_seconds: PositiveInt + + @field_validator("python") + @classmethod + def absolute_python(cls, value: str) -> str: + return PreparedFile.absolute_path(value) + + @model_validator(mode="after") + def environment_contract(self) -> Self: + validate_environment(self.env, self.env_unset) + return self + + +class ResultMetadata(StrictModel): + hw: str + model: str + model_prefix: str + image: str + framework: Literal["vllm"] + precision: Literal["fp4"] + spec_decoding: Literal["mtp"] + tp: PositiveInt + pp: Literal[1] + dcp_size: Literal[1] + pcp_size: Literal[1] + ep: Literal[1] + dp_attention: Literal[False] + total_cpu_dram_gb: Annotated[int, Field(ge=0)] + recipe_fingerprint: Digest + + def normalizer_env(self, concurrency: int) -> dict[str, str]: + return { + "RUNNER_TYPE": self.hw, + "MODEL": self.model, + "MODEL_PREFIX": self.model_prefix, + "IMAGE": self.image, + "FRAMEWORK": self.framework, + "PRECISION": self.precision, + "SPEC_DECODING": self.spec_decoding, + "RECIPE_FINGERPRINT": self.recipe_fingerprint, + "TP": str(self.tp), + "PP_SIZE": str(self.pp), + "DCP_SIZE": str(self.dcp_size), + "PCP_SIZE": str(self.pcp_size), + "EP_SIZE": str(self.ep), + "DP_ATTENTION": "false", + "CONC": str(concurrency), + "IS_MULTINODE": "false", + "DISAGG": "false", + "KV_OFFLOADING": "none", + "TOTAL_CPU_DRAM_GB": str(self.total_cpu_dram_gb), + } + + +class AgentXSpec(StrictModel): + schema_version: Literal[1] + runtime: RuntimeSpec + metadata: ResultMetadata + concurrency: PositiveInt + result_filename: Annotated[str, Field(pattern=r"^[A-Za-z0-9][A-Za-z0-9_.-]*$")] + tokenizer: str + dataset_revision: Annotated[str, Field(pattern=r"^[0-9a-f]{40}$")] + dataset_loader: Literal["semianalysis_cc_traces_weka_062126"] + dataset_repository: Literal["semianalysisai/cc-traces-weka-062126"] + dataset_entries: Literal[393] + duration_seconds: Literal[3600] + warmup_requests_per_lane: Literal[10] + warmup_grace_seconds: PositiveInt + trace_idle_gap_cap_seconds: PositiveInt + live_failed_request_threshold: Annotated[float, Field(ge=0, le=1)] + failed_request_threshold: Literal[0.1] + random_seed: Literal[42] + required_server_metric_prefix: Literal["vllm:"] + + @model_validator(mode="after") + def client_contract(self) -> Self: + if self.tokenizer != self.metadata.model: + raise ValueError("the pilot tokenizer must be the model identity") + if "aiperf" not in self.runtime.distributions: + raise ValueError("the prepared runtime must identify aiperf") + cache = self.runtime.env.get("AIPERF_DATASET_MMAP_CACHE_DIR") + if not cache or not Path(cache).is_absolute(): + raise ValueError("the prepared runtime requires an explicit absolute mmap cache base") + return self + + +class EvalSpec(StrictModel): + schema_version: Literal[1] + runtime: RuntimeSpec + metadata: ResultMetadata + concurrency: PositiveInt + task: PreparedFile + task_name: Literal["gsm8k"] + expected_documents: Literal[1319] + max_length: Literal[16384] + max_tokens: Literal[12288] + minimum_score: Annotated[float, Field(ge=0, le=1)] + # Independent prepared identities bind the full test split, not self-reported n-samples. + # JSON object maps string doc_id to the harness doc_hash. + document_identities: PreparedFile + + @model_validator(mode="after") + def client_contract(self) -> Self: + if "lm-eval" not in self.runtime.distributions: + raise ValueError("the prepared runtime must identify lm-eval") + return self diff --git a/infx/results/evals.py b/infx/results/evals.py index e66467e3dd..11e165b378 100644 --- a/infx/results/evals.py +++ b/infx/results/evals.py @@ -291,6 +291,17 @@ def build_row(meta: dict[str, Any], m: dict[str, Any]) -> dict[str, Any]: "integration_error": m.get("integration_error"), } + for field in ( + "disagg", + "num_gpus", + "deployment", + "recipe_fingerprint", + "point_id", + "execution_id", + ): + if field in meta: + row[field] = meta[field] + if "eval_suite" in meta: row["eval_suite"] = meta["eval_suite"] diff --git a/infx/results/publication_receipt.py b/infx/results/publication_receipt.py new file mode 100644 index 0000000000..fb5cd37376 --- /dev/null +++ b/infx/results/publication_receipt.py @@ -0,0 +1,469 @@ +"""Immutable measurement receipts issued by independently selected hosted tooling. + +The issuer identity is supplied by the trusted workflow, never inferred from worker files. +The caller must authorize the source run and generate the expected contract from approved +immutable configuration. This module verifies API ownership and archive bytes, not actor policy. +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import math +import re +import stat +import subprocess +import zipfile +from importlib.resources import files +from pathlib import Path, PurePosixPath +from typing import Annotated, Any, Literal + +from pydantic import BaseModel, ConfigDict, Field, StringConstraints, model_validator + +from infx.benchmarks.common import decode_json, require_finite +from infx.results.evals import build_rows + +Sha256 = Annotated[str, StringConstraints(pattern=r"^[a-f0-9]{64}$")] +GitSha = Annotated[str, StringConstraints(pattern=r"^[a-f0-9]{40}$")] +Positive = Annotated[int, Field(strict=True, gt=0)] +RunId = Annotated[str, StringConstraints(pattern=r"^[1-9][0-9]*$")] + + +class StrictModel(BaseModel): + model_config = ConfigDict(extra="forbid", strict=True) + + +class Member(StrictModel): + path: str + sha256: Sha256 + size: Annotated[int, Field(strict=True, ge=0)] + + +class Artifact(StrictModel): + id: Positive + name: str + sha256: Sha256 + run_id: RunId + members: list[Member] + + +class Issuer(StrictModel): + repository: str + run_id: RunId + job: str + workflow_sha: GitSha + collector_sha: GitSha + + +class Topology(StrictModel): + kind: Literal["aggregate"] + nodes: Literal[1] + serving_gpus: Positive + tp: Positive + ep: Positive + + +class Point(StrictModel): + point_id: Sha256 + bundle_digest: Sha256 + execution_id: str + source_run_id: RunId + source_attempt: Positive + kind: Literal["throughput", "eval"] + concurrency: Positive + topology: Topology + artifact_ids: list[Positive] + execution_artifact_id: Positive + execution_path: str + native_manifest_sha256: Sha256 + normalized_artifact_id: Positive + normalized_path: str + normalized_format: Literal["normalized", "lm-eval"] = "normalized" + metadata_path: str | None = None + required_metrics: list[str] + config: dict[str, str] + task: str | None = None + filters: list[str] = Field(default_factory=list) + sample_count: Annotated[int, Field(strict=True, ge=0)] = 0 + samples_artifact_id: Positive | None = None + samples_path: str | None = None + dataset: dict[str, str | int] = Field(default_factory=dict) + + +class Contracts(StrictModel): + raw: Literal["aiperf-1.4"] + normalized: Literal["agentx-v1"] + publication: Literal[1] + + +class ExpectedContract(StrictModel): + repository: str + source_run_id: RunId + source_attempt: Positive + source_head_sha: GitSha + bundle_digest: Sha256 + contracts: Contracts + points: list[Point] + + @model_validator(mode="after") + def unique_points(self) -> ExpectedContract: + if not self.points or len({point.point_id for point in self.points}) != len(self.points): + raise ValueError("Expected points must be nonempty and unique") + for point in self.points: + if not point.config or not point.required_metrics: + raise ValueError("Expected config and metric requirements cannot be empty") + if not point.execution_id or len(point.artifact_ids) != len(set(point.artifact_ids)): + raise ValueError("Missing execution identity or duplicate artifact binding") + if point.execution_artifact_id not in point.artifact_ids: + raise ValueError("Execution artifact must belong to its point") + if point.normalized_artifact_id not in point.artifact_ids: + raise ValueError("Normalized artifact must belong to its point") + if point.normalized_format == "lm-eval" and ( + point.kind != "eval" or not point.metadata_path + ): + raise ValueError("Raw lm-eval format requires metadata path and eval mode") + if point.kind == "eval" and ( + not point.task + or not point.filters + or point.sample_count <= 0 + or point.samples_artifact_id not in point.artifact_ids + or not point.samples_path + ): + raise ValueError("Eval contract must require task, filters, count and raw samples") + return self + + +class SourceReceipt(ExpectedContract): + kind: Literal["source-measurement-receipt"] + version: Literal[1] + receipt_id: Sha256 + issuer: Issuer + artifacts: list[Artifact] + + +class PublicationRecord(StrictModel): + kind: Literal["publication-record"] + version: Literal[1] + receipt_id: Sha256 + receipt_artifact_id: Positive + receipt_artifact_sha256: Sha256 + source_run_id: RunId + merge_run_id: RunId + merge_sha: GitSha + changelog_artifact_id: Positive + changelog_artifact_sha256: Sha256 + ingest_sha: GitSha + app_sha: GitSha + + +def canonical_bytes(value: dict[str, Any]) -> bytes: + """Receipt fields contain strings and integers; no floating-point canonicalization.""" + return json.dumps( + value, ensure_ascii=False, sort_keys=True, separators=(",", ":"), allow_nan=False + ).encode() + + +def safe_member(name: str) -> str: + parts = name.removesuffix("/").split("/") + if ( + not name + or "\\" in name + or "\0" in name + or name.startswith("/") + or re.match(r"^[A-Za-z]:", name) + or any(part in {"", ".", ".."} for part in parts) + ): + raise ValueError(f"Unsafe archive member: {name}") + return str(PurePosixPath(*parts)) + + +def inspect_archive(archive: Path) -> list[Member]: + members: list[Member] = [] + names: set[str] = set() + files: set[str] = set() + total = 0 + with zipfile.ZipFile(archive) as source: + for item in source.infolist(): + name = safe_member(item.filename) + kind = stat.S_IFMT(item.external_attr >> 16) + if name in names or kind not in {0, stat.S_IFREG, stat.S_IFDIR}: + raise ValueError(f"Duplicate/link/special archive member: {name}") + names.add(name) + if item.is_dir(): + continue + total += item.file_size + if total > 20 * 1024**3 or item.file_size > 10 * 1024**3: + raise ValueError("Artifact exceeds extraction budget") + payload = source.read(item) + members.append( + Member(path=name, size=len(payload), sha256=hashlib.sha256(payload).hexdigest()) + ) + files.add(name) + for name in files: + if any(str(parent) in files for parent in PurePosixPath(name).parents): + raise ValueError(f"File/directory collision: {name}") + return sorted(members, key=lambda member: member.path) + + +def verify_receipt(data: dict[str, Any]) -> SourceReceipt: + receipt = SourceReceipt.model_validate(data) + payload = receipt.model_dump(exclude={"receipt_id"}) + if hashlib.sha256(canonical_bytes(payload)).hexdigest() != receipt.receipt_id: + raise ValueError("Receipt digest mismatch") + required_ids = {artifact_id for point in receipt.points for artifact_id in point.artifact_ids} + actual_ids = [artifact.id for artifact in receipt.artifacts] + if len(actual_ids) != len(set(actual_ids)) or required_ids != set(actual_ids): + raise ValueError("Receipt artifact set mismatch") + return receipt + + +def validate_point_content(point: Point, archives: Path) -> None: + """Validate normalized/raw eval semantics independently before receipt issuance.""" + with zipfile.ZipFile(archives / f"{point.normalized_artifact_id}.zip") as archive: + value = decode_json(archive.read(safe_member(point.normalized_path)).decode()) + require_finite(value) + if point.normalized_format == "lm-eval": + meta = decode_json(archive.read(safe_member(point.metadata_path or "")).decode()) + require_finite(meta) + if point.task == "gsm8k" and point.sample_count == 1319: + config = value.get("config", {}) + model_args = config.get("model_args", {}) + if ( + config.get("model") != "local-chat-completions" + or config.get("limit") is not None + or model_args.get("num_concurrent") != point.concurrency + or model_args.get("max_length") != 16384 + or model_args.get("tokenized_requests") is not False + or config.get("gen_kwargs") + != {"max_tokens": 12288, "temperature": 0, "top_p": 1} + or set(value.get("results", {})) != {"gsm8k"} + ): + raise ValueError( + "Pilot eval task, concurrency or context/generation contract differs" + ) + if ( + point.config.get("model") == "dsv41flash" + and model_args.get("model") != "deepseek-ai/DeepSeek-V4.1-Flash" + ): + raise ValueError("Pilot eval served model differs from independent expectation") + rows = build_rows(value, meta, source=point.normalized_path) + else: + rows = value if isinstance(value, list) else [value] + matching = [ + row + for row in rows + if isinstance(row, dict) + and row.get("conc", row.get("users")) == point.concurrency + and (point.kind == "throughput" or row.get("task") == point.task) + ] + if len(matching) != 1: + raise ValueError("Expected exactly one normalized point at required concurrency/task") + row = matching[0] + aliases = { + "model": "infmax_model_prefix", + "hardware": "hw", + "specMethod": "spec_decoding", + "recipeFingerprint": "recipe_fingerprint", + } + for key, expected in point.config.items(): + value = row.get(aliases.get(key, key)) + if key == "model": + value = row.get("infmax_model_prefix", row.get("model_prefix")) + if key == "hardware" and isinstance(value, str): + value = value.lower().split("-", 1)[0] + if value != expected: + raise ValueError(f"Normalized configuration mismatch: {key}") + topology = row.get("deployment") + if topology is not None and topology != point.topology.model_dump(): + raise ValueError("Normalized deployment differs from expected topology") + if row.get("disagg") is not False or row.get("is_multinode") is not False: + raise ValueError("Pilot output must explicitly identify aggregate single-node topology") + if ( + row.get("tp") != point.topology.tp + or row.get("ep") != point.topology.ep + or row.get("num_gpus") != point.topology.serving_gpus + ): + raise ValueError("Normalized GPU/parallelism mismatch") + metric_paths = { + "output_tput_tps": ("request_metrics", "throughput", "output", "tokens_per_second"), + "total_tput_tps": ("request_metrics", "throughput", "total", "tokens_per_second"), + "duration_seconds": ("request_metrics", "throughput", "duration_seconds"), + } + for key in point.required_metrics: + value = row.get(key) + if value is None and key in metric_paths: + value = row + for field in metric_paths[key]: + value = value.get(field) if isinstance(value, dict) else None + if ( + isinstance(value, bool) + or not isinstance(value, int | float) + or not math.isfinite(value) + or value < 0 + ): + raise ValueError(f"Required finite non-negative metric missing: {key}") + if point.dataset and row.get("dataset") != point.dataset: + raise ValueError("Dataset identity differs from independent expectation") + if point.kind == "eval": + with zipfile.ZipFile(archives / f"{point.samples_artifact_id}.zip") as archive: + text = archive.read(safe_member(point.samples_path or "")).decode() + identities = None + if point.task == "gsm8k" and point.sample_count == 1319: + identities = decode_json( + files("infx.benchmarks") + .joinpath("resources/gsm8k-test-doc-hashes.json") + .read_text() + ) + observed: set[tuple[int, str]] = set() + strict_passed = 0 + for line in text.splitlines(): + if not line.strip(): + continue + sample = decode_json(line) + require_finite(sample) + doc_id, filter_name = sample.get("doc_id"), sample.get("filter") + if ( + type(doc_id) is not int + or doc_id < 0 + or filter_name not in point.filters + or sample.get("task_name", point.task) != point.task + or (doc_id, filter_name) in observed + ): + raise ValueError("Invalid/duplicate evaluation sample identity") + if identities is not None: + document_hash = hashlib.sha256( + json.dumps(sample.get("doc"), indent=2, ensure_ascii=False).encode() + ).hexdigest() + target = sample.get("target") + if ( + identities.get(str(doc_id)) != document_hash + or sample.get("doc_hash") != document_hash + or target != sample.get("doc", {}).get("answer") + or sample.get("target_hash") != hashlib.sha256(str(target).encode()).hexdigest() + ): + raise ValueError("Pilot eval document/target differs from prepared full split") + observed.add((doc_id, filter_name)) + if filter_name == "strict-match": + score = sample.get("exact_match,strict-match", sample.get("exact_match")) + if type(score) not in (int, float) or score not in (0, 1): + raise ValueError("GSM8K strict sample requires a binary score") + strict_passed += int(score) + documents = {doc for doc, _ in observed} + if ( + documents != set(range(point.sample_count)) + or len(observed) != point.sample_count * len(point.filters) + or row.get("n_eff") != point.sample_count + ): + raise ValueError("Incomplete evaluation sample/filter coverage") + if not math.isclose( + row["em_strict"], strict_passed / point.sample_count, rel_tol=0, abs_tol=1e-12 + ): + raise ValueError("Evaluation strict summary differs from raw samples") + + +def seal_receipt( + expected: ExpectedContract, issuer: Issuer, inventory: list[dict[str, Any]], archives: Path +) -> SourceReceipt: + required = {artifact_id for point in expected.points for artifact_id in point.artifact_ids} + rows = {int(row["id"]): row for row in inventory} + if len(rows) != len(inventory) or set(rows) != required: + raise ValueError("Requested artifact set differs from independently fetched API inventory") + artifacts: list[Artifact] = [] + for artifact_id in sorted(required): + row = rows[artifact_id] + run_ids = { + point.source_run_id for point in expected.points if artifact_id in point.artifact_ids + } + if ( + len(run_ids) != 1 + or str(row.get("workflow_run", {}).get("id")) not in run_ids + or row.get("expired") + ): + raise ValueError(f"Wrong-run/expired artifact {artifact_id}") + archive = archives / f"{artifact_id}.zip" + with archive.open("rb") as source: + digest = hashlib.file_digest(source, "sha256").hexdigest() + if row.get("digest") != f"sha256:{digest}": + raise ValueError(f"API/archive digest mismatch: {artifact_id}") + artifacts.append( + Artifact( + id=artifact_id, + name=row["name"], + sha256=digest, + run_id=next(iter(run_ids)), + members=inspect_archive(archive), + ) + ) + for point in expected.points: + validate_point_content(point, archives) + with zipfile.ZipFile(archives / f"{point.execution_artifact_id}.zip") as archive: + execution = decode_json(archive.read(safe_member(point.execution_path)).decode()) + if not isinstance(execution, dict): + raise ValueError("Execution evidence must be an object") + native = execution.get("native_receipt", {}) + source = execution.get("source", {}) + if ( + not isinstance(native, dict) + or not isinstance(source, dict) + or type(execution.get("schema_version")) is not int + or execution.get("schema_version") != 1 + or execution.get("point_id") != point.point_id + or execution.get("execution_id") != point.execution_id + or execution.get("bundle_digest") != point.bundle_digest + or execution.get("mode") != point.kind + or type(execution.get("client_exit_code")) is not int + or execution.get("client_exit_code") != 0 + or source.get("repository") != expected.repository + or str(source.get("run_id")) != point.source_run_id + or type(source.get("attempt")) is not int + or source.get("attempt") != point.source_attempt + or source.get("head_sha") != expected.source_head_sha + or native.get("state") != "COMPLETED" + or not re.fullmatch(r"[1-9][0-9]*", str(native.get("job_id", ""))) + or native.get("manifest_sha256") != point.native_manifest_sha256 + ): + raise ValueError( + f"Execution evidence differs from independently expected point: {point.point_id}" + ) + payload = expected.model_dump() | { + "kind": "source-measurement-receipt", + "version": 1, + "issuer": issuer.model_dump(), + "artifacts": [artifact.model_dump() for artifact in artifacts], + } + return verify_receipt( + payload | {"receipt_id": hashlib.sha256(canonical_bytes(payload)).hexdigest()} + ) + + +def fetch_and_seal(expected: ExpectedContract, issuer: Issuer, archives: Path) -> SourceReceipt: + if not re.fullmatch(r"[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+", expected.repository): + raise ValueError("Invalid repository") + archives.mkdir(parents=True, exist_ok=True) + inventory = [] + for artifact_id in sorted({value for point in expected.points for value in point.artifact_ids}): + endpoint = f"repos/{expected.repository}/actions/artifacts/{artifact_id}" + row = json.loads(subprocess.check_output(["gh", "api", endpoint])) + inventory.append(row) + with (archives / f"{artifact_id}.zip").open("xb") as output: + subprocess.run(["gh", "api", f"{endpoint}/zip"], stdout=output, check=True) + return seal_receipt(expected, issuer, inventory, archives) + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--expected", type=Path, required=True) + parser.add_argument("--issuer", type=Path, required=True) + parser.add_argument("--archives", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + expected = ExpectedContract.model_validate_json(args.expected.read_text()) + issuer = Issuer.model_validate_json(args.issuer.read_text()) + receipt = fetch_and_seal(expected, issuer, args.archives) + with args.output.open("x") as output: + output.write(json.dumps(receipt.model_dump(), indent=2, ensure_ascii=False) + "\n") + + +if __name__ == "__main__": + main() diff --git a/infx/srt_slurm/contracts.py b/infx/srt_slurm/contracts.py new file mode 100644 index 0000000000..35002feb61 --- /dev/null +++ b/infx/srt_slurm/contracts.py @@ -0,0 +1,89 @@ +"""Versioned execution references and content identities for native SRT jobs.""" + +from __future__ import annotations + +import hashlib +import json +from pathlib import Path, PurePosixPath +from typing import Any, Literal + +import yaml +from pydantic import BaseModel, ConfigDict, Field, field_validator + + +class ExecutionReference(BaseModel): + model_config = ConfigDict(extra="forbid", strict=True, populate_by_name=True) + + runtime: Literal["srt-slurm"] + contract_version: Literal[1] = Field(alias="contract-version") + recipe: str + profile: str + runtime_lock: str = Field(alias="runtime-lock") + client_policy: str = Field(alias="client-policy") + input_digests: dict[str, str] | None = Field(default=None, alias="input-digests") + + @field_validator("recipe", "profile", "runtime_lock", "client_policy") + @classmethod + def repository_path(cls, value: str) -> str: + path = PurePosixPath(value) + if not value or path.is_absolute() or ".." in path.parts or "\\" in value: + raise ValueError("execution references must be contained repository paths") + return value + + +class UniqueKeyLoader(yaml.SafeLoader): + """Reject accidental duplicate mappings before native schema validation.""" + + +def _unique_mapping(loader: UniqueKeyLoader, node: yaml.MappingNode) -> dict: + result = {} + for key_node, value_node in node.value: + key = loader.construct_object(key_node, deep=True) + if key in result: + raise ValueError(f"Duplicate YAML key {key!r} at {key_node.start_mark}") + result[key] = loader.construct_object(value_node, deep=True) + return result + + +UniqueKeyLoader.add_constructor(yaml.resolver.BaseResolver.DEFAULT_MAPPING_TAG, _unique_mapping) + + +def load_mapping(path: Path) -> dict[str, Any]: + value = yaml.load(path.read_text(), Loader=UniqueKeyLoader) # noqa: S506 + if not isinstance(value, dict): + raise ValueError(f"Expected a mapping: {path}") + return value + + +def digest(value: Any) -> str: + encoded = json.dumps(value, sort_keys=True, separators=(",", ":"), allow_nan=False) + return hashlib.sha256(encoded.encode()).hexdigest() + + +def referenced_inputs(reference: ExecutionReference, root: Path) -> dict[str, str]: + """Bind referenced bytes from the caller's selected checkout, never a live fallback.""" + inputs = {} + resolved_root = root.resolve() + names = [reference.recipe, reference.profile, reference.runtime_lock, reference.client_policy] + policy_path = (root / reference.client_policy).resolve(strict=True) + if not policy_path.is_relative_to(resolved_root): + raise ValueError(f"Execution input escapes checkout: {reference.client_policy}") + policy = load_mapping(policy_path) + golden = policy.get("golden_curve") + if not isinstance(golden, str): + raise ValueError("client policy must name its committed golden curve") + names.append(ExecutionReference.repository_path(golden)) + for name in names: + path = (root / name).resolve(strict=True) + if not path.is_relative_to(resolved_root): + raise ValueError(f"Execution input escapes checkout: {name}") + inputs[name] = hashlib.sha256(path.read_bytes()).hexdigest() + return inputs + + +def resolve_reference(raw: dict[str, Any], root: Path) -> dict[str, Any]: + reference = ExecutionReference.model_validate(raw) + actual = referenced_inputs(reference, root) + if reference.input_digests is not None and reference.input_digests != actual: + raise ValueError("Execution inputs changed after matrix generation") + return reference.model_copy(update={"input_digests": actual}).model_dump(by_alias=True) diff --git a/infx/workflows/phase1_publication.py b/infx/workflows/phase1_publication.py new file mode 100644 index 0000000000..7984bf57b6 --- /dev/null +++ b/infx/workflows/phase1_publication.py @@ -0,0 +1,222 @@ +"""Seal the H100 qualification against a separately reviewed prepared expectation. + +Approval JSON belongs to trusted default-branch control code. Never populate its +identities from worker execution.json, infer approval from artifact names, or run +candidate code in this collector. +""" + +from __future__ import annotations + +import argparse +import json +import os +import subprocess +import tempfile +from pathlib import Path +from typing import Annotated, Literal + +from pydantic import BaseModel, ConfigDict, Field, model_validator + +from infx.benchmarks.common import decode_json, read_json +from infx.results.publication_receipt import ( + Contracts, + ExpectedContract, + Issuer, + Point, + Topology, + fetch_and_seal, + inspect_archive, +) +from infx.srt_slurm.contracts import digest + +Sha = Annotated[str, Field(pattern=r"^[0-9a-f]{64}$")] +GitSha = Annotated[str, Field(pattern=r"^[0-9a-f]{40}$")] + + +class ApprovedPoint(BaseModel): + model_config = ConfigDict(extra="forbid", strict=True) + + point_id: Sha + execution_id: Sha + bundle_digest: Sha + native_manifest_sha256: Sha + kind: Literal["throughput", "eval"] + concurrency: int = Field(gt=0) + dataset_revision: GitSha | None + + +class Approval(BaseModel): + model_config = ConfigDict(extra="forbid", strict=True) + + schema_version: Literal[1] + repository: Literal["SemiAnalysisAI/InferenceX"] + source_run_id: str = Field(pattern=r"^[1-9][0-9]*$") + source_attempt: int = Field(gt=0) + source_head_sha: GitSha + points: list[ApprovedPoint] + + @model_validator(mode="after") + def complete_pilot(self) -> Approval: + expected = {("throughput", c) for c in (1, 2, 4, 8, 16, 20, 24, 28)} | {("eval", 28)} + if len(self.points) != 9 or {(p.kind, p.concurrency) for p in self.points} != expected: + raise ValueError( + "Phase 1 requires exactly eight throughput points and the real c28 eval" + ) + if len({p.point_id for p in self.points}) != len(self.points): + raise ValueError("approved points must have distinct semantic identities") + if any(p.kind == "throughput" and p.dataset_revision is None for p in self.points): + raise ValueError("throughput approval requires its actual immutable corpus revision") + return self + + +def api(path: str, *, paginate: bool = False) -> object: + argv = ["gh", "api", path] + if paginate: + argv.extend(("--paginate", "--slurp")) + return decode_json(subprocess.run(argv, capture_output=True, text=True, check=True).stdout) + + +def expected_contract(approval: Approval) -> ExpectedContract: + run = api(f"repos/{approval.repository}/actions/runs/{approval.source_run_id}") + if ( + not isinstance(run, dict) + or str(run.get("id")) != approval.source_run_id + or run.get("head_sha") != approval.source_head_sha + or run.get("run_attempt") != approval.source_attempt + or run.get("status") != "completed" + or run.get("conclusion") != "success" + or run.get("path", "").split("@", 1)[0] != ".github/workflows/run-sweep.yml" + ): + raise ValueError("approved source is not the completed successful sweep/attempt/head") + pages = api( + f"repos/{approval.repository}/actions/runs/{approval.source_run_id}/artifacts?per_page=100", + paginate=True, + ) + inventory = [item for page in pages for item in page["artifacts"] if not item["expired"]] + + def artifact(name: str) -> int: + matching = [item for item in inventory if item["name"] == name] + if len(matching) != 1: + raise ValueError(f"expected one unexpired artifact named {name}") + return int(matching[0]["id"]) + + points = [] + for approved in approval.points: + execution = artifact(f"native-execution-{approved.point_id}") + normalized_path = f"{approved.point_id}.json" + samples_path = None + metadata_path = None + if approved.kind == "throughput": + normalized = artifact(f"bmk_agentic_{approved.point_id}") + ids = [execution, normalized, artifact(f"agentic_{approved.point_id}")] + metrics = ["output_tput_tps", "total_tput_tps", "duration_seconds"] + else: + normalized = artifact(f"eval_{approved.point_id}_lm-eval__{approval.source_attempt}") + ids = [execution, normalized] + with tempfile.TemporaryDirectory(prefix="infx-eval-members-") as temporary: + archive = Path(temporary) / "eval.zip" + with archive.open("xb") as stream: + subprocess.run( + [ + "gh", + "api", + f"repos/{approval.repository}/actions/artifacts/{normalized}/zip", + ], + stdout=stream, + check=True, + ) + members = [member.path for member in inspect_archive(archive)] + results = [ + name + for name in members + if name.startswith("results") + and name.endswith("_conc28.json") + and "/" not in name + ] + samples = [ + name + for name in members + if name.startswith("samples") + and name.endswith("_conc28.jsonl") + and "/" not in name + ] + if len(results) != 1 or len(samples) != 1 or "meta_env.json" not in members: + raise ValueError( + "eval artifact lacks one complete c28 result/sample/metadata set" + ) + normalized_path, samples_path, metadata_path = ( + results[0], + samples[0], + "meta_env.json", + ) + metrics = ["em_strict", "em_strict_se", "em_flexible", "em_flexible_se", "n_eff"] + points.append( + Point( + **approved.model_dump(exclude={"dataset_revision"}), + source_run_id=approval.source_run_id, + source_attempt=approval.source_attempt, + topology=Topology(kind="aggregate", nodes=1, serving_gpus=8, tp=8, ep=1), + artifact_ids=ids, + execution_artifact_id=execution, + execution_path="execution.json", + normalized_artifact_id=normalized, + normalized_path=normalized_path, + normalized_format="lm-eval" if approved.kind == "eval" else "normalized", + metadata_path=metadata_path, + required_metrics=metrics, + config={ + "model": "dsv41flash", + "hardware": "h100", + "framework": "vllm", + "precision": "fp4", + "specMethod": "mtp", + }, + task="gsm8k" if approved.kind == "eval" else None, + filters=["strict-match", "flexible-extract"] if approved.kind == "eval" else [], + sample_count=1319 if approved.kind == "eval" else 0, + samples_artifact_id=normalized if approved.kind == "eval" else None, + samples_path=samples_path, + dataset={ + "source_type": "public_dataset", + "loader": "semianalysis_cc_traces_weka_062126", + "hf_dataset_name": "semianalysisai/cc-traces-weka-062126", + "hf_split": "train", + "num_dataset_entries": 393, + "hf_revision": approved.dataset_revision, + } + if approved.kind == "throughput" + else {}, + ) + ) + return ExpectedContract( + repository=approval.repository, + source_run_id=approval.source_run_id, + source_attempt=approval.source_attempt, + source_head_sha=approval.source_head_sha, + bundle_digest=digest({point.point_id: point.bundle_digest for point in approval.points}), + contracts=Contracts(raw="aiperf-1.4", normalized="agentx-v1", publication=1), + points=points, + ) + + +def main() -> None: + parser = argparse.ArgumentParser() + parser.add_argument("--approval", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + parser.add_argument("--archives", type=Path, required=True) + args = parser.parse_args() + approval = Approval.model_validate(read_json(args.approval)) + issuer = Issuer( + repository=os.environ["GITHUB_REPOSITORY"], + run_id=os.environ["GITHUB_RUN_ID"], + job=os.environ["GITHUB_JOB"], + workflow_sha=os.environ["TRUSTED_WORKFLOW_SHA"], + collector_sha=os.environ["TRUSTED_WORKFLOW_SHA"], + ) + receipt = fetch_and_seal(expected_contract(approval), issuer, args.archives) + with args.output.open("x") as stream: + stream.write(json.dumps(receipt.model_dump(), indent=2) + "\n") + + +if __name__ == "__main__": + main() diff --git a/infx/workflows/phase1_record.py b/infx/workflows/phase1_record.py new file mode 100644 index 0000000000..dfc2e238ee --- /dev/null +++ b/infx/workflows/phase1_record.py @@ -0,0 +1,143 @@ +"""Bind an immutable source receipt to a later publication without changing the source.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import os +import re +import subprocess +import tempfile +import zipfile +from pathlib import Path +from typing import Any + +from infx.benchmarks.common import decode_json, read_json +from infx.results.publication_receipt import PublicationRecord, inspect_archive, verify_receipt +from infx.workflows.phase1_publication import api + + +def verified_artifact( + repository: str, artifact_id: int, expected_digest: str, *, name: str +) -> tuple[dict[str, Any], bytes]: + endpoint = f"repos/{repository}/actions/artifacts/{artifact_id}" + metadata = api(endpoint) + if ( + not isinstance(metadata, dict) + or metadata.get("id") != artifact_id + or metadata.get("expired") + or metadata.get("name") != name + ): + raise ValueError("required immutable artifact is missing, expired or misnamed") + if metadata.get("digest") != "sha256:" + expected_digest: + raise ValueError("artifact API digest disagrees with approved publication") + payload = subprocess.run( + ["gh", "api", endpoint + "/zip"], capture_output=True, check=True + ).stdout + if hashlib.sha256(payload).hexdigest() != expected_digest: + raise ValueError("downloaded artifact differs from approved digest") + return metadata, payload + + +def validate_record( + record: PublicationRecord, repository: str, issuer_shas: set[str], reader_sha: str +) -> PublicationRecord: + if ( + not re.fullmatch(r"[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+", repository) + or not issuer_shas + or any(not re.fullmatch(r"[a-f0-9]{40}", value) for value in issuer_shas) + or not re.fullmatch(r"[a-f0-9]{40}", reader_sha) + ): + raise ValueError("publication requires valid deployed repository/issuer/reader policy") + metadata, payload = verified_artifact( + repository, + record.receipt_artifact_id, + record.receipt_artifact_sha256, + name="measurement-receipt", + ) + issuer_id = metadata.get("workflow_run", {}).get("id") + if type(issuer_id) is not int or issuer_id <= 0: + raise ValueError("source receipt has no valid API issuer identity") + issuer = api(f"repos/{repository}/actions/runs/{issuer_id}") + if ( + not isinstance(issuer, dict) + or issuer.get("id") != issuer_id + or issuer.get("head_sha") not in issuer_shas + or issuer.get("status") != "completed" + or issuer.get("conclusion") != "success" + or issuer.get("head_branch") != "main" + or issuer.get("event") != "workflow_dispatch" + or issuer.get("path") != ".github/workflows/phase1-receipt.yml" + ): + raise ValueError("source receipt issuer is not an approved completed trusted workflow") + with tempfile.TemporaryDirectory(prefix="infx-source-receipt-") as temporary: + archive = Path(temporary) / "receipt.zip" + archive.write_bytes(payload) + if [member.path for member in inspect_archive(archive)] != ["receipt.json"]: + raise ValueError("source receipt artifact must contain only receipt.json") + with zipfile.ZipFile(archive) as stream: + receipt = verify_receipt(decode_json(stream.read("receipt.json").decode())) + if ( + receipt.repository != repository + or receipt.issuer.repository != repository + or receipt.receipt_id != record.receipt_id + or receipt.source_run_id != record.source_run_id + or receipt.issuer.run_id != str(issuer["id"]) + or receipt.issuer.workflow_sha != issuer["head_sha"] + ): + raise ValueError("publication attempts to replace source measurement identity") + source = api( + f"repos/{repository}/actions/runs/{record.source_run_id}/attempts/{receipt.source_attempt}" + ) + if ( + not isinstance(source, dict) + or str(source.get("id")) != record.source_run_id + or source.get("run_attempt") != receipt.source_attempt + or source.get("head_sha") != receipt.source_head_sha + or source.get("status") != "completed" + or source.get("conclusion") != "success" + ): + raise ValueError("source receipt does not match its completed original attempt") + merge = api(f"repos/{repository}/actions/runs/{record.merge_run_id}") + if ( + not isinstance(merge, dict) + or str(merge.get("id")) != record.merge_run_id + or merge.get("head_sha") != record.merge_sha + or merge.get("head_branch") != "main" + or merge.get("event") != "push" + or merge.get("status") != "completed" + or merge.get("conclusion") != "success" + or merge.get("path") != ".github/workflows/run-sweep.yml" + ): + raise ValueError("publication merge must be the completed successful main sweep") + changelog, _ = verified_artifact( + repository, + record.changelog_artifact_id, + record.changelog_artifact_sha256, + name="changelog-metadata", + ) + if str(changelog["workflow_run"]["id"]) != record.merge_run_id: + raise ValueError("publication changelog belongs to a different merge run") + if record.app_sha != reader_sha or record.ingest_sha != reader_sha: + raise ValueError("publication reader revision does not match the deployed reader pin") + return record + + +def main() -> None: + parser = argparse.ArgumentParser() + parser.add_argument("--approval", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + record = validate_record( + PublicationRecord.model_validate(read_json(args.approval)), + os.environ["GITHUB_REPOSITORY"], + {value.strip() for value in os.environ["TRUSTED_ISSUER_SHAS"].split(",") if value.strip()}, + os.environ["DEPLOYED_READER_SHA"], + ) + with args.output.open("x") as stream: + stream.write(json.dumps(record.model_dump(), indent=2) + "\n") + + +if __name__ == "__main__": + main() diff --git a/infx/workflows/receipt_transport.py b/infx/workflows/receipt_transport.py new file mode 100644 index 0000000000..a79caac90c --- /dev/null +++ b/infx/workflows/receipt_transport.py @@ -0,0 +1,278 @@ +"""Resolve immutable receipts using read-only APIs and deployed issuer policy.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import os +import re +import subprocess +import tempfile +import zipfile +from dataclasses import dataclass +from pathlib import Path +from typing import Any + +from infx import github +from infx.benchmarks.common import decode_json +from infx.results.publication_receipt import ( + PublicationRecord, + SourceReceipt, + inspect_archive, + verify_receipt, +) + + +class PendingReceiptError(ValueError): + """The native source has not yet passed its separate trusted publication stage.""" + + +@dataclass(frozen=True) +class SealedArtifact: + artifact_id: int + archive_sha256: str + content_sha256: str + issuer_run_id: str + issuer_sha: str + document: dict[str, Any] + + +def read_sealed( + repo: str, artifact: dict[str, Any], issuer: dict[str, Any], member: str +) -> SealedArtifact: + artifact_id = artifact.get("id") + if type(artifact_id) is not int or artifact_id <= 0 or artifact.get("expired"): + raise ValueError("Invalid/expired receipt artifact") + metadata = github.api(repo, f"/actions/artifacts/{artifact_id}") + if ( + metadata.get("id") != artifact_id + or metadata.get("workflow_run", {}).get("id") != issuer["id"] + or metadata.get("expired") + ): + raise ValueError("Receipt API artifact ownership mismatch") + with tempfile.TemporaryDirectory(prefix="infx-sealed-artifact-") as temporary: + archive = Path(temporary) / f"{artifact_id}.zip" + with archive.open("xb") as stream: + subprocess.run( + ["gh", "api", f"repos/{repo}/actions/artifacts/{artifact_id}/zip"], + stdout=stream, + check=True, + ) + if archive.stat().st_size > 10 * 1024**2: + raise ValueError("Compact receipt archive exceeds size budget") + with archive.open("rb") as stream: + archive_sha = hashlib.file_digest(stream, "sha256").hexdigest() + if metadata.get("digest") != f"sha256:{archive_sha}": + raise ValueError("Receipt archive differs from API digest") + members = inspect_archive(archive) + if len(members) != 1 or members[0].path != member or members[0].size > 10 * 1024**2: + raise ValueError("Receipt archive must contain exactly its versioned JSON document") + with zipfile.ZipFile(archive) as source: + content = source.read(member) + document = decode_json(content.decode()) + if not isinstance(document, dict): + raise ValueError("Sealed receipt must be an object") + return SealedArtifact( + artifact_id, + archive_sha, + hashlib.sha256(content).hexdigest(), + str(issuer["id"]), + issuer["head_sha"], + document, + ) + + +def resolve_transport( + repo: str, + source_run_id: str, + merge_run_id: str, + allowed_shas: set[str], + *, + workflow: str = ".github/workflows/phase1-receipt.yml", + publication_required: bool = False, +) -> dict[str, str]: + if not re.fullmatch(r"[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+", repo) or any( + not re.fullmatch(r"[1-9][0-9]*", value) for value in (source_run_id, merge_run_id) + ): + raise ValueError("Invalid repository/run identity") + inventory = github.paginate( + repo, f"/actions/runs/{source_run_id}/artifacts", item_key="artifacts" + ) + if not any(str(item.get("name", "")).startswith("native-execution-") for item in inventory): + return {"receipt-required": "false"} + if not allowed_shas or any(not re.fullmatch(r"[a-f0-9]{40}", value) for value in allowed_shas): + raise ValueError("Native receipt requires deployed issuer revision allowlist") + if not re.fullmatch(r"\.github/workflows/[A-Za-z0-9_.-]+\.ya?ml", workflow): + raise ValueError("Invalid deployed issuer workflow") + source_receipts: list[tuple[SealedArtifact, SourceReceipt]] = [] + publications: list[SealedArtifact] = [] + runs = github.paginate( + repo, + f"/actions/workflows/{workflow.rsplit('/', 1)[-1]}/runs", + item_key="workflow_runs", + params={"status": "completed"}, + ) + for run in runs: + if ( + run.get("head_sha") not in allowed_shas + or run.get("status") != "completed" + or run.get("conclusion") != "success" + or run.get("path") != workflow + or run.get("head_branch") != "main" + or run.get("event") != "workflow_dispatch" + ): + continue + artifacts = github.paginate( + repo, f"/actions/runs/{run['id']}/artifacts", item_key="artifacts" + ) + for artifact in artifacts: + if artifact.get("expired"): + continue + if artifact.get("name") == "measurement-receipt": + sealed = read_sealed(repo, artifact, run, "receipt.json") + if ( + sealed.document.get("repository") != repo + or sealed.document.get("source_run_id") != source_run_id + ): + continue + receipt = verify_receipt(sealed.document) + if ( + receipt.issuer.repository != repo + or receipt.issuer.run_id != sealed.issuer_run_id + or receipt.issuer.workflow_sha != sealed.issuer_sha + ): + raise ValueError("Receipt claims an issuer different from its actual API owner") + attempt = github.api( + repo, f"/actions/runs/{source_run_id}/attempts/{receipt.source_attempt}" + ) + if ( + str(attempt.get("id")) != source_run_id + or attempt.get("run_attempt") != receipt.source_attempt + or attempt.get("head_sha") != receipt.source_head_sha + or attempt.get("status") != "completed" + or attempt.get("conclusion") != "success" + ): + raise ValueError( + "Receipt source attempt/head is not a completed successful execution" + ) + source_receipts.append((sealed, receipt)) + elif artifact.get("name") == "publication-record": + publications.append(read_sealed(repo, artifact, run, "publication.json")) + if not source_receipts: + raise PendingReceiptError( + "Native source awaits an independently sealed measurement receipt; no ingest dispatched" + ) + if len(source_receipts) != 1: + raise ValueError( + "Multiple source receipts require explicit accepted-snapshot resolution; refusing newest-by-name selection" + ) + sealed, receipt = source_receipts[0] + for artifact in receipt.artifacts: + metadata = github.api(repo, f"/actions/artifacts/{artifact.id}") + if ( + metadata.get("id") != artifact.id + or metadata.get("expired") + or str(metadata.get("workflow_run", {}).get("id")) != artifact.run_id + or metadata.get("digest") != f"sha256:{artifact.sha256}" + ): + raise ValueError("Accepted artifact set is missing, expired, wrong-run or changed") + payload = { + "receipt-required": "true", + "receipt-artifact-id": str(sealed.artifact_id), + "receipt-artifact-sha256": sealed.archive_sha256, + "receipt-sha256": sealed.content_sha256, + "receipt-issuer-run-id": sealed.issuer_run_id, + "receipt-issuer-sha": sealed.issuer_sha, + } + if publication_required or source_run_id != merge_run_id: + matches: list[tuple[SealedArtifact, PublicationRecord]] = [] + for publication in publications: + if ( + publication.document.get("source_run_id") != source_run_id + or publication.document.get("merge_run_id") != merge_run_id + ): + continue + record = PublicationRecord.model_validate(publication.document) + if ( + record.source_run_id == source_run_id + and record.merge_run_id == merge_run_id + and record.receipt_id == receipt.receipt_id + ): + matches.append((publication, record)) + if not matches: + raise PendingReceiptError( + "Accepted source awaits its later publication record; no ingest dispatched" + ) + if len(matches) != 1: + raise ValueError( + "Multiple publication records require explicit accepted-snapshot resolution" + ) + publication, record = matches[0] + merge = github.api(repo, f"/actions/runs/{merge_run_id}") + changelog = github.api(repo, f"/actions/artifacts/{record.changelog_artifact_id}") + if ( + record.receipt_artifact_id != sealed.artifact_id + or record.receipt_artifact_sha256 != sealed.archive_sha256 + or merge.get("head_sha") != record.merge_sha + or merge.get("head_branch") != "main" + or merge.get("event") != "push" + or merge.get("status") != "completed" + or merge.get("conclusion") != "success" + or changelog.get("expired") + or str(changelog.get("workflow_run", {}).get("id")) != merge_run_id + or changelog.get("name") != "changelog-metadata" + or changelog.get("digest") != f"sha256:{record.changelog_artifact_sha256}" + ): + raise ValueError("Publication record differs from accepted source/merge/changelog") + payload.update( + { + "publication-artifact-id": str(publication.artifact_id), + "publication-artifact-sha256": publication.archive_sha256, + "publication-sha256": publication.content_sha256, + "publication-issuer-run-id": publication.issuer_run_id, + "publication-issuer-sha": publication.issuer_sha, + } + ) + return payload + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--repository", default=os.environ.get("GITHUB_REPOSITORY"), required=False) + parser.add_argument("--source-run-id", default=os.environ.get("SOURCE_RUN_ID"), required=False) + parser.add_argument("--merge-run-id", default=os.environ.get("MERGE_RUN_ID"), required=False) + parser.add_argument("--publication-required", action="store_true") + parser.add_argument("--defer-unsealed", action="store_true") + args = parser.parse_args() + if not all((args.repository, args.source_run_id, args.merge_run_id)): + parser.error("repository, source run ID and merge run ID are required") + allowed = { + value.strip() + for value in os.environ.get("INFX_RECEIPT_ISSUER_SHAS", "").split(",") + if value.strip() + } + try: + payload = resolve_transport( + args.repository, + args.source_run_id, + args.merge_run_id, + allowed, + workflow=os.environ.get("INFX_RECEIPT_ISSUER_WORKFLOW", ""), + publication_required=args.publication_required, + ) + outputs = {"ready": "true", "payload": json.dumps(payload, separators=(",", ":"))} + except PendingReceiptError as exc: + if not args.defer_unsealed: + raise + outputs = {"ready": "false", "payload": "{}"} + print(str(exc)) + output = os.environ.get("GITHUB_OUTPUT") + if output: + with Path(output).open("a") as stream: + stream.write("".join(f"{key}={value}\n" for key, value in outputs.items())) + print(json.dumps(outputs)) + + +if __name__ == "__main__": + main() diff --git a/utils/test_benchmark_preparation.py b/utils/test_benchmark_preparation.py new file mode 100644 index 0000000000..f76897026a --- /dev/null +++ b/utils/test_benchmark_preparation.py @@ -0,0 +1,306 @@ +"""Preparation and derived-cache behavior under file mutation and process contention.""" + +from __future__ import annotations + +import fcntl +import json +import shutil +import subprocess +import sys +from pathlib import Path + +import pytest + +from infx.benchmarks.cache import MmapCache +from infx.benchmarks.common import read_json +from infx.benchmarks.identity import ( + capture_identity, + require_source_revision, + verify_runtime, +) +from infx.benchmarks.prepare import ClientSite, collect_assets, prepare +from infx.benchmarks.spec import RuntimeSpec + + +@pytest.fixture +def installed_child(tmp_path): + environment = tmp_path / "child-env" + subprocess.run( + [sys.executable, "-m", "venv", "--without-pip", str(environment)], check=True + ) + python = environment / "bin/python" + result = subprocess.run( + [ + str(python), + "-I", + "-c", + "import sysconfig; print(sysconfig.get_path('purelib'))", + ], + capture_output=True, + text=True, + check=True, + ) + site = Path(result.stdout.strip()) + package = site / "lm_eval" + package.mkdir() + implementation = package / "__init__.py" + implementation.write_text("meaning = 42\n") + metadata = site / "lm_eval-0.1.dist-info" + metadata.mkdir() + (metadata / "METADATA").write_text( + "Metadata-Version: 2.1\nName: lm-eval\nVersion: 0.1\n" + ) + (metadata / "direct_url.json").write_text( + json.dumps( + { + "url": "https://github.com/EleutherAI/lm-evaluation-harness.git", + "vcs_info": { + "vcs": "git", + "commit_id": "b315ef3b05176acc9732bb7fdec116abe1ecc476", + }, + } + ) + ) + (metadata / "RECORD").write_text( + "\n".join( + f"{path.relative_to(site)},," + for path in [implementation, *metadata.iterdir(), metadata / "RECORD"] + ) + + "\n" + ) + return python, implementation + + +@pytest.fixture +def client_site(tmp_path, installed_child): + model = tmp_path / "model" + model.mkdir() + (model / "config.json").write_text('{"model_type":"fixture"}') + (model / "tokenizer.json").write_text('{"type":"fixture"}') + (model / "model.safetensors.index.json").write_text( + '{"weight_map":{"weight":"model-00001.safetensors"}}' + ) + (model / "model-00001.safetensors").write_bytes(b"controlled model bytes") + dataset = tmp_path / "hub/datasets--openai--gsm8k" + reference = dataset / "refs/main" + reference.parent.mkdir(parents=True) + reference.write_text("b" * 40) + payload = dataset / "snapshots" / ("b" * 40) / "train.parquet" + payload.parent.mkdir(parents=True) + payload.write_bytes(b"controlled cached dataset") + return ClientSite( + python=str(installed_child[0]), + distributions=["lm-eval"], + env={ + "HF_HUB_OFFLINE": "1", + "HF_DATASETS_OFFLINE": "1", + "HF_HUB_CACHE": str(tmp_path / "hub"), + "HF_DATASETS_CACHE": str(tmp_path / "datasets"), + }, + env_unset=[], + asset_roots=[str(model), str(dataset)], + asset_files=[], + model_path=str(model), + timeout_seconds=10, + terminate_grace_seconds=1, + ) + + +def test_preparation_binds_installed_bytes_and_packaged_resources_from_other_cwd( + tmp_path, installed_child, client_site, monkeypatch +): + outside = tmp_path / "unrelated" + outside.mkdir() + monkeypatch.chdir(outside) + prepared = tmp_path / "prepared" + resources = prepare(client_site, "eval", prepared) + runtime = RuntimeSpec.model_validate(read_json(prepared / "runtime.json")) + original = verify_runtime(runtime, dataset_loader=None) + assert original["distributions"]["lm-eval"]["version"] == "0.1" + # Both files are real installed resources copied outside the checkout. A changed + # task is detected through its bound content, before the harness can execute it. + task = Path(resources["task"]["path"]) + assert task.parent == prepared + task.chmod(0o644) + task.write_text("task: unrelated\n") + from infx.benchmarks.common import verify_file + from infx.benchmarks.spec import PreparedFile + + with pytest.raises(ValueError, match="prepared file missing or changed"): + verify_file(PreparedFile.model_validate(resources["task"])) + # Same package name and version, different actual implementation bytes. + installed_child[1].write_text("meaning = -1\n") + with pytest.raises(ValueError, match="identity differs"): + verify_runtime(runtime, dataset_loader=None) + + +def test_model_config_alone_cannot_pass_asset_preparation(client_site): + (Path(client_site.model_path) / "model-00001.safetensors").unlink() + with pytest.raises(ValueError, match="shard is missing or unbound"): + collect_assets(client_site) + + +def test_installed_interpreter_identity_exposes_external_base_paths(installed_child): + python, _implementation = installed_child + identity = capture_identity(str(python), ["lm-eval"], dataset_loader=None) + paths = identity["python_paths"] + assert paths["executable"] == str(python) + assert paths["executable_resolved"] == str(python.resolve()) + assert paths["prefix"] == str(python.parent.parent) + assert paths["base_prefix"] != paths["prefix"] + assert Path(paths["base_prefix"]).is_dir() + # A same-path mount of the venv alone cannot make its external interpreter + # and standard library available inside a client container. + assert not Path(paths["executable_resolved"]).is_relative_to(paths["prefix"]) + + +def test_client_source_and_secret_inputs_are_rejected(client_site): + with pytest.raises(ValueError, match="immutable reviewed revision"): + require_source_revision( + {"distributions": {"lm-eval": {"version": "0.1"}}}, "lm-eval", "a" * 40 + ) + with pytest.raises(ValueError, match="secret-bearing"): + ClientSite.model_validate( + { + **client_site.model_dump(), + "env": {**client_site.env, "HF_TOKEN": "not-persisted"}, + } + ) + + +def test_installed_wheel_prepares_and_verifies_resources_without_checkout( + tmp_path, client_site +): + uv = shutil.which("uv") or str(Path(sys.executable).with_name("uv")) + repository = Path(__file__).resolve().parents[1] + wheels = tmp_path / "wheels" + subprocess.run( + [uv, "build", "--wheel", "--out-dir", str(wheels)], + cwd=repository, + capture_output=True, + text=True, + check=True, + timeout=120, + ) + environment = tmp_path / "installed-wrapper" + subprocess.run( + [uv, "venv", "--python", sys.executable, str(environment)], + capture_output=True, + text=True, + check=True, + timeout=30, + ) + python = environment / "bin/python" + subprocess.run( + [ + uv, + "pip", + "install", + "--python", + str(python), + str(next(wheels.glob("*.whl"))), + ], + capture_output=True, + text=True, + check=True, + timeout=120, + ) + site_path = tmp_path / "site.json" + site_path.write_text(client_site.model_dump_json()) + outside = tmp_path / "outside-checkout" + outside.mkdir() + code = """ +import json +from pathlib import Path +from infx.benchmarks.prepare import ClientSite, prepare +from infx.benchmarks.common import read_json, verify_file +from infx.benchmarks.spec import PreparedFile +import sys +resources = prepare(ClientSite.model_validate(read_json(Path(sys.argv[1]))), 'eval', Path('prepared')) +task = PreparedFile.model_validate(resources['task']) +Path(task.path).chmod(0o644) +Path(task.path).write_text('task: changed-after-preparation\\n') +try: + verify_file(task) +except ValueError as error: + print(json.dumps({'error': str(error), 'resource_directory': str(Path(task.path).parent)})) +else: + raise SystemExit('mutated resource was accepted') +""" + result = subprocess.run( + [str(python), "-I", "-c", code, str(site_path)], + cwd=outside, + capture_output=True, + text=True, + check=True, + timeout=30, + ) + observed = json.loads(result.stdout) + assert "prepared file missing or changed" in observed["error"] + assert observed["resource_directory"] == str(outside / "prepared") + + +def populate(root, payload=b"client-generated bytes"): + key = "a" * 32 + entry = root / key + entry.mkdir() + (entry / "dataset.dat").write_bytes(payload) + (entry / "index.dat").write_bytes(b"index bytes") + (entry / "manifest.json").write_text( + json.dumps({"cache_key": key, "compressed": False}) + ) + return entry + + +def test_cache_cold_warm_copy_and_corrupt_snapshot_rebuild(tmp_path): + first = MmapCache( + tmp_path / "cache", {"dataset": "prepared"}, lock_timeout_seconds=0.1 + ) + source = populate(first.prepare()) + first.publish() + first.close() + canonical_payload = first.canonical / source.name / "dataset.dat" + warm = MmapCache( + tmp_path / "cache", {"dataset": "prepared"}, lock_timeout_seconds=0.1 + ) + restored = warm.prepare() / source.name / "dataset.dat" + assert restored.read_bytes() == b"client-generated bytes" + assert restored.stat().st_ino != canonical_payload.stat().st_ino + restored.write_bytes(b"private mutation") + assert canonical_payload.read_bytes() == b"client-generated bytes" + warm.close() + canonical_payload.write_bytes(b"truncated") + repair = MmapCache( + tmp_path / "cache", {"dataset": "prepared"}, lock_timeout_seconds=0.1 + ) + assert list(repair.prepare().iterdir()) == [] + assert len(list(repair.root.glob("quarantine-*"))) == 1 + populate(repair.run_dir, b"rebuilt bytes") + repair.publish() + repair.close() + assert canonical_payload.read_bytes() == b"rebuilt bytes" + + +def test_incomplete_cache_receipt_and_lock_contention_never_authorize_unlocked_publication( + tmp_path, +): + cache = MmapCache( + tmp_path / "cache", {"dataset": "prepared"}, lock_timeout_seconds=0.05 + ) + cache.canonical.mkdir() + populate(cache.canonical) + # A producer crashed before committing its integrity record. + assert list(cache.prepare().iterdir()) == [] + cache.close() + lock = cache.root / "publication.lock" + with lock.open("a+b") as stream: + fcntl.flock(stream.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB) + other = MmapCache( + tmp_path / "cache", {"dataset": "prepared"}, lock_timeout_seconds=0.05 + ) + populate(other.prepare()) + other.publish() + assert not other.canonical.exists() + assert any("independent cold cache" in event for event in other.events) + assert any("publication skipped" in event for event in other.events) + other.close() diff --git a/utils/test_phase1_receipt_control.py b/utils/test_phase1_receipt_control.py new file mode 100644 index 0000000000..690cc6a4bf --- /dev/null +++ b/utils/test_phase1_receipt_control.py @@ -0,0 +1,410 @@ +"""Behavioral checks for the independently reviewed qualification and publication controls.""" + +import hashlib +import io +import json +import subprocess +import tempfile +import unittest +import zipfile +from pathlib import Path +from unittest.mock import patch + +from infx.results.publication_receipt import ( + ExpectedContract, + Issuer, + PublicationRecord, + canonical_bytes, + seal_receipt, +) +from infx.workflows.phase1_publication import Approval, expected_contract +from infx.workflows.phase1_record import validate_record + + +def archive_bytes(name, content): + buffer = io.BytesIO() + with zipfile.ZipFile(buffer, "w") as archive: + archive.writestr(name, content) + return buffer.getvalue() + + +class QualificationTests(unittest.TestCase): + def setUp(self): + self.approval = { + "schema_version": 1, + "repository": "SemiAnalysisAI/InferenceX", + "source_run_id": "100", + "source_attempt": 1, + "source_head_sha": "a" * 40, + "points": [ + { + "point_id": f"{index + 1:064x}", + "execution_id": f"{index + 10:064x}", + "bundle_digest": f"{index + 20:064x}", + "native_manifest_sha256": f"{index + 30:064x}", + "kind": kind, + "concurrency": concurrency, + "dataset_revision": "c" * 40 if kind == "throughput" else None, + } + for index, (kind, concurrency) in enumerate( + [("throughput", c) for c in [1, 2, 4, 8, 16, 20, 24, 28]] + + [("eval", 28)] + ) + ], + } + self.inventory = [] + for index, point in enumerate(self.approval["points"]): + point_id = point["point_id"] + self.inventory.append( + { + "id": 1001 + index * 10, + "name": f"native-execution-{point_id}", + "expired": False, + } + ) + if point["kind"] == "throughput": + self.inventory.extend( + [ + { + "id": 1000 + index * 10, + "name": f"bmk_agentic_{point_id}", + "expired": False, + }, + { + "id": 1002 + index * 10, + "name": f"agentic_{point_id}", + "expired": False, + }, + ] + ) + else: + self.inventory.append( + { + "id": 1080, + "name": f"eval_{point_id}_lm-eval__1", + "expired": False, + } + ) + self.run = { + "id": 100, + "head_sha": "a" * 40, + "run_attempt": 1, + "status": "completed", + "conclusion": "success", + "path": ".github/workflows/run-sweep.yml", + } + self.patchers = [ + patch("infx.workflows.phase1_publication.api", side_effect=self.api), + patch( + "infx.workflows.phase1_publication.subprocess.run", + side_effect=self.download, + ), + ] + for patcher in self.patchers: + patcher.start() + self.addCleanup(patcher.stop) + + def api(self, endpoint, **kwargs): + if endpoint.endswith("/actions/runs/100"): + return self.run + if endpoint.endswith("/actions/runs/100/artifacts?per_page=100"): + return [{"artifacts": self.inventory}] + raise AssertionError(endpoint) + + def download(self, argv, *, stdout, check): + self.assertEqual( + argv[-1], "repos/SemiAnalysisAI/InferenceX/actions/artifacts/1080/zip" + ) + with zipfile.ZipFile(stdout, "w") as archive: + for name in [ + "results_fixed_conc28.json", + "samples_gsm8k_fixed_conc28.jsonl", + "meta_env.json", + ]: + archive.writestr(name, "{}") + return subprocess.CompletedProcess(argv, 0) + + def test_complete_pilot_binds_each_artifact_and_preserves_corpus_and_precision( + self, + ): + contract = expected_contract(Approval.model_validate(self.approval)) + self.assertEqual(len(contract.points), 9) + self.assertEqual(contract.points[0].artifact_ids, [1001, 1000, 1002]) + self.assertEqual(contract.points[0].config["precision"], "fp4") + self.assertEqual(contract.points[0].dataset["hf_revision"], "c" * 40) + self.assertEqual(contract.points[0].topology.serving_gpus, 8) + self.assertNotEqual( + contract.points[0].bundle_digest, contract.points[1].bundle_digest + ) + evaluation = contract.points[-1] + self.assertEqual(evaluation.artifact_ids, [1081, 1080]) + self.assertEqual(evaluation.normalized_format, "lm-eval") + self.assertEqual(evaluation.normalized_path, "results_fixed_conc28.json") + self.assertEqual(evaluation.samples_path, "samples_gsm8k_fixed_conc28.jsonl") + self.assertEqual(evaluation.sample_count, 1319) + self.assertEqual(evaluation.filters, ["strict-match", "flexible-extract"]) + + def test_partial_pilot_wrong_source_and_ambiguous_artifacts_fail(self): + with self.assertRaisesRegex(ValueError, "exactly eight throughput"): + Approval.model_validate( + self.approval | {"points": self.approval["points"][:-1]} + ) + self.run["id"] = 101 + with self.assertRaisesRegex(ValueError, "completed successful sweep"): + expected_contract(Approval.model_validate(self.approval)) + self.run["id"] = 100 + self.inventory.append(self.inventory[0] | {"id": 9999}) + with self.assertRaisesRegex(ValueError, "expected one unexpired artifact"): + expected_contract(Approval.model_validate(self.approval)) + + +class PublicationTests(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory() + self.addCleanup(self.temporary.cleanup) + root = Path(self.temporary.name) + with zipfile.ZipFile(root / "101.zip", "w") as archive: + archive.writestr( + "result.json", + json.dumps( + { + "conc": 1, + "tp": 8, + "ep": 1, + "num_gpus": 8, + "disagg": False, + "is_multinode": False, + "infmax_model_prefix": "dsr1", + "output_tput_tps": 100, + } + ), + ) + archive.writestr( + "execution.json", + json.dumps( + { + "schema_version": 1, + "point_id": "c" * 64, + "execution_id": "owned:55", + "bundle_digest": "b" * 64, + "source": { + "repository": "org/repo", + "run_id": 100, + "attempt": 1, + "head_sha": "a" * 40, + }, + "mode": "throughput", + "native_receipt": { + "job_id": "55", + "manifest_sha256": "f" * 64, + "state": "COMPLETED", + }, + "client_exit_code": 0, + } + ), + ) + expected = ExpectedContract.model_validate( + { + "repository": "org/repo", + "source_run_id": "100", + "source_attempt": 1, + "source_head_sha": "a" * 40, + "bundle_digest": "b" * 64, + "contracts": { + "raw": "aiperf-1.4", + "normalized": "agentx-v1", + "publication": 1, + }, + "points": [ + { + "point_id": "c" * 64, + "bundle_digest": "b" * 64, + "execution_id": "owned:55", + "source_run_id": "100", + "source_attempt": 1, + "kind": "throughput", + "concurrency": 1, + "topology": { + "kind": "aggregate", + "nodes": 1, + "serving_gpus": 8, + "tp": 8, + "ep": 1, + }, + "artifact_ids": [101], + "execution_artifact_id": 101, + "execution_path": "execution.json", + "native_manifest_sha256": "f" * 64, + "normalized_artifact_id": 101, + "normalized_path": "result.json", + "required_metrics": ["output_tput_tps"], + "config": {"model": "dsr1"}, + } + ], + } + ) + receipt = seal_receipt( + expected, + Issuer( + repository="org/repo", + run_id="200", + job="seal", + workflow_sha="d" * 40, + collector_sha="d" * 40, + ), + [ + { + "id": 101, + "name": "native-execution-point", + "expired": False, + "workflow_run": {"id": 100}, + "digest": "sha256:" + + hashlib.sha256((root / "101.zip").read_bytes()).hexdigest(), + } + ], + root, + ) + self.receipt = receipt.model_dump() + self.archives = { + 301: archive_bytes("receipt.json", json.dumps(self.receipt)), + 501: archive_bytes("changelog-metadata.json", "{}"), + } + self.metadata = { + key: { + "id": key, + "name": name, + "expired": False, + "workflow_run": {"id": owner}, + "digest": "sha256:" + hashlib.sha256(self.archives[key]).hexdigest(), + } + for key, name, owner in [ + (301, "measurement-receipt", 200), + (501, "changelog-metadata", 150), + ] + } + self.runs = { + "200": { + "id": 200, + "head_sha": "d" * 40, + "status": "completed", + "conclusion": "success", + "head_branch": "main", + "event": "workflow_dispatch", + "path": ".github/workflows/phase1-receipt.yml", + }, + "100/attempts/1": { + "id": 100, + "run_attempt": 1, + "head_sha": "a" * 40, + "status": "completed", + "conclusion": "success", + }, + "150": { + "id": 150, + "head_sha": "e" * 40, + "status": "completed", + "conclusion": "success", + "head_branch": "main", + "event": "push", + "path": ".github/workflows/run-sweep.yml", + }, + } + self.record = PublicationRecord.model_validate( + { + "kind": "publication-record", + "version": 1, + "receipt_id": receipt.receipt_id, + "receipt_artifact_id": 301, + "receipt_artifact_sha256": self.metadata[301]["digest"][7:], + "source_run_id": "100", + "merge_run_id": "150", + "merge_sha": "e" * 40, + "changelog_artifact_id": 501, + "changelog_artifact_sha256": self.metadata[501]["digest"][7:], + "ingest_sha": "f" * 40, + "app_sha": "f" * 40, + } + ) + for patcher in [ + patch("infx.workflows.phase1_record.api", side_effect=self.api), + patch( + "infx.workflows.phase1_record.subprocess.run", side_effect=self.download + ), + ]: + patcher.start() + self.addCleanup(patcher.stop) + + def api(self, endpoint): + prefix = "repos/org/repo/actions/" + if endpoint.startswith(prefix + "artifacts/"): + return self.metadata[int(endpoint.rsplit("/", 1)[-1])] + if endpoint.startswith(prefix + "runs/"): + return self.runs[endpoint.removeprefix(prefix + "runs/")] + raise AssertionError(endpoint) + + def download(self, argv, **kwargs): + return subprocess.CompletedProcess( + argv, 0, stdout=self.archives[int(argv[-1].split("/")[-2])] + ) + + def validate(self): + return validate_record(self.record, "org/repo", {"d" * 40}, "f" * 40) + + def replace_receipt_archive(self, text): + self.archives[301] = archive_bytes("receipt.json", text) + digest = hashlib.sha256(self.archives[301]).hexdigest() + self.metadata[301]["digest"] = "sha256:" + digest + self.record = self.record.model_copy(update={"receipt_artifact_sha256": digest}) + + def test_publication_preserves_original_measurement_and_pins_later_merge(self): + record = self.validate() + self.assertEqual(record.source_run_id, "100") + self.assertEqual(record.merge_run_id, "150") + self.assertEqual(record.receipt_artifact_id, 301) + self.assertEqual(record.changelog_artifact_id, 501) + + def test_wrong_api_owner_source_attempt_and_reader_pin_fail(self): + for mapping, key, invalid, message in [ + (self.metadata[301], "id", 999, "missing, expired or misnamed"), + ( + self.runs["200"], + "event", + "pull_request", + "approved completed trusted workflow", + ), + ( + self.runs["100/attempts/1"], + "head_sha", + "b" * 40, + "completed original attempt", + ), + (self.runs["150"], "id", 151, "completed successful main sweep"), + (self.metadata[501]["workflow_run"], "id", 151, "different merge run"), + ]: + with self.subTest(key=key, invalid=invalid): + original, mapping[key] = mapping[key], invalid + with self.assertRaisesRegex(ValueError, message): + self.validate() + mapping[key] = original + self.record = self.record.model_copy(update={"app_sha": "9" * 40}) + with self.assertRaisesRegex(ValueError, "deployed reader pin"): + self.validate() + + def test_receipt_repository_and_duplicate_json_keys_cannot_change_authority(self): + original = json.dumps(self.receipt) + self.replace_receipt_archive('{"receipt_id":"' + "0" * 64 + '",' + original[1:]) + with self.assertRaisesRegex(ValueError, "duplicate JSON key"): + self.validate() + self.receipt["issuer"]["repository"] = "another/repository" + payload = { + key: value for key, value in self.receipt.items() if key != "receipt_id" + } + self.receipt["receipt_id"] = hashlib.sha256( + canonical_bytes(payload) + ).hexdigest() + self.record = self.record.model_copy( + update={"receipt_id": self.receipt["receipt_id"]} + ) + self.replace_receipt_archive(json.dumps(self.receipt)) + with self.assertRaisesRegex(ValueError, "replace source measurement identity"): + self.validate() diff --git a/utils/test_publication_receipt.py b/utils/test_publication_receipt.py new file mode 100644 index 0000000000..6d8dbfd929 --- /dev/null +++ b/utils/test_publication_receipt.py @@ -0,0 +1,276 @@ +"""Behavioral checks for independently issued immutable artifact snapshots.""" + +import hashlib +import json +import stat +import tempfile +import unittest +import zipfile +from pathlib import Path + +from infx.results.publication_receipt import ( + ExpectedContract, + Issuer, + inspect_archive, + seal_receipt, + verify_receipt, +) + + +class ReceiptTests(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.root = Path(self.temp.name) + self.addCleanup(self.temp.cleanup) + with zipfile.ZipFile(self.root / "101.zip", "w") as archive: + archive.writestr( + "result.json", + '{"conc":1,"tp":8,"ep":1,"num_gpus":8,"disagg":false,"is_multinode":false,"infmax_model_prefix":"dsr1","output_tput_tps":100}', + ) + archive.writestr( + "execution.json", + json.dumps( + { + "schema_version": 1, + "point_id": "c" * 64, + "execution_id": "intent:cluster:55:0", + "bundle_digest": "b" * 64, + "source": { + "repository": "org/repo", + "run_id": 100, + "attempt": 1, + "head_sha": "a" * 40, + }, + "mode": "throughput", + "native_receipt": { + "job_id": "55", + "manifest_sha256": "f" * 64, + "state": "COMPLETED", + }, + "client_exit_code": 0, + } + ), + ) + self.expected = ExpectedContract.model_validate( + { + "repository": "org/repo", + "source_run_id": "100", + "source_attempt": 2, + "source_head_sha": "a" * 40, + "bundle_digest": "b" * 64, + "contracts": { + "raw": "aiperf-1.4", + "normalized": "agentx-v1", + "publication": 1, + }, + "points": [ + { + "point_id": "c" * 64, + "bundle_digest": "b" * 64, + "execution_id": "intent:cluster:55:0", + "source_run_id": "100", + "source_attempt": 1, + "kind": "throughput", + "concurrency": 1, + "topology": { + "kind": "aggregate", + "nodes": 1, + "serving_gpus": 8, + "tp": 8, + "ep": 1, + }, + "artifact_ids": [101], + "normalized_artifact_id": 101, + "normalized_path": "result.json", + "required_metrics": ["output_tput_tps"], + "config": {"model": "dsr1"}, + "execution_artifact_id": 101, + "execution_path": "execution.json", + "native_manifest_sha256": "f" * 64, + } + ], + } + ) + self.issuer = Issuer( + repository="org/repo", + run_id="200", + job="validate", + workflow_sha="d" * 40, + collector_sha="e" * 40, + ) + digest = hashlib.sha256((self.root / "101.zip").read_bytes()).hexdigest() + self.inventory = [ + { + "id": 101, + "name": "bmk_result", + "expired": False, + "workflow_run": {"id": 100}, + "digest": f"sha256:{digest}", + } + ] + + def test_preserves_original_execution_and_exact_uploaded_member(self): + receipt = seal_receipt(self.expected, self.issuer, self.inventory, self.root) + self.assertEqual(receipt.points[0].source_attempt, 1) + self.assertEqual(receipt.source_attempt, 2) + self.assertEqual(receipt.artifacts[0].members[1].path, "result.json") + self.assertEqual( + verify_receipt(json.loads(receipt.model_dump_json())).receipt_id, + receipt.receipt_id, + ) + changed = receipt.model_dump() + changed["points"][0]["source_attempt"] = 2 + with self.assertRaisesRegex(ValueError, "digest mismatch"): + verify_receipt(changed) + + def test_missing_wrong_run_expired_and_changed_bytes_fail(self): + bad_inventory = [ + [], + [self.inventory[0] | {"workflow_run": {"id": 99}}], + [self.inventory[0] | {"expired": True}], + [self.inventory[0] | {"digest": "sha256:" + "0" * 64}], + ] + for rows in bad_inventory: + with self.subTest(rows=rows), self.assertRaises(ValueError): + seal_receipt(self.expected, self.issuer, rows, self.root) + + def test_rejects_archive_traversal_duplicate_and_link_members(self): + for names in [["../escape"], ["same", "same"], ["/absolute"], ["a\\b"]]: + with self.subTest(names=names): + archive_path = self.root / "bad.zip" + with zipfile.ZipFile(archive_path, "w") as archive: + for name in names: + archive.writestr(name, "data") + with self.assertRaises(ValueError): + inspect_archive(archive_path) + with zipfile.ZipFile(self.root / "link.zip", "w") as archive: + item = zipfile.ZipInfo("link") + item.external_attr = (stat.S_IFLNK | 0o777) << 16 + archive.writestr(item, "target") + with self.assertRaisesRegex(ValueError, "link"): + inspect_archive(self.root / "link.zip") + + def test_semantic_failure_is_not_repaired_by_a_fresh_api_digest(self): + with zipfile.ZipFile(self.root / "101.zip") as archive: + execution = archive.read("execution.json") + for change in ( + {"output_tput_tps": -1}, + {"infmax_model_prefix": "wrong-model"}, + {"conc": 28}, + {"num_gpus": 4}, + ): + with self.subTest(change=change): + row = { + "conc": 1, + "tp": 8, + "ep": 1, + "num_gpus": 8, + "disagg": False, + "is_multinode": False, + "infmax_model_prefix": "dsr1", + "output_tput_tps": 100, + } | change + with zipfile.ZipFile(self.root / "101.zip", "w") as archive: + archive.writestr("result.json", json.dumps(row)) + archive.writestr("execution.json", execution) + inventory = [ + self.inventory[0] + | { + "digest": "sha256:" + + hashlib.sha256( + (self.root / "101.zip").read_bytes() + ).hexdigest() + } + ] + with self.assertRaises(ValueError): + seal_receipt(self.expected, self.issuer, inventory, self.root) + + def test_raw_eval_gate_matches_summary_to_both_filter_sets(self): + from infx.results.publication_receipt import Point, validate_point_content + + point = Point.model_validate( + self.expected.points[0].model_dump() + | { + "kind": "eval", + "concurrency": 28, + "normalized_format": "lm-eval", + "metadata_path": "meta_env.json", + "task": "gsm8k", + "sample_count": 2, + "filters": ["strict-match", "flexible-extract"], + "samples_artifact_id": 101, + "samples_path": "samples.jsonl", + "required_metrics": ["em_strict", "n_eff"], + } + ) + meta = { + "conc": 28, + "tp": 8, + "ep": 1, + "num_gpus": 8, + "disagg": False, + "is_multinode": False, + "infmax_model_prefix": "dsr1", + "prefill_num_workers": 0, + "decode_num_workers": 0, + } + samples = [ + { + "doc_id": doc, + "task_name": "gsm8k", + "filter": name, + "exact_match": int(doc == 0 or name == "flexible-extract"), + } + for doc in range(2) + for name in point.filters + ] + for summary, succeeds in [(0.5, True), (1.0, False)]: + with zipfile.ZipFile(self.root / "101.zip", "w") as archive: + archive.writestr( + "result.json", + json.dumps( + { + "results": {"gsm8k": {"exact_match,strict-match": summary}}, + "configs": { + "gsm8k": { + "metric_list": [{"metric": "exact_match"}], + "filter_list": [ + {"name": "strict-match"}, + {"name": "flexible-extract"}, + ], + } + }, + "n-samples": {"gsm8k": {"effective": 2}}, + } + ), + ) + archive.writestr("meta_env.json", json.dumps(meta)) + archive.writestr( + "samples.jsonl", "\n".join(json.dumps(sample) for sample in samples) + ) + if succeeds: + validate_point_content(point, self.root) + else: + with self.assertRaisesRegex(ValueError, "strict summary"): + validate_point_content(point, self.root) + + def test_execution_evidence_rejects_ambiguous_json_and_boolean_status(self): + with zipfile.ZipFile(self.root / "101.zip") as archive: + result = archive.read("result.json") + execution = json.loads(archive.read("execution.json")) + candidates = [ + json.dumps(execution | {"schema_version": True}), + json.dumps(execution | {"client_exit_code": False}), + '{"client_exit_code":1,' + json.dumps(execution)[1:], + ] + for candidate in candidates: + with self.subTest(candidate=candidate): + with zipfile.ZipFile(self.root / "101.zip", "w") as archive: + archive.writestr("result.json", result) + archive.writestr("execution.json", candidate) + metadata = self.inventory[0] | { + "digest": "sha256:" + + hashlib.sha256((self.root / "101.zip").read_bytes()).hexdigest() + } + with self.assertRaises(ValueError): + seal_receipt(self.expected, self.issuer, [metadata], self.root) diff --git a/utils/test_python_benchmark_clients.py b/utils/test_python_benchmark_clients.py new file mode 100644 index 0000000000..6cfabb2bb2 --- /dev/null +++ b/utils/test_python_benchmark_clients.py @@ -0,0 +1,594 @@ +"""Behavioral contracts for prepared clients; no GPU, server, or package install required.""" + +from __future__ import annotations + +import hashlib +import json +import os +import signal +import subprocess +import sys +import time +from pathlib import Path + +import pytest +import yaml + +from infx.benchmarks.agentx import ( + build_argv as agentx_argv, +) +from infx.benchmarks.agentx import ( + normalize, + replay_environment, + validate_scenario, + verify_corpus, +) +from infx.benchmarks.agentx import ( + run as run_agentx, +) +from infx.benchmarks.common import read_json, run_child +from infx.benchmarks.eval import ( + build_argv as eval_argv, +) +from infx.benchmarks.eval import ( + eval_metadata, + packaged_task_path, + stage_outputs, + validate_outputs, +) +from infx.benchmarks.identity import verify_runtime +from infx.benchmarks.spec import AgentXSpec, EvalSpec + + +def bound_file(path: Path) -> dict: + return {"path": str(path), "sha256": hashlib.sha256(path.read_bytes()).hexdigest()} + + +@pytest.fixture +def spec_inputs(tmp_path): + identity = tmp_path / "identity.json" + identity.write_text( + '{"dataset_resolution":{"metadata":{"hf_dataset_name":"semianalysisai/cc-traces-weka-062126"}},' + '"distributions":{"aiperf":{"direct_url":{"vcs_info":{"commit_id":' + '"754356e9a39acc6cc6afb242d123bb57c3fb6f75"}}}}}' + ) + dataset = tmp_path / "hub/datasets--semianalysisai--cc-traces-weka-062126" + reference = dataset / "refs/main" + reference.parent.mkdir(parents=True) + reference.write_text("f" * 40) + data = dataset / "snapshots" / ("f" * 40) / "traces.jsonl" + data.parent.mkdir(parents=True) + data.write_text('{"trace_id":"one"}\n') + return { + "schema_version": 1, + "runtime": { + "python": sys.executable, + "identity": bound_file(identity), + "distributions": ["aiperf"], + "env": { + "HF_HUB_OFFLINE": "1", + "HF_DATASETS_OFFLINE": "1", + "HF_HUB_CACHE": str(tmp_path / "hub"), + "HF_DATASETS_CACHE": str(tmp_path / "datasets"), + "AIPERF_DATASET_MMAP_CACHE_DIR": str(tmp_path / "mmap"), + }, + "env_unset": ["UNWANTED_CLIENT_SETTING"], + "assets": [bound_file(reference), bound_file(data)], + "timeout_seconds": 10, + "terminate_grace_seconds": 1, + }, + "metadata": { + "hw": "cluster:h100-fixture", + "model": "fixture/model", + "model_prefix": "fixture", + "image": "fixture@sha256:" + "1" * 64, + "framework": "vllm", + "precision": "fp4", + "spec_decoding": "mtp", + "tp": 8, + "pp": 1, + "dcp_size": 1, + "pcp_size": 1, + "ep": 1, + "dp_attention": False, + "total_cpu_dram_gb": 1024, + "recipe_fingerprint": "2" * 64, + }, + "concurrency": 28, + } + + +@pytest.fixture +def agentx_spec(spec_inputs): + return AgentXSpec.model_validate( + { + **spec_inputs, + "result_filename": "agg_fixture", + "tokenizer": "fixture/model", + "dataset_revision": "f" * 40, + "dataset_loader": "semianalysis_cc_traces_weka_062126", + "dataset_repository": "semianalysisai/cc-traces-weka-062126", + "dataset_entries": 393, + "duration_seconds": 3600, + "warmup_requests_per_lane": 10, + "warmup_grace_seconds": 1800, + "trace_idle_gap_cap_seconds": 300, + "live_failed_request_threshold": 0.1, + "failed_request_threshold": 0.1, + "random_seed": 42, + "required_server_metric_prefix": "vllm:", + } + ) + + +def export_config(): + return { + "request_count": {"avg": 2}, + "error_request_count": {"avg": 0}, + "metadata": { + "scenario": "inferencex-agentx-mvp", + "submission_valid": True, + "dataset": { + "source_type": "public_dataset", + "loader": "semianalysis_cc_traces_weka_062126", + "hf_dataset_name": "semianalysisai/cc-traces-weka-062126", + "hf_split": "train", + "num_dataset_entries": 393, + }, + "metric_duration_coverage": [ + { + "expected_duration_seconds": 3600.0, + "required_ratio": 0.95, + "ttft_ratio": 0.96, + "inter_token_latency_ratio": 0.2, + } + ], + }, + "input_config": { + "models": {"items": [{"name": "fixture/model"}]}, + "endpoint": { + "urls": ["http://worker.example:9123"], + "type": "chat", + "path": "/v1/chat/completions", + "streaming": True, + "use_server_token_count": True, + }, + "tokenizer": {"name": "fixture/model"}, + "phases": [ + { + "kind": "profiling", + "type": "concurrency", + "timing_mode": "agentic_replay", + "duration": 3600, + "concurrency": 28, + "trajectory_start_min_ratio": 0.25, + "trajectory_start_max_ratio": 0.75, + "system_idle_gap_cap_seconds": 10, + "warmup_requests_per_lane": 10, + "agentic_warmup_grace_period": 1800, + "failed_request_threshold": 0.1, + } + ], + "datasets": [ + { + "dataset": "semianalysis_cc_traces_weka_062126", + "entries": 393, + "random_seed": 42, + "trace_idle_gap_cap_seconds": 300, + } + ], + }, + } + + +def record(start, end, tokens, *, phase="profiling", error=None): + return { + "metadata": { + "request_start_ns": start, + "request_end_ns": end, + "benchmark_phase": phase, + }, + "metrics": { + "output_sequence_length": {"value": tokens, "unit": "tokens"}, + "input_sequence_length": {"value": 10, "unit": "tokens"}, + }, + "error": error, + } + + +def raw_agentx(root): + raw = root / "aiperf_artifacts" + raw.mkdir(parents=True) + (raw / "profile_export.jsonl").write_text( + "\n".join( + json.dumps(row) + for row in [ + record(1_000_000_000, 3_000_000_000, 30), + record(4_000_000_000, 7_000_000_000, 60), + record(0, 8_000_000_000, 500, phase="warmup"), + record(0, 9_000_000_000, 900, error={"type": "cancelled"}), + ] + ) + + "\n" + ) + (raw / "profile_export_aiperf.json").write_text(json.dumps(export_config())) + (raw / "server_metrics_export.json").write_text( + '{"vllm:request_success_total": {}}' + ) + (raw / "server_metrics_export.csv").write_text( + "metric,value\nvllm:request_success_total,2\n" + ) + return raw + + +def test_one_child_process_retains_raw_and_existing_normalization( + tmp_path, agentx_spec, monkeypatch +): + fixture = tmp_path / "producer" + raw_agentx(fixture) + external = tmp_path / "installed child" + external.write_text( + f"#!{sys.executable}\n" + "import json,os,pathlib,shutil,sys\n" + "if '--distribution' in sys.argv:\n" + f" print(pathlib.Path({agentx_spec.runtime.identity.path!r}).read_text())\n" + "else:\n" + " out=pathlib.Path(sys.argv[sys.argv.index('--output-artifact-dir')+1])\n" + f" shutil.copytree({str(fixture / 'aiperf_artifacts')!r},out)\n" + " (out.parent/'child-observed.json').write_text(json.dumps({'argv':sys.argv[1:]," + "'ambient':os.environ.get('UNWANTED_CLIENT_SETTING')," + "'poison':os.environ.get('AIPERF_DATASET_RANDOM_SEED')," + "'cwd':os.getcwd()}))\n" + ) + external.chmod(0o755) + spec = agentx_spec.model_copy( + update={ + "runtime": agentx_spec.runtime.model_copy(update={"python": str(external)}) + } + ) + monkeypatch.setenv("UNWANTED_CLIENT_SETTING", "leaked") + monkeypatch.setenv("AIPERF_DATASET_RANDOM_SEED", "1") + output = tmp_path / "outputs with spaces" + assert run_agentx(spec, "http://worker.example:9123", output) == 0 + observed = read_json(output / "child-observed.json") + assert observed["argv"].count("profile") == 1 + assert observed["ambient"] is None and observed["poison"] is None + assert observed["cwd"] == str(output) + aggregate = read_json(output / "agg_fixture.json") + # Two successful requests produce 90 tokens over their six-second span. + assert ( + aggregate["request_metrics"]["throughput"]["output"]["tokens_per_second"] == 15 + ) + assert aggregate["request_metrics"]["throughput"]["duration_seconds"] == 6 + assert aggregate["request_accounting"]["records_warmup_dropped"] == 1 + assert aggregate["request_accounting"]["records_error_dropped"] == 1 + assert aggregate["num_gpus"] == 8 and aggregate["is_multinode"] is False + assert ( + read_json(output / "diagnostics/client-audit.json")["status"]["returncode"] == 0 + ) + + +def test_literal_argv_and_resolved_remote_endpoint(tmp_path, agentx_spec): + argv = agentx_argv( + agentx_spec, "http://remote.example:9009/", tmp_path / "literal ; $(no)" + ) + assert argv[argv.index("--url") + 1] == "http://remote.example:9009" + assert ( + argv[argv.index("--server-metrics") + 1] == "http://remote.example:9009/metrics" + ) + assert argv[argv.index("--output-artifact-dir") + 1].endswith( + "literal ; $(no)/aiperf_artifacts" + ) + assert "--max-context-length" not in argv + assert "--unsafe-override" not in argv + + +def test_normalizer_uses_native_serving_log_context(tmp_path, agentx_spec, monkeypatch): + output = tmp_path / "output" + raw_agentx(output) + logs = tmp_path / "native-logs" + logs.mkdir() + (logs / "node_agg_w0.out").write_text("INFO GPU KV cache size: 1,234,567 tokens\n") + # A different point's old log must not override the supplied native context. + (tmp_path / "watchtower-other.out").write_text( + "INFO GPU KV cache size: 8,000,000 tokens\n" + ) + monkeypatch.setenv("SRT_LOG_DIR", str(logs)) + normalized = read_json(normalize(agentx_spec, output)) + assert normalized["kv_cache_pool_tokens"] == 1_234_567 + + +@pytest.mark.parametrize( + "change,message", + [ + ( + lambda value: value["input_config"]["phases"][0].update( + warmup_requests_per_lane=1 + ), + "warmup_requests", + ), + ( + lambda value: value["input_config"]["datasets"][0].update( + max_context_length=100 + ), + "context cap", + ), + ( + lambda value: value["metadata"].update(submission_valid=False), + "scenario validity", + ), + ( + lambda value: value["metadata"]["metric_duration_coverage"][0].update( + ttft_ratio=0.94 + ), + "neither TTFT", + ), + ], +) +def test_scenario_canonical_contract_is_independent_of_scenario_verdict( + agentx_spec, change, message +): + aggregate = export_config() + change(aggregate) + errors = validate_scenario(aggregate, agentx_spec, "http://worker.example:9123") + assert any(message in error for error in errors) + + +def test_mutated_prepared_dataset_fails_before_client(agentx_spec): + data = Path(agentx_spec.runtime.assets[-1].path) + data.write_text("changed") + with pytest.raises(ValueError, match="prepared file missing or changed"): + verify_runtime(agentx_spec.runtime, dataset_loader=agentx_spec.dataset_loader) + + +def test_unbound_dataset_content_and_ambient_scenario_override_are_rejected( + agentx_spec, +): + Path(agentx_spec.runtime.assets[-1].path).with_name("extra.json").write_text("{}") + with pytest.raises(ValueError, match="absent from the prepared asset list"): + verify_corpus(agentx_spec) + runtime = agentx_spec.runtime.model_copy( + update={"env": {**agentx_spec.runtime.env, "AIPERF_UNSAFE_OVERRIDE": "true"}} + ) + with pytest.raises(ValueError, match="unqualified"): + replay_environment(agentx_spec.model_copy(update={"runtime": runtime})) + + +def test_nonfinite_json_and_duplicate_keys_fail(tmp_path): + path = tmp_path / "value.json" + for text in ('{"metric": 1e400}', '{"metric":NaN}', '{"metric":1,"metric":2}'): + path.write_text(text) + with pytest.raises(ValueError): + read_json(path) + + +def test_client_timeout_kills_process_group_and_preserves_log(tmp_path): + code = "import signal,time; signal.signal(signal.SIGTERM,signal.SIG_IGN); print('started',flush=True); time.sleep(30)" + result = run_child( + [sys.executable, "-c", code], + env=os.environ, + cwd=tmp_path, + log=tmp_path / "client.log", + timeout_seconds=0.2, + terminate_grace_seconds=0.1, + ) + assert result == { + "returncode": -signal.SIGKILL, + "cancelled_by_signal": None, + "timed_out": True, + "orphaned_descendants": False, + } + assert (tmp_path / "client.log").read_text() == "started\n" + + +def test_repeated_term_is_forwarded_and_does_not_restart_shutdown(tmp_path): + ready = tmp_path / "ready" + child_code = ( + "import pathlib,signal,time; " + "signal.signal(signal.SIGTERM,signal.SIG_IGN); " + f"pathlib.Path({str(ready)!r}).write_text('ready'); time.sleep(30)" + ) + parent_code = ( + "import json,os,sys; from pathlib import Path; " + f"sys.path.insert(0,{str(Path(__file__).resolve().parents[1])!r}); " + "from infx.benchmarks.common import run_child; " + f"status=run_child([sys.executable,'-c',{child_code!r}],env=os.environ," + "cwd=Path.cwd(),log=Path('child.log'),timeout_seconds=10,terminate_grace_seconds=0.2); " + "Path('status.json').write_text(json.dumps(status))" + ) + parent = subprocess.Popen([sys.executable, "-c", parent_code], cwd=tmp_path) + try: + deadline = time.monotonic() + 5 + while not ready.exists() and time.monotonic() < deadline: + time.sleep(0.01) + assert ready.exists() + parent.send_signal(signal.SIGTERM) + time.sleep(0.03) + parent.send_signal(signal.SIGTERM) + assert parent.wait(timeout=2) == 0 + finally: + if parent.poll() is None: + parent.kill() + parent.wait() + status = read_json(tmp_path / "status.json") + assert status["cancelled_by_signal"] == signal.SIGTERM + assert status["timed_out"] is False + assert status["returncode"] == -signal.SIGKILL + + +def test_successful_child_with_surviving_writer_is_not_accepted(tmp_path): + descendant = "import signal,time; signal.signal(signal.SIGTERM,signal.SIG_IGN); time.sleep(30)" + code = ( + "import subprocess,sys; " + f"subprocess.Popen([sys.executable,'-c',{descendant!r}]); " + "print('leader exited',flush=True)" + ) + status = run_child( + [sys.executable, "-c", code], + env=os.environ, + cwd=tmp_path, + log=tmp_path / "client.log", + timeout_seconds=5, + terminate_grace_seconds=0.1, + ) + assert status["returncode"] == 0 + assert status["orphaned_descendants"] is True + + +@pytest.fixture +def eval_case(tmp_path, spec_inputs): + documents = { + str(index): {"question": f"What is {index}+1?", "answer": f"#### {index + 1}"} + for index in range(1319) + } + identities = { + key: hashlib.sha256( + json.dumps(doc, indent=2, ensure_ascii=False).encode() + ).hexdigest() + for key, doc in documents.items() + } + identity_path = tmp_path / "documents.json" + identity_path.write_text(json.dumps(identities)) + inputs = { + **spec_inputs, + "runtime": {**spec_inputs["runtime"], "distributions": ["lm-eval"]}, + "task": bound_file(packaged_task_path()), + "document_identities": bound_file(identity_path), + "task_name": "gsm8k", + "expected_documents": 1319, + "max_length": 16384, + "max_tokens": 12288, + "minimum_score": 0.9, + } + spec = EvalSpec.model_validate(inputs) + task = yaml.safe_load(packaged_task_path().read_text()) + task["generation_kwargs"].update(max_tokens=12288, temperature=0, top_p=1) + result = { + "config": { + "model": "local-chat-completions", + "limit": None, + "model_args": { + "model": "fixture/model", + "base_url": "http://worker:9000/v1/chat/completions", + "num_concurrent": 28, + "max_length": 16384, + "tokenized_requests": False, + }, + "gen_kwargs": {"max_tokens": 12288, "temperature": 0, "top_p": 1}, + }, + "configs": {"gsm8k": task}, + "n-samples": {"gsm8k": {"original": 1319, "effective": 1319}}, + "results": { + "gsm8k": { + "exact_match,strict-match": 1318 / 1319, + "exact_match,flexible-extract": 1318 / 1319, + "exact_match_stderr,strict-match": 0.01, + } + }, + } + result_path = tmp_path / "results.json" + result_path.write_text(json.dumps(result)) + samples = [] + # Reversed filter order must preserve coverage and score checks. + for name in ("flexible-extract", "strict-match"): + for key, doc in documents.items(): + samples.append( + { + "doc_id": int(key), + "filter": name, + "doc": doc, + "doc_hash": identities[key], + "target": doc["answer"], + "target_hash": hashlib.sha256(doc["answer"].encode()).hexdigest(), + "exact_match": 0.0 if key == "0" else 1.0, + } + ) + sample_path = tmp_path / "samples.jsonl" + sample_path.write_text("\n".join(json.dumps(row) for row in samples) + "\n") + return spec, result_path, sample_path, result, samples + + +def test_real_eval_contract_accepts_complete_split_and_canonical_aggregate( + eval_case, tmp_path +): + spec, results, samples, _, _ = eval_case + assert validate_outputs(spec, "http://worker:9000", [results], [samples]) == [] + argv = eval_argv(spec, "http://worker:9000", tmp_path) + assert ( + "base_url=http://worker:9000/v1/chat/completions" + in argv[argv.index("--model_args") + 1] + ) + assert "max_length=16384" in argv[argv.index("--model_args") + 1] + assert ( + argv[argv.index("--gen_kwargs") + 1] == "max_tokens=12288,temperature=0,top_p=1" + ) + metadata = eval_metadata(spec, complete=True) + assert ( + metadata["num_gpus"], + metadata["prefill_num_workers"], + metadata["decode_num_workers"], + ) == (8, 0, 0) + assert metadata["deployment"] == { + "kind": "aggregate", + "nodes": 1, + "serving_gpus": 8, + "tp": 8, + "ep": 1, + } + + +@pytest.mark.parametrize( + "mutation,match", + [ + (lambda r: r["config"]["model_args"].update(max_length=1048576), "max_length"), + (lambda r: r["config"].update(limit=0.1), "full-split"), + ( + lambda r: r["results"]["gsm8k"].update({"exact_match,strict-match": 1.0}), + "disagrees with raw samples", + ), + ], +) +def test_eval_rejects_self_consistent_wrong_budget_smoke_and_score( + eval_case, mutation, match +): + spec, result_path, sample_path, result, _ = eval_case + mutation(result) + result_path.write_text(json.dumps(result)) + assert any( + match in error + for error in validate_outputs( + spec, "http://worker:9000", [result_path], [sample_path] + ) + ) + + +def test_eval_rejects_duplicate_missing_and_changed_document(eval_case): + spec, result_path, sample_path, _, samples = eval_case + samples[-1] = samples[0] + samples[1]["doc"]["question"] = "unrelated task" + sample_path.write_text("\n".join(json.dumps(row) for row in samples) + "\n") + errors = validate_outputs(spec, "http://worker:9000", [result_path], [sample_path]) + assert any("duplicate eval sample" in error for error in errors) + assert any("coverage incomplete" in error for error in errors) + assert any("document bytes/hash" in error for error in errors) + + +def test_eval_rejects_nonfinite_secondary_metric(eval_case): + spec, result_path, sample_path, result, _ = eval_case + result["results"]["gsm8k"]["exact_match_stderr,strict-match"] = float("inf") + result_path.write_text(json.dumps(result)) + with pytest.raises(ValueError, match="non-finite"): + validate_outputs(spec, "http://worker:9000", [result_path], [sample_path]) + + +def test_staging_keeps_partial_evidence_and_rejects_collision(eval_case, tmp_path): + spec, _, _, _, _ = eval_case + raw = tmp_path / "harness/nested" + raw.mkdir(parents=True) + (raw / "results_failure.json").write_bytes(b'{"failed": true}\n') + (raw / "samples_partial.jsonl").write_bytes(b'{"doc_id":0}\n') + results, samples = stage_outputs(spec, tmp_path) + assert results[0].read_bytes() == b'{"failed": true}\n' + assert samples[0].read_bytes() == b'{"doc_id":0}\n' + with pytest.raises(FileExistsError): + stage_outputs(spec, tmp_path) diff --git a/utils/test_receipt_transport.py b/utils/test_receipt_transport.py new file mode 100644 index 0000000000..404ea22503 --- /dev/null +++ b/utils/test_receipt_transport.py @@ -0,0 +1,265 @@ +"""Read-only transport resolution against controlled GitHub API/archive responses.""" + +import hashlib +import io +import json +import subprocess +import tempfile +import unittest +import zipfile +from pathlib import Path +from unittest.mock import patch + +from infx.results.publication_receipt import ExpectedContract, Issuer, seal_receipt +from infx.workflows.receipt_transport import PendingReceiptError, resolve_transport + + +def archive_bytes(name, payload): + buffer = io.BytesIO() + with zipfile.ZipFile(buffer, "w") as archive: + archive.writestr(name, json.dumps(payload)) + return buffer.getvalue() + + +class TransportTests(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory() + self.addCleanup(self.temporary.cleanup) + root = Path(self.temporary.name) + execution = { + "schema_version": 1, + "point_id": "c" * 64, + "execution_id": "owned:55", + "bundle_digest": "b" * 64, + "source": { + "repository": "org/repo", + "run_id": 100, + "attempt": 1, + "head_sha": "a" * 40, + }, + "mode": "throughput", + "native_receipt": { + "job_id": "55", + "manifest_sha256": "f" * 64, + "state": "COMPLETED", + }, + "client_exit_code": 0, + } + with zipfile.ZipFile(root / "101.zip", "w") as archive: + archive.writestr( + "result.json", + json.dumps( + { + "conc": 1, + "tp": 8, + "ep": 1, + "num_gpus": 8, + "disagg": False, + "is_multinode": False, + "infmax_model_prefix": "dsr1", + "output_tput_tps": 100, + } + ), + ) + archive.writestr("execution.json", json.dumps(execution)) + self.source_archive_sha = hashlib.sha256( + (root / "101.zip").read_bytes() + ).hexdigest() + expected = ExpectedContract.model_validate( + { + "repository": "org/repo", + "source_run_id": "100", + "source_attempt": 1, + "source_head_sha": "a" * 40, + "bundle_digest": "b" * 64, + "contracts": { + "raw": "aiperf-1.4", + "normalized": "agentx-v1", + "publication": 1, + }, + "points": [ + { + "point_id": "c" * 64, + "bundle_digest": "b" * 64, + "execution_id": "owned:55", + "source_run_id": "100", + "source_attempt": 1, + "kind": "throughput", + "concurrency": 1, + "topology": { + "kind": "aggregate", + "nodes": 1, + "serving_gpus": 8, + "tp": 8, + "ep": 1, + }, + "artifact_ids": [101], + "execution_artifact_id": 101, + "execution_path": "execution.json", + "native_manifest_sha256": "f" * 64, + "normalized_artifact_id": 101, + "normalized_path": "result.json", + "required_metrics": ["output_tput_tps"], + "config": {"model": "dsr1"}, + } + ], + } + ) + source_meta = { + "id": 101, + "name": "native-execution-point", + "expired": False, + "workflow_run": {"id": 100}, + "digest": "sha256:" + self.source_archive_sha, + } + receipt = seal_receipt( + expected, + Issuer( + repository="org/repo", + run_id="200", + job="seal", + workflow_sha="d" * 40, + collector_sha="d" * 40, + ), + [source_meta], + root, + ) + self.receipt_id = receipt.receipt_id + self.archives = {301: archive_bytes("receipt.json", receipt.model_dump())} + receipt_zip_sha = hashlib.sha256(self.archives[301]).hexdigest() + publication = { + "kind": "publication-record", + "version": 1, + "receipt_id": receipt.receipt_id, + "receipt_artifact_id": 301, + "receipt_artifact_sha256": receipt_zip_sha, + "source_run_id": "100", + "merge_run_id": "150", + "merge_sha": "f" * 40, + "changelog_artifact_id": 501, + "changelog_artifact_sha256": "0" * 64, + "ingest_sha": "1" * 40, + "app_sha": "2" * 40, + } + self.archives[401] = archive_bytes("publication.json", publication) + self.metadata = { + 101: source_meta, + 301: { + "id": 301, + "name": "measurement-receipt", + "workflow_run": {"id": 200}, + "digest": "sha256:" + receipt_zip_sha, + "expired": False, + }, + 401: { + "id": 401, + "name": "publication-record", + "workflow_run": {"id": 300}, + "digest": "sha256:" + hashlib.sha256(self.archives[401]).hexdigest(), + "expired": False, + }, + 501: { + "id": 501, + "name": "changelog-metadata", + "workflow_run": {"id": 150}, + "digest": "sha256:" + "0" * 64, + "expired": False, + }, + } + self.runs = [ + { + "id": run, + "head_sha": sha * 40, + "status": "completed", + "conclusion": "success", + "path": ".github/workflows/phase1-receipt.yml", + "head_branch": "main", + "event": "workflow_dispatch", + } + for run, sha in [(200, "d"), (300, "e")] + ] + self.native = True + self.patchers = [ + patch("infx.workflows.receipt_transport.github.api", side_effect=self.api), + patch( + "infx.workflows.receipt_transport.github.paginate", + side_effect=self.pages, + ), + patch( + "infx.workflows.receipt_transport.subprocess.run", + side_effect=self.download, + ), + ] + for patcher in self.patchers: + patcher.start() + self.addCleanup(patcher.stop) + + def api(self, repo, endpoint, *args, **kwargs): + if endpoint.startswith("/actions/artifacts/"): + return self.metadata[int(endpoint.rsplit("/", 1)[-1])] + if endpoint == "/actions/runs/100/attempts/1": + return { + "id": 100, + "run_attempt": 1, + "head_sha": "a" * 40, + "status": "completed", + "conclusion": "success", + } + if endpoint == "/actions/runs/150": + return { + "id": 150, + "head_sha": "f" * 40, + "status": "completed", + "conclusion": "success", + "head_branch": "main", + "event": "push", + } + raise AssertionError(endpoint) + + def pages(self, repo, endpoint, *args, **kwargs): + if endpoint == "/actions/runs/100/artifacts": + return [self.metadata[101]] if self.native else [] + if endpoint.endswith("/phase1-receipt.yml/runs"): + return self.runs + if endpoint == "/actions/runs/200/artifacts": + return [self.metadata[301]] + if endpoint == "/actions/runs/300/artifacts": + return [self.metadata[401]] + raise AssertionError(endpoint) + + def download(self, args, *, stdout, check): + artifact = int(args[-1].split("/")[-2]) + stdout.write(self.archives[artifact]) + return subprocess.CompletedProcess(args, 0) + + def test_preserves_source_receipt_and_separate_later_publication(self): + payload = resolve_transport( + "org/repo", "100", "150", {"d" * 40, "e" * 40}, publication_required=True + ) + self.assertEqual(payload["receipt-artifact-id"], "301") + self.assertEqual(payload["receipt-issuer-run-id"], "200") + self.assertEqual(payload["publication-artifact-id"], "401") + self.assertEqual(payload["publication-issuer-run-id"], "300") + self.assertEqual(payload["receipt-required"], "true") + + def test_missing_required_receipt_never_enters_legacy(self): + self.runs = [] + with self.assertRaises(PendingReceiptError): + resolve_transport("org/repo", "100", "100", {"d" * 40}) + self.native = False + self.assertEqual( + resolve_transport("org/repo", "100", "100", set()), + {"receipt-required": "false"}, + ) + + def test_wrong_issuer_digest_and_missing_accepted_artifact_fail(self): + with self.assertRaises(PendingReceiptError): + resolve_transport("org/repo", "100", "100", {"9" * 40}) + original = self.metadata[301]["digest"] + self.metadata[301]["digest"] = "sha256:" + "0" * 64 + with self.assertRaisesRegex(ValueError, "API digest"): + resolve_transport("org/repo", "100", "100", {"d" * 40}) + self.metadata[301]["digest"] = original + self.metadata[101]["expired"] = True + with self.assertRaisesRegex(ValueError, "Accepted artifact set"): + resolve_transport("org/repo", "100", "100", {"d" * 40}) From 52deac0c03cdf54b073e058a469f420f1fe45a1f Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sat, 19 Sep 2026 19:24:14 -0400 Subject: [PATCH 02/16] feat: enable the prepared native H100 Phase 1 pilot MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 中文:在独立的可信回执前置能力之上启用 H100 聚合 TP8 原生试点,保留八个吞吐点和独立 c28 真实 eval,加入显式节点需求、预备式任务身份、原生提交恢复及所有权清理、双语部署说明和跨仓库行为验证。正式硬件与发布验收仍待 GitHub sweep 完成。 --- .github/workflows/benchmark-tmpl.yml | 51 +- .../srt-slurm/phase1/client-policy.json | 23 + .../srt-slurm/phase1/h100-dsv41flash.yaml | 54 ++ benchmarks/srt-slurm/phase1/runtime-lock.json | 13 + configs/nvidia-master.yaml | 7 + docs/configuration-procedures.md | 2 + docs/configuration-procedures_zh.md | 2 + docs/index.md | 1 + docs/index_zh.md | 1 + docs/srt-slurm-phase1.md | 115 ++++ docs/srt-slurm-phase1_zh.md | 115 ++++ infx/matrix/generate.py | 14 +- infx/matrix/validation.py | 21 + infx/srt_slurm/client_guard.py | 37 ++ infx/srt_slurm/job.py | 127 +++++ infx/srt_slurm/launch.py | 524 +++++++++++++++++ infx/srt_slurm/render.py | 258 +++++++++ infx/srt_slurm/workflow.py | 80 +++ infx/workflows/merge_source.py | 102 ++++ perf-changelog.yaml | 9 + runners/srt-slurm/h100-phase1.yaml | 11 + .../test_merge_with_reuse.py | 10 +- .../fixtures/native_pilot/client-policy.json | 23 + utils/fixtures/native_pilot/golden.yaml | 23 + utils/fixtures/native_pilot/profile.yaml | 11 + utils/fixtures/native_pilot/recipe.yaml | 54 ++ utils/fixtures/native_pilot_probe.py | 93 +++ utils/merge_with_reuse.sh | 31 +- utils/test_merge_source.py | 37 ++ utils/test_native_phase1_matrix_evals.py | 67 +++ utils/test_native_pilot.py | 535 ++++++++++++++++++ 31 files changed, 2417 insertions(+), 34 deletions(-) create mode 100644 benchmarks/srt-slurm/phase1/client-policy.json create mode 100644 benchmarks/srt-slurm/phase1/h100-dsv41flash.yaml create mode 100644 benchmarks/srt-slurm/phase1/runtime-lock.json create mode 100644 docs/srt-slurm-phase1.md create mode 100644 docs/srt-slurm-phase1_zh.md create mode 100644 infx/srt_slurm/client_guard.py create mode 100644 infx/srt_slurm/job.py create mode 100644 infx/srt_slurm/launch.py create mode 100644 infx/srt_slurm/render.py create mode 100644 infx/srt_slurm/workflow.py create mode 100644 infx/workflows/merge_source.py create mode 100644 runners/srt-slurm/h100-phase1.yaml create mode 100644 utils/fixtures/native_pilot/client-policy.json create mode 100644 utils/fixtures/native_pilot/golden.yaml create mode 100644 utils/fixtures/native_pilot/profile.yaml create mode 100644 utils/fixtures/native_pilot/recipe.yaml create mode 100644 utils/fixtures/native_pilot_probe.py create mode 100644 utils/test_merge_source.py create mode 100644 utils/test_native_phase1_matrix_evals.py create mode 100644 utils/test_native_pilot.py diff --git a/.github/workflows/benchmark-tmpl.yml b/.github/workflows/benchmark-tmpl.yml index 11b14aadec..f33d79b52e 100644 --- a/.github/workflows/benchmark-tmpl.yml +++ b/.github/workflows/benchmark-tmpl.yml @@ -165,7 +165,7 @@ jobs: ${{ fromJSON( vars.PRIORITY_SCHEDULER_ENABLED == 'true' && ( - vars.NODE_SLOT_SCHEDULER_ENABLED == 'true' && + (vars.NODE_SLOT_SCHEDULER_ENABLED == 'true' || fromJSON(inputs.config).execution.runtime == 'srt-slurm') && ( inputs.skip-queue-pr != '' && format( @@ -220,6 +220,7 @@ jobs: fi - name: Resource cleanup (pre-run) + if: ${{ fromJSON(inputs.config).execution.runtime != 'srt-slurm' }} run: &resource-cleanup | # Cleanup Docker resources if command -v docker >/dev/null 2>&1 && docker info >/dev/null 2>&1; then @@ -243,6 +244,7 @@ jobs: fi - name: Repair stale git state (pre-run) + if: ${{ fromJSON(inputs.config).execution.runtime != 'srt-slurm' }} run: | # Interrupted runs leave index.lock files and corrupt submodule git # dirs behind in the persistent self-hosted workspace, and both make @@ -275,7 +277,45 @@ jobs: submodules: true persist-credentials: false + - name: Set up uv for native pilot preparation + if: ${{ fromJSON(inputs.config).execution.runtime == 'srt-slurm' }} + uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1 + + - name: Launch prepared native pilot + if: ${{ fromJSON(inputs.config).execution.runtime == 'srt-slurm' }} + env: + # Prepared native clients and serving assets are offline; retain no legacy download/scoring credentials. + HF_TOKEN: '' + MODAL_TOKEN_ID: '' + MODAL_TOKEN_SECRET: '' + NATIVE_CONFIG_JSON: ${{ inputs.config }} + NATIVE_PRIORITY: ${{ inputs.priority }} + NATIVE_QUEUE_TOKEN: ${{ inputs.queue-token }} + NATIVE_RUN_EVAL: ${{ toJSON(inputs.run-eval) }} + NATIVE_EVAL_ONLY: ${{ toJSON(inputs.eval-only) }} + NATIVE_EVAL_FRAMEWORK: ${{ inputs.eval-framework }} + NATIVE_EVAL_SUITE: ${{ inputs.eval-suite }} + NATIVE_AGENTX_FAST: ${{ toJSON(inputs.agentx-fast) }} + NATIVE_EVAL_LIMIT: ${{ inputs.eval-limit }} + NATIVE_REQUIRE_POWER: ${{ toJSON(inputs.require-power) }} + NATIVE_SITE_JSON: ${{ vars.INFX_H100_PHASE1_SITE_JSON }} + NATIVE_READER_REVISION: ${{ vars.INFX_PHASE1_READER_REVISION }} + NATIVE_COLLECTOR_REVISION: ${{ vars.INFX_PHASE1_COLLECTOR_REVISION }} + shell: python + run: | + import subprocess + subprocess.run(['uv', 'run', '--locked', '--python', '3.12', 'python', '-m', 'infx.srt_slurm.workflow'], check=True) + + - name: Upload native execution provenance + if: ${{ always() && fromJSON(inputs.config).execution.runtime == 'srt-slurm' && env.NATIVE_POINT_ID != '' }} + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: native-execution-${{ env.NATIVE_POINT_ID }} + path: native-execution/** + if-no-files-found: error + - name: Launch job script + if: ${{ fromJSON(inputs.config).execution.runtime != 'srt-slurm' }} env: RUNNER_NAME: ${{ runner.name }} RUNNER_TYPE: ${{ inputs.runner }} @@ -383,6 +423,9 @@ jobs: server.log results/*.log results/*_config.json + results/native/**/*.log + results/native/**/*.json + results/native/**/*.yaml if-no-files-found: ignore - name: Upload GPU metrics @@ -444,11 +487,11 @@ jobs: if-no-files-found: ${{ inputs.eval-only && 'error' || 'ignore' }} - name: Verify eval scores - if: ${{ (success() || failure()) && inputs.eval-only }} + if: ${{ (success() || failure()) && inputs.eval-only && fromJSON(inputs.config).execution.runtime != 'srt-slurm' }} run: python3 utils/evals/validate_scores.py - name: Cleanup eval outputs (post-upload) - if: ${{ always() && (env.RUN_EVAL == 'true' || inputs.eval-only) }} + if: ${{ always() && (env.RUN_EVAL == 'true' || inputs.eval-only) && fromJSON(inputs.config).execution.runtime != 'srt-slurm' }} run: | rm -f meta_env.json || true # Remove any eval results JSONs that were moved into workspace @@ -467,5 +510,5 @@ jobs: fi - name: Resource cleanup (post-run) - if: always() + if: ${{ always() && fromJSON(inputs.config).execution.runtime != 'srt-slurm' }} run: *resource-cleanup diff --git a/benchmarks/srt-slurm/phase1/client-policy.json b/benchmarks/srt-slurm/phase1/client-policy.json new file mode 100644 index 0000000000..a8d42281b2 --- /dev/null +++ b/benchmarks/srt-slurm/phase1/client-policy.json @@ -0,0 +1,23 @@ +{ + "schema_version": 1, + "golden_curve": "golden_al_distribution/dsv41flash_dspark.yaml", + "golden_model": "deepseek-v4.1-flash", + "thinking_mode": "thinking_on", + "dataset_repository": "semianalysisai/cc-traces-weka-062126", + "dataset_loader": "semianalysis_cc_traces_weka_062126", + "dataset_entries": 393, + "duration_seconds": 3600, + "warmup_requests_per_lane": 10, + "warmup_grace_seconds": 1800, + "trace_idle_gap_cap_seconds": 300, + "live_failed_request_threshold": 0.1, + "failed_request_threshold": 0.1, + "random_seed": 42, + "required_server_metric_prefix": "vllm:", + "eval_task": "gsm8k", + "eval_documents": 1319, + "eval_max_length": 16384, + "eval_max_tokens": 12288, + "eval_minimum_score": 0.9, + "telemetry": "temporary-parity-exception-no-native-power" +} diff --git a/benchmarks/srt-slurm/phase1/h100-dsv41flash.yaml b/benchmarks/srt-slurm/phase1/h100-dsv41flash.yaml new file mode 100644 index 0000000000..bf431b2c71 --- /dev/null +++ b/benchmarks/srt-slurm/phase1/h100-dsv41flash.yaml @@ -0,0 +1,54 @@ +schema: 2 +name: inferencex-h100-dsv41flash-agentx +slurm: + account: customer + partition: hpc-gpu-1 + time_limit: '08:00:00' +model: + path: deepseek-ai/DeepSeek-V4.1-Flash + container: vllm/vllm-openai:nightly-cd10ed6f9f6b37a8ace9cf380007e66fe12ec0c3 + precision: fp4 +resources: + gpu_type: h100 + gpus_per_node: 8 +engine: vllm +frontend: + type: vllm + enable_multiple_frontends: false +roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + env: + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_USE_RUST_FRONTEND: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + PYTHONUNBUFFERED: '1' + PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True + HF_HUB_OFFLINE: '1' + HF_DATASETS_OFFLINE: '1' + args: + served-model-name: deepseek-ai/DeepSeek-V4.1-Flash + tensor-parallel-size: 8 + language-model-only: true + tokenizer-mode: deepseek_v41 + tool-call-parser: deepseek_v41 + enable-auto-tool-choice: true + reasoning-parser: deepseek_v41 + engram-config: {cpu_offload: true} + speculative-config: + method: dspark + num_speculative_tokens: 5 + draft_sample_method: probabilistic + rejection_sample_method: block + enable_adaptive_verification: true + max-model-len: 1048576 + max-num-batched-tokens: 4096 + gpu-memory-utilization: 0.92 + disable-uvicorn-access-log: true +health_check: + max_attempts: 360 + interval_seconds: 10 +benchmark: + type: custom diff --git a/benchmarks/srt-slurm/phase1/runtime-lock.json b/benchmarks/srt-slurm/phase1/runtime-lock.json new file mode 100644 index 0000000000..9536c7efd0 --- /dev/null +++ b/benchmarks/srt-slurm/phase1/runtime-lock.json @@ -0,0 +1,13 @@ +{ + "schema_version": 1, + "repository": "https://github.com/SemiAnalysisAI/srt-slurm.git", + "revision": "8e459d10d217cec7636a4d257e1e0f69d2a45a81", + "uv_lock_sha256": "f7c3ef25605ebe27bac9c6a54aa2acef7332210321fff918c3fa72b7049d4dea", + "capabilities": [ + "prepared-v1", + "durable-intent-v1", + "custom-argv-v1", + "controller-observation-v1", + "bounded-cleanup-v1" + ] +} diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index 06e8c65fc1..a61f7baced 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -8056,6 +8056,13 @@ dsv41flash-fp4-h100-vllm-agentic-dspark: precision: fp4 framework: vllm multinode: false + execution: + runtime: srt-slurm + contract-version: 1 + recipe: benchmarks/srt-slurm/phase1/h100-dsv41flash.yaml + profile: runners/srt-slurm/h100-phase1.yaml + runtime-lock: benchmarks/srt-slurm/phase1/runtime-lock.json + client-policy: benchmarks/srt-slurm/phase1/client-policy.json scenarios: agentic-coding: - dram-utilization: 0.80 diff --git a/docs/configuration-procedures.md b/docs/configuration-procedures.md index 4c6abfccfa..f9a7c4eba1 100644 --- a/docs/configuration-procedures.md +++ b/docs/configuration-procedures.md @@ -8,6 +8,8 @@ Use this page for benchmark configuration, recipe, image, and runner changes. It is a procedure, not a field catalog: the linked implementation and schema remain authoritative. +The Phase 1 H100 aggregate pilot uses an explicit versioned `execution` reference and an isolated native runtime pin. See [Phase 1](./srt-slurm-phase1.md) for preparation, same-path mounts, the reader-first release gate and rollback. Its legacy script remains until hardware qualification. + ## Source map | Source of truth | What it controls | diff --git a/docs/configuration-procedures_zh.md b/docs/configuration-procedures_zh.md index f877035f18..4cabe2ae83 100644 --- a/docs/configuration-procedures_zh.md +++ b/docs/configuration-procedures_zh.md @@ -8,6 +8,8 @@ 本页用于基准配置、配方、镜像和 runner 变更。它是操作规程而非字段目录;所链接的实现和 schema 始终是权威来源。 +阶段 1 H100 聚合试点使用明确、带版本的 `execution` 引用及独立原生运行时 pin。准备流程、同路径挂载、reader 优先发布 gate 与回退见[阶段 1](./srt-slurm-phase1_zh.md)。旧脚本保留至硬件验收完成。 + ## 权威来源图 | 权威来源 | 控制内容 | diff --git a/docs/index.md b/docs/index.md index c0014bc903..4e042af21e 100644 --- a/docs/index.md +++ b/docs/index.md @@ -16,6 +16,7 @@ This is the mandatory low-context router for InferenceX work. Pick the one page | [`agent-guide.md`](./agent-guide.md) / [`agent-guide_zh.md`](./agent-guide_zh.md) | Agent onboarding, safe start, invariants, and verification | | [`procedures.md`](./procedures.md) / [`procedures_zh.md`](./procedures_zh.md) | Routing from a recurring task to one focused operational checklist | | [`architecture.md`](./architecture.md) / [`architecture_zh.md`](./architecture_zh.md) | Config-to-result flow, ownership boundaries, artifacts, and InferenceX-app handoff | +| [`srt-slurm-phase1.md`](./srt-slurm-phase1.md) / [`srt-slurm-phase1_zh.md`](./srt-slurm-phase1_zh.md) | Phase 1 native H100 execution: code ownership, provisioning, recovery, receipt publication and qualification ledger | | [`configuration-procedures.md`](./configuration-procedures.md) / [`configuration-procedures_zh.md`](./configuration-procedures_zh.md) | Config, runner, image, recipe, llm-d, srt-slurm, and MTP changes | | [`ci-procedures.md`](./ci-procedures.md) / [`ci-procedures_zh.md`](./ci-procedures_zh.md) | Matrix generation, validation, dispatch, PR sweeps, reuse, staging, and artifact downloads | | [`eval-agentx-procedures.md`](./eval-agentx-procedures.md) / [`eval-agentx-procedures_zh.md`](./eval-agentx-procedures_zh.md) | Eval and AgentX selection, execution, scoring, evidence, and live-run diagnosis | diff --git a/docs/index_zh.md b/docs/index_zh.md index 91a2bb705c..69d7739859 100644 --- a/docs/index_zh.md +++ b/docs/index_zh.md @@ -16,6 +16,7 @@ | [`agent-guide.md`](./agent-guide.md) / [`agent-guide_zh.md`](./agent-guide_zh.md) | Agent 入门、安全开始、关键约束与验证 | | [`procedures.md`](./procedures.md) / [`procedures_zh.md`](./procedures_zh.md) | 从常见任务路由到一份聚焦运维清单 | | [`architecture.md`](./architecture.md) / [`architecture_zh.md`](./architecture_zh.md) | 配置到结果的流程、所有权边界、产物与 InferenceX-app 交接 | +| [`srt-slurm-phase1.md`](./srt-slurm-phase1.md) / [`srt-slurm-phase1_zh.md`](./srt-slurm-phase1_zh.md) | 阶段 1 原生 H100 执行:代码职责、部署准备、恢复、回执发布及验收账本 | | [`configuration-procedures.md`](./configuration-procedures.md) / [`configuration-procedures_zh.md`](./configuration-procedures_zh.md) | 配置、Runner、镜像、Recipe、llm-d、srt-slurm 与 MTP 变更 | | [`ci-procedures.md`](./ci-procedures.md) / [`ci-procedures_zh.md`](./ci-procedures_zh.md) | 矩阵生成、校验、派发、PR 扫描、复用、暂存与产物下载 | | [`eval-agentx-procedures.md`](./eval-agentx-procedures.md) / [`eval-agentx-procedures_zh.md`](./eval-agentx-procedures_zh.md) | Eval 与 AgentX 选择、执行、打分、证据与实时运行诊断 | diff --git a/docs/srt-slurm-phase1.md b/docs/srt-slurm-phase1.md new file mode 100644 index 0000000000..4076e3c184 --- /dev/null +++ b/docs/srt-slurm-phase1.md @@ -0,0 +1,115 @@ +# Phase 1: prepared H100 aggregate execution + +**English** | [中文](./srt-slurm-phase1_zh.md) + +Phase 1 implements the first native srt-slurm lane. Hardware qualification, reader deployment and publication remain open. Passing local tests does not close this phase. The approved migration plan’s Phase 1 acceptance contract is restated below. The full plan and its research archive remain in the separate planning worktree. + +## Scope and ownership + +Only `dsv41flash-fp4-h100-vllm-agentic-dspark` uses the new `execution` reference. One exclusive H100 node runs one direct vLLM TP8 worker. Throughput uses c1,2,4,8,16,20,24,28; a separate c28 GSM8K job uses real verification. `--all-evals` also selects only c28 for this version-one native contract. The existing image, 1,048,576 context, 4096 batched tokens, five-token DSpark, 480-minute allocation and golden AL resource are preserved. Power telemetry is an explicit temporary parity exception; `require-power` is rejected. + +The shared `utils/srt-slurm` gitlink remains unchanged. The pilot's separate `runtime-lock.json` pins its native source and dependency lock. NVIDIA and AMD reference checkouts remain clean. Native srt-slurm Bash templates, wrappers and setup scripts remain supported dependency code. Phase 1 does not port AMD, MoRI or ATOM. + +```mermaid +flowchart LR + M[Master execution reference] --> Q[Typed matrix and scheduling envelope] + Q --> P[Prepare installed clients and offline assets] + P --> B[Freeze recipe, profile, client and identities] + B --> N[Native prepare: allocation and cardinality] + N --> J[Native durable intent journal] + J --> S[One exclusive Slurm allocation] + S --> V[Direct vLLM TP8] + V --> C[Python AgentX or real eval] + C --> A[Raw and normalized artifacts] + A --> R[Trusted source receipt] + R --> I[App validated import] + R --> U[Later publication record] + U --> I +``` + +## Call and file map + +```mermaid +flowchart TD + W[benchmark-tmpl.yml / native step] --> F[infx.srt_slurm.workflow.main] + F --> J[infx.srt_slurm.job.parse_job] + F --> P[infx.srt_slurm.launch.prepare] + P --> CP[infx.benchmarks.prepare.prepare] + P --> R[infx.srt_slurm.render.render_recipe] + R --> Y[benchmarks/srt-slurm/phase1/h100-dsv41flash.yaml] + R --> H[runners/srt-slurm/h100-phase1.yaml] + P --> NP[srtctl prepare] + F --> X[infx.srt_slurm.launch.execute] + X --> NI[srtctl intent-path / submit-prepared] + X --> NW[srtctl wait / reconcile / cancel-known / wait-known] + NI --> G[infx.srt_slurm.client_guard.main] + G --> AX[infx.benchmarks.agentx.run] + G --> EV[infx.benchmarks.eval.run] + AX --> AIP[Pinned Python 3.11 AIPerf child] + EV --> LM[Pinned lm-eval child] + X --> O[Closed output staging and failure diagnostics] +``` + +`ExecutionReference` binds recipe, profile, runtime lock, client policy and the policy's golden YAML bytes. Changed inputs, duplicate YAML keys, unsupported scope or missing explicit queue demand fail before allocation. `priority` and `queue-token` are scheduling metadata; they do not change the requested semantic point. + +Preparation records the actual installed native/wrapper/client files, interpreter identities, plugin resolution, asset-path/content bindings, immutable model/dataset revisions and image bytes. The installed wrapper must match the selected checkout and must not be editable. `requested_point_id` identifies the requested row; `point_id` additionally binds these resolved identities. `bundle_digest` binds the full executable snapshot. `execution_id` identifies one repository/run/attempt/requested-point intent. Existing `recipe_fingerprint` remains the compatible matrix family label; native publication requires the stronger receipt identities. + +## Provisioning before the first GPU run + +Provision on shared Linux storage visible to the H100 login host and compute container. Do not reuse the macOS test environments. The native, wrapper and selected client interpreters, their standard libraries, installed distributions, native source, prepared bundles and client caches need explicit same-path mounts. Mount roots must be canonical paths; symlink aliases are rejected. Python 3.12 is required for native/wrapper execution and Python 3.11 for the pinned AgentX child. Outputs and writable caches must stay outside `/workspace`. The model's HF snapshot must retain access to its sibling blob directory. The native model argument preserves that full-cache mount. + +1. Install the pinned native source using its committed `uv.lock` and a noneditable environment (`uv sync --frozen --no-editable --no-dev --python 3.12`). Keep that source checkout clean. Preserve the hashed Linux-built wheel and its build-tool constraints: `uv.lock` freezes runtime dependencies but does not pin the upstream Hatch build dependencies. If rebuilding, fetch and verify NVIDIA’s `v2.2.1` tag at `984180e5b8755aef85e9995048b5a16cb5336bce` to retain the same hatch-vcs version lineage. +2. Install a noneditable InferenceX wheel from the exact measured checkout into a shared Python 3.12 environment. Its installed package bytes are compared against the checkout before allocation. +3. Materialize separate client environments and retain their resolved package artifacts/locks. AgentX must come from `754356e9a39acc6cc6afb242d123bb57c3fb6f75`; lm-eval must come from `b315ef3b05176acc9732bb7fdec116abe1ecc476`. Editable and wrong-source installations are rejected. Preparation captures every installed distribution, not just the named entry point. +4. Materialize the complete model/tokenizer snapshot, exact `semianalysisai/cc-traces-weka-062126` snapshot and GSM8K cache. Capture their real revisions; do not invent or substitute a revision. Prepare the unchanged serving image as a verified squash file and record its provenance/hash. +5. Write one `ClientSite` JSON for AgentX and one for eval. These explicitly provide the interpreter, distributions, offline cache environment, environment removals, asset roots/files, model snapshot, timeout and termination grace. `RuntimeSpec` rejects credentials; execution strips ambient credentials and unqualified AIPerf overrides. The packaged task and 1,319 independent document hashes are included in installed wheels. +6. Write the `PilotSite` JSON with these two client-site paths, source/interpreter/model/image paths, mounts and actual deployed reader/collector revisions. The Pydantic models in [`render.py`](../infx/srt_slurm/render.py) and [`prepare.py`](../infx/benchmarks/prepare.py) are the exact schemas. + +Preparation validates existing assets; it does not install packages, download models or repair incomplete snapshots on compute nodes. The derived mmap cache uses an owned namespace, file-integrity receipts, independent verified copies and corruption quarantine. Cold preparation on lock contention is explicit and bounded. + +Before enabling sweeps, deploy the app reader and migration `016_measurement_snapshots.sql`, then land/deploy the trusted collector. Configure `INFX_H100_PHASE1_SITE_JSON`, `INFX_PHASE1_READER_REVISION` and `INFX_PHASE1_COLLECTOR_REVISION` in InferenceX. Configure `INFX_RECEIPT_ISSUER_SHAS` and `INFX_RECEIPT_ISSUER_WORKFLOW` in both repositories; the workflow is `.github/workflows/phase1-receipt.yml`. These values are absent in the inspected repository configuration. A source branch containing the code alone is not a deployed reader. + +## Preparation, execution and recovery + +The standalone adapter accepts explicit files: + +```text +python -m infx.srt_slurm.launch --job job.json --site site.json --root CHECKOUT --source source.json --prepare-only +python -m infx.srt_slurm.launch --job job.json --site site.json --root CHECKOUT --source source.json +``` + +`job.json` contains the generated row plus explicit `priority`, `queue-token` and `node-count: 1`. `source.json` contains `repository`, numeric `run_id`, numeric `attempt` and the full measured `head_sha`. Do not manufacture GitHub run identity. `--prepare-only` performs no allocation. Independently inspect and retain these prepared expectations before running the corresponding points; a worker's later `execution.json` is evidence to compare, not authority for expected identities. + +Native preparation must resolve exactly `{nodes:1,gpus_per_node:8,serving_gpus:8,workers:1,cardinality:1}`. Throughput renders synthetic rejection from the committed golden curve with adaptive verification off. Eval renders real block rejection with adaptive verification on. Direct port 8000 is an explicit exclusive-node policy: a bind collision is a failure, not permission to contact another server. + +Slurm allocation, claims, accepted IDs, scheduler observation and cancellation belong to the native runtime. The adapter obtains the journal path before the interruptible submit. It never repeats an ambiguous submission or cancels by runner name. Active controller state takes precedence over stale accounting; a failed-but-active requeue is not closed. Known owned allocations are cancelled and observed to terminal closure with bounded waits. An unresolved intent stays fenced for inspection. + +Successful publication requires both native terminal success and a closed client audit with no error, timeout, signal or orphaned writer. Failure diagnostics retain raw outputs, client audit, frozen inputs and native logs without producing an accepted execution manifest. The broad legacy pre/post runner cleanup is skipped for this lane. The legacy H100 launcher/script remains available for rollback until qualification and a reviewed retirement diff. + +## Measurement receipt and publication + +The complete source contract is eight throughput points and one real c28 eval. GSM8K requires all 1,319 documents and both filters (2,638 scored rows), the preserved 16,384 context / 12,288 generation budgets, finite scores and complete sample identities. Aggregate eval metadata has `disagg:false`, `is_multinode:false`, eight serving GPUs and zero prefill/decode worker counts. + +1. Create a reviewed `qualification/phase1/*.json` expectation using the `Approval` schema in [`phase1_publication.py`](../infx/workflows/phase1_publication.py). Copy point/execution/bundle/native-manifest identities from the independently prepared control records, not worker archives. Require the complete nine-point set and actual corpus revision. +2. Run `phase1-receipt.yml` on `main` with `kind: measurement`. Trusted code resolves exact source artifact IDs, verifies API ownership/run/attempt, ZIP digest and safe members, then validates execution identity, normalized metrics/config/topology/dataset and raw eval coverage before sealing `receipt.json`. +3. Staging resolves the accepted receipt through the deployed issuer allowlist. Missing native receipts fail closed. The app verifies the snapshot before database writes or a staging reset; partial import resumes only the same immutable source receipt. +4. After the reviewed merge/publication run completes, approve a `PublicationRecord` JSON linking the original receipt artifact/digest, merge SHA/run, changelog artifact/digest and deployed app/ingest revision. Run the same issuer with `kind: publication`. The original source receipt is not rewritten. +5. Use the supported staging/recovery dispatch. Automatic main ingest defers while required source/publication sealing is pending; it never falls back to legacy native ingestion. Recovery carries exact receipt and publication references. App checks include exact-run and latest curves, trace detail, aggregate topology and strict-filter eval visibility. + +The merge helper preserves the latest explicit authorized `/use RUN_ID` (or `/reuse-sweep-run RUN_ID`). A newer diagnostic run cannot silently replace it. Unavailable authorized evidence requires an explicit new decision. + +## Qualification ledger + +| Gate | Status / required evidence | +| --- | --- | +| Native and client behavior | CPU tests and installed-wheel checks; no GPU claim | +| Receipt, app and recovery | Local unit, database and browser smoke checks; deployment pending | +| H100 throughput | c1,2,4,8,16,20,24,28 pending on the unchanged image | +| Real evaluation | New c28 run pending; historical full raw eval passes validator | +| Cancellation/cleanup | Local ownership/race/closure tests; real Slurm signal qualification pending | +| Measurement equivalence | Compare qualified metrics, failures, warmup/drain, server settings and raw schemas against the retained baseline | +| Publication | Trusted receipt, later publication record and refreshed app evidence pending | +| Power | Explicit temporary parity exception; measured power not claimed | +| Retirement | Legacy H100 script retained pending all exit evidence | + +Record actual InferenceX/native/collector/app commits, source run/attempt, prepared expectation, nine artifact bindings, source receipt, publication record and app verification report in this ledger when available. No fabricated IDs or placeholder success entries may close a gate. diff --git a/docs/srt-slurm-phase1_zh.md b/docs/srt-slurm-phase1_zh.md new file mode 100644 index 0000000000..072ca0d729 --- /dev/null +++ b/docs/srt-slurm-phase1_zh.md @@ -0,0 +1,115 @@ +# 阶段 1:预备式 H100 聚合执行 + +[English](./srt-slurm-phase1.md) | **中文** + +阶段 1 实现首条原生 srt-slurm 路径。硬件验收、reader 部署与发布仍待完成;本地测试通过不代表阶段结束。下文重述已批准迁移计划中阶段 1 的验收要求。完整计划及研究资料仍保留在独立的规划 worktree。 + +## 范围与职责 + +只有 `dsv41flash-fp4-h100-vllm-agentic-dspark` 使用新的 `execution` 引用。一台独占 H100 节点运行一个直连 vLLM TP8 worker。吞吐并发为 c1、2、4、8、16、20、24、28;另有 c28 GSM8K 真实验证任务。`--all-evals` 对该版本 1 原生契约也仅选择 c28。保留现有镜像、1,048,576 上下文、4096 batched tokens、五 token DSpark、480 分钟分配以及 golden AL 资源。功耗遥测是明确的临时一致性例外;拒绝 `require-power`。 + +共享 `utils/srt-slurm` gitlink 不变。试点通过独立 `runtime-lock.json` 固定原生源代码与依赖锁。NVIDIA、AMD 参考仓库保持干净。原生 srt-slurm 的 Bash 模板、包装器及安装脚本仍属于允许保留的依赖代码。本阶段不迁移 AMD、MoRI 或 ATOM。 + +```mermaid +flowchart LR + M[Master execution 引用] --> Q[类型化矩阵与调度信封] + Q --> P[准备已安装客户端与离线资源] + P --> B[冻结配方、profile、客户端与身份] + B --> N[原生 prepare:分配与基数] + N --> J[原生持久化 intent 日志] + J --> S[一次独占 Slurm 分配] + S --> V[直连 vLLM TP8] + V --> C[Python AgentX 或真实 eval] + C --> A[原始及规范化产物] + A --> R[受信任源测量回执] + R --> I[App 校验后导入] + R --> U[后续发布记录] + U --> I +``` + +## 调用与文件关系 + +```mermaid +flowchart TD + W[benchmark-tmpl.yml / native step] --> F[infx.srt_slurm.workflow.main] + F --> J[infx.srt_slurm.job.parse_job] + F --> P[infx.srt_slurm.launch.prepare] + P --> CP[infx.benchmarks.prepare.prepare] + P --> R[infx.srt_slurm.render.render_recipe] + R --> Y[benchmarks/srt-slurm/phase1/h100-dsv41flash.yaml] + R --> H[runners/srt-slurm/h100-phase1.yaml] + P --> NP[srtctl prepare] + F --> X[infx.srt_slurm.launch.execute] + X --> NI[srtctl intent-path / submit-prepared] + X --> NW[srtctl wait / reconcile / cancel-known / wait-known] + NI --> G[infx.srt_slurm.client_guard.main] + G --> AX[infx.benchmarks.agentx.run] + G --> EV[infx.benchmarks.eval.run] + AX --> AIP[固定 Python 3.11 AIPerf 子进程] + EV --> LM[固定 lm-eval 子进程] + X --> O[写入者关闭后的产物整理与失败诊断] +``` + +`ExecutionReference` 绑定配方、profile、runtime lock、client policy 以及 policy 指向的 golden YAML 字节。输入发生变化、YAML 重复键、超出范围或缺少明确排队节点需求,均在分配前失败。`priority` 与 `queue-token` 仅属于调度信息,不改变请求测量点。 + +准备阶段记录实际安装的原生运行时、wrapper、客户端文件,解释器身份、插件解析、资源路径与内容绑定、不可变模型/数据集 revision、镜像字节。已安装 wrapper 必须与所选 checkout 一致,且不可为 editable。`requested_point_id` 标识请求行;`point_id` 进一步绑定解析后的实际身份;`bundle_digest` 绑定完整可执行快照;`execution_id` 标识一次仓库/run/attempt/请求点 intent。现有 `recipe_fingerprint` 继续用于兼容矩阵 family;原生发布必须验证更强的回执身份。 + +## 首次 GPU 运行前的部署 + +在 H100 登录节点与计算容器均可访问的共享 Linux 存储上部署,不复用 macOS 测试环境。原生、wrapper 及所选客户端的 Python 解释器、标准库、已安装依赖、原生源代码、prepared bundle 与客户端缓存均需明确的同路径挂载。挂载根目录必须是规范路径,不接受符号链接别名。原生及 wrapper 使用 Python 3.12,固定的 AgentX 子进程使用 Python 3.11。输出与可写缓存不得放在 `/workspace` 下。HF 模型 snapshot 必须仍能访问相邻 blob 目录;原生模型参数会保留完整缓存路径。 + +1. 使用原生源代码提交中的 `uv.lock` 安装非 editable 环境:`uv sync --frozen --no-editable --no-dev --python 3.12`。保持该源码 checkout 干净。保留带哈希的 Linux wheel 及构建工具约束:`uv.lock` 固定运行依赖,但未固定上游 Hatch 构建依赖。重新构建时,获取并核实 NVIDIA 的 `v2.2.1` tag 指向 `984180e5b8755aef85e9995048b5a16cb5336bce`,保留相同 hatch-vcs 版本谱系。 +2. 从实际测量 checkout 构建并安装非 editable InferenceX wheel,使用共享 Python 3.12 环境。分配前逐文件比较已安装包与 checkout。 +3. 准备独立客户端环境并保留实际解析的包产物与锁。AgentX 必须来自 `754356e9a39acc6cc6afb242d123bb57c3fb6f75`;lm-eval 必须来自 `b315ef3b05176acc9732bb7fdec116abe1ecc476`。拒绝 editable 或错误来源。准备阶段记录所有已安装 distribution,而非仅入口包。 +4. 完整准备模型/tokenizer snapshot、准确的 `semianalysisai/cc-traces-weka-062126` snapshot 与 GSM8K 缓存。记录真实 revision,不编造或替换。准备原样服务镜像的已校验 squash 文件,记录来源和哈希。 +5. 分别编写 AgentX、eval 的 `ClientSite` JSON:解释器、distribution、离线缓存环境、移除变量、资源根目录/文件、模型 snapshot、超时与终止宽限。`RuntimeSpec` 拒绝凭证;执行时移除继承凭证及未验收的 AIPerf 覆盖项。wheel 包含 eval task 与 1,319 个独立文档哈希。 +6. 编写 `PilotSite` JSON,包含两个客户端配置路径、源码/解释器/模型/镜像路径、挂载以及实际部署的 reader/collector revision。准确 schema 见 [`render.py`](../infx/srt_slurm/render.py) 与 [`prepare.py`](../infx/benchmarks/prepare.py)。 + +准备阶段校验已有资源,不在计算节点安装包、下载模型或修复不完整 snapshot。派生 mmap 缓存使用独立所属 namespace、文件完整性回执、独立校验副本及损坏隔离;锁竞争时的冷准备有明确界限。 + +先部署 app reader 与 `016_measurement_snapshots.sql` migration,再合入/部署受信任 collector。InferenceX 需配置 `INFX_H100_PHASE1_SITE_JSON`、`INFX_PHASE1_READER_REVISION`、`INFX_PHASE1_COLLECTOR_REVISION`。两个仓库均需配置 `INFX_RECEIPT_ISSUER_SHAS`、`INFX_RECEIPT_ISSUER_WORKFLOW`,workflow 路径为 `.github/workflows/phase1-receipt.yml`。检查时这些变量尚不存在。分支中有代码不等于 reader 已部署。 + +## 准备、执行与恢复 + +独立适配器接收明确的文件参数: + +```text +python -m infx.srt_slurm.launch --job job.json --site site.json --root CHECKOUT --source source.json --prepare-only +python -m infx.srt_slurm.launch --job job.json --site site.json --root CHECKOUT --source source.json +``` + +`job.json` 包含生成的矩阵行及明确的 `priority`、`queue-token`、`node-count: 1`。`source.json` 包含 `repository`、数值 `run_id`、数值 `attempt`、完整测量 `head_sha`。不得编造 GitHub run 身份。`--prepare-only` 不申请资源。运行对应测量点前,独立检查并保留准备期预期;worker 后来的 `execution.json` 只用于对照,不能成为预期身份的权威来源。 + +原生准备必须解析为 `{nodes:1,gpus_per_node:8,serving_gpus:8,workers:1,cardinality:1}`。吞吐从已提交 golden 曲线渲染 synthetic rejection,关闭 adaptive verification;eval 使用真实 block rejection,开启 adaptive verification。直连端口 8000 采用明确的独占节点策略:端口冲突即失败,不能连接其他服务器。 + +Slurm 分配、claim、已接受 ID、调度器观察与取消均由原生运行时负责。适配器在可中断 submit 前取得日志路径,不重试不明确的提交,也不按 runner 名批量取消。活跃 controller 状态优先于陈旧 accounting;失败但仍活跃的 requeue 不算关闭。取消所有已确认属于该 intent 的资源,并在有界等待内观察终止;未解决的 intent 保持 fenced,等待检查。 + +成功发布要求原生终止成功及客户端写入者关闭:无错误、超时、信号或孤儿 writer。失败诊断保留原始输出、client audit、冻结输入及原生日志,但不生成已接受 execution manifest。该路径跳过旧 runner 的宽泛前后清理。H100 旧 launcher/script 保留至硬件验收及退休差异审阅完成,以便回退。 + +## 测量回执与发布 + +完整源契约为八个吞吐点加一个真实 c28 eval。GSM8K 必须包含全部 1,319 文档及两种 filter(2,638 个评分行),保留 16,384 上下文 / 12,288 生成预算,验证有限分数和完整样本身份。聚合 eval 元数据为 `disagg:false`、`is_multinode:false`、八个服务 GPU、prefill/decode worker 数均为零。 + +1. 按 [`phase1_publication.py`](../infx/workflows/phase1_publication.py) 的 `Approval` schema 创建经审阅的 `qualification/phase1/*.json`。点、执行、bundle、原生 manifest 身份必须来自独立准备期控制记录,不得从 worker archive 推导。要求完整九点集合及实际数据集 revision。 +2. 在 `main` 运行 `phase1-receipt.yml`,`kind: measurement`。受信任代码解析准确 artifact ID,校验 API 所属关系/run/attempt、ZIP 摘要、安全成员,再验证执行身份、规范化指标/config/拓扑/数据集及原始 eval 覆盖,最后封存 `receipt.json`。 +3. Staging 根据部署的 issuer allowlist 解析回执。缺少原生回执时关闭导入。App 在写数据库或重置 staging 前校验完整快照;部分导入仅能续传同一不可变回执。 +4. 审阅通过的合并/发布 run 结束后,批准 `PublicationRecord` JSON,连接原始回执 artifact/digest、merge SHA/run、changelog artifact/digest 与部署的 app/ingest revision。以 `kind: publication` 运行同一 issuer,不重写原始源回执。 +5. 使用支持的 staging/recovery dispatch。自动 main ingest 在源回执或发布记录尚未封存时延后,不回退到旧原生导入路径。恢复传递准确回执与发布引用;app 校验 exact-run/latest 曲线、trace detail、聚合拓扑及 strict-filter eval 可见性。 + +Merge helper 保留最近明确授权的 `/use RUN_ID` 或 `/reuse-sweep-run RUN_ID`。较新的诊断 run 不会静默替换它;授权证据失效必须重新明确决定。 + +## 验收账本 + +| Gate | 状态 / 所需证据 | +| --- | --- | +| 原生及客户端行为 | CPU 测试、已安装 wheel 检查;不声称 GPU 验收 | +| 回执、app、恢复 | 本地单元/数据库/浏览器 smoke 检查;待部署 | +| H100 吞吐 | 原镜像 c1、2、4、8、16、20、24、28 待运行 | +| 真实 eval | 新 c28 待运行;历史完整原始 eval 通过新 validator | +| 取消与清理 | 本地所属关系/race/closure 测试;实际 Slurm 信号验收待做 | +| 测量等价性 | 对照保留基线比较指标、失败、warmup/drain、服务设置及原始 schema | +| 发布 | 受信任回执、后续发布记录及刷新后 app 证据待完成 | +| 功耗 | 明确临时一致性例外;不声称实测功耗 | +| 退休 | 保留旧 H100 script,等待全部出口证据 | + +证据就绪后记录实际 InferenceX/native/collector/app 提交、source run/attempt、准备期预期、九点 artifact 绑定、源回执、发布记录与 app 验证报告。不得用编造 ID 或占位成功条目关闭 gate。 diff --git a/infx/matrix/generate.py b/infx/matrix/generate.py index 5bbad46d3c..ee8f50b558 100644 --- a/infx/matrix/generate.py +++ b/infx/matrix/generate.py @@ -11,6 +11,7 @@ import yaml from infx.config import repository_root +from infx.srt_slurm.contracts import resolve_reference from .validation import ( DEFAULT_AGENTIC_DURATION_SECONDS, @@ -675,7 +676,8 @@ def mark_all_eval_entries(matrix_values: list[dict]) -> list[dict]: Kimi K3 and MiniMax M3 agentic rows remain one eval job per generated concurrency, using the model's vendor validator. Other agentic entries use GSM8K through lm-eval. Their multi-node rows are merged by topology and - select the highest resulting concurrency. + select the highest resulting concurrency. The version-one native Phase 1 + execution contract permits only its representative c28 eval. Fixed-sequence evals only run at 8k1k. Multi-node rows with the same engine topology are merged into one eval row that runs every concurrency @@ -688,6 +690,12 @@ def mark_all_eval_entries(matrix_values: list[dict]) -> list[dict]: target_isl, target_osl = seq_len_stoi["8k1k"] for entry in matrix_values: + execution = entry.get("execution") or {} + if execution.get("runtime") == "srt-slurm" and execution.get("contract-version") == 1: + entry[Fields.RUN_EVAL.value] = entry[Fields.CONC.value] == 28 + expanded_entries.append(entry) + continue + automatic_eval = automatic_agentic_vendor_eval(entry) if automatic_eval is not None: eval_framework, eval_suite = automatic_eval @@ -982,6 +990,10 @@ def _agentic_entries( if is_multinode: entry[Fields.DISAGG.value] = disagg entry[Fields.SCENARIO_TYPE.value] = "agentic-coding" + if config.get("execution") is not None: + if is_multinode: + raise ValueError("The native pilot execution contract requires one physical node") + entry["execution"] = resolve_reference(config["execution"], repository_root()) if kv_offload_backend is not None: entry[Fields.KV_OFFLOAD_BACKEND.value] = kv_offload_backend entry.update(component_metadata(benchmark, config)) diff --git a/infx/matrix/validation.py b/infx/matrix/validation.py index d3503c7f2f..c74ff9a890 100644 --- a/infx/matrix/validation.py +++ b/infx/matrix/validation.py @@ -12,6 +12,8 @@ model_validator, ) +from infx.srt_slurm.contracts import ExecutionReference + CLUSTER_LABEL_PREFIX = "cluster:" DEFAULT_AGENTIC_DURATION_SECONDS = 3600 @@ -306,6 +308,8 @@ class SingleNodeAgenticMatrixEntry(BaseModel): model_config = ConfigDict(extra="forbid", populate_by_name=True) + execution: ExecutionReference | None = None + image: str model: str model_prefix: str = Field(alias=Fields.MODEL_PREFIX.value) @@ -874,6 +878,8 @@ class SingleNodeMasterConfigEntry(BaseModel): model_config = ConfigDict(extra="forbid", populate_by_name=True) + execution: ExecutionReference | None = None + image: str model: str model_prefix: str = Field(alias=Fields.MODEL_PREFIX.value) @@ -885,6 +891,21 @@ class SingleNodeMasterConfigEntry(BaseModel): router: ComponentMetadata | None = None scenarios: SingleNodeScenarios + @model_validator(mode="after") + def validate_native_execution_scope(self) -> Self: + if self.execution is not None and ( + self.model_prefix != "dsv41flash" + or self.framework != "vllm" + or self.precision != "fp4" + or self.runner != "cluster:h100-dgxc" + or self.scenarios.fixed_seq_len + or not self.scenarios.agentic_coding + ): + raise ValueError( + "native execution is currently restricted to the H100 DSV4.1-Flash AgentX pilot" + ) + return self + @model_validator(mode="after") def validate_agentic_runner(self) -> Self: _validate_agentic_runner_is_cluster(self.runner, self.scenarios) diff --git a/infx/srt_slurm/client_guard.py b/infx/srt_slurm/client_guard.py new file mode 100644 index 0000000000..4fb2a127ce --- /dev/null +++ b/infx/srt_slurm/client_guard.py @@ -0,0 +1,37 @@ +"""Verify prepared wrapper/files inside the container before starting a client.""" + +from __future__ import annotations + +import argparse +from pathlib import Path + +from infx.benchmarks import agentx, eval as real_eval +from infx.benchmarks.identity import capture_identity +from infx.benchmarks.spec import AgentXSpec, EvalSpec +from infx.srt_slurm.job import read_json +from infx.srt_slurm.launch import verify_bundle + + +def main() -> int: + import os + + parser = argparse.ArgumentParser() + parser.add_argument("--bundle", type=Path, required=True) + parser.add_argument("--client", choices=("agentx", "eval"), required=True) + parser.add_argument("--spec", type=Path, required=True) + parser.add_argument("--artifact-root", type=Path, required=True) + args = parser.parse_args() + bundle = read_json(args.bundle) + verify_bundle(bundle) + identity = capture_identity(bundle["site"]["wrapper_python"], ["infx"], dataset_loader=None) + if identity != bundle["identity"]["wrapper_identity"]: + raise ValueError("installed InferenceX wrapper changed after preparation") + endpoint = os.environ["SRT_ENDPOINT"] + spec = read_json(args.spec) + if args.client == "eval": + return real_eval.run(EvalSpec.model_validate(spec), endpoint, args.artifact_root) + return agentx.run(AgentXSpec.model_validate(spec), endpoint, args.artifact_root) + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/infx/srt_slurm/job.py b/infx/srt_slurm/job.py new file mode 100644 index 0000000000..37fd2d30c1 --- /dev/null +++ b/infx/srt_slurm/job.py @@ -0,0 +1,127 @@ +"""Strict workflow-to-runtime boundary for the first aggregate migration lane.""" + +from __future__ import annotations + +import hashlib +from pathlib import Path +from typing import Any, Literal + +from pydantic import BaseModel, ConfigDict, Field, field_validator, model_validator + +from infx.benchmarks.common import decode_json +from infx.srt_slurm.contracts import digest, resolve_reference +from infx.workflows.benchmark_schema import AgenticConfig + + +class SchedulingEnvelope(BaseModel): + model_config = ConfigDict(extra="forbid", strict=True, populate_by_name=True) + + priority: str + queue_token: str = Field(alias="queue-token", min_length=1) + node_count: Literal[1] = Field(alias="node-count") + + @field_validator("node_count", mode="before") + @classmethod + def integer_node_count(cls, value: Any) -> int: + if type(value) is not int: + raise ValueError("node-count must be an integer, not a boolean or coerced value") + return value + + @field_validator("priority") + @classmethod + def finite_priority(cls, value: str) -> str: + from decimal import Decimal, InvalidOperation + + try: + priority = Decimal(value) + except InvalidOperation as error: + raise ValueError("priority must be a finite decimal") from error + if not priority.is_finite(): + raise ValueError("priority must be a finite decimal") + return value + + +class JobSpec(BaseModel): + model_config = ConfigDict(extra="forbid", strict=True) + + schema_version: Literal[1] + row: AgenticConfig + scheduling: SchedulingEnvelope + + @model_validator(mode="after") + def pilot_boundary(self) -> JobSpec: + row = self.row + if row.execution is None: + raise ValueError("native execution reference is required") + if ( + row.model_prefix != "dsv41flash" + or row.model != "deepseek-ai/DeepSeek-V4.1-Flash" + or row.framework != "vllm" + or row.precision != "fp4" + or row.runner != "cluster:h100-dgxc" + or (row.tp, row.pp, row.dcp_size, row.pcp_size, row.ep) != (8, 1, 1, 1, 1) + or row.dp_attn + or row.kv_offloading != "none" + or row.spec_decoding != "mtp" + or row.duration != 3600 + or row.conc not in (1, 2, 4, 8, 16, 20, 24, 28) + or row.router is not None + or row.eval_suite not in (None, "") + ): + raise ValueError("execution is restricted to the Phase 1 H100 aggregate pilot") + if row.run_eval and not row.eval_only: + raise ValueError("throughput and real eval must be separate jobs") + if row.eval_only and (row.eval_framework != "lm-eval" or row.conc != 28): + raise ValueError("the pilot requires its representative c28 lm-eval job") + return self + + @property + def mode(self) -> str: + return "eval" if self.row.eval_only else "throughput" + + def semantic_inputs(self) -> dict[str, Any]: + row = self.row.model_dump(by_alias=True, exclude_none=True) + for name in ("exp-name", "recipe-fingerprint", "run-eval", "eval-only"): + row.pop(name, None) + return {"schema_version": self.schema_version, "mode": self.mode, "row": row} + + @property + def point_id(self) -> str: + return digest(self.semantic_inputs()) + + +def parse_job(raw: dict[str, Any], root: Path, scheduling: dict[str, Any] | None = None) -> JobSpec: + """Validate separately supplied or priority-annotated workflow inputs without defaults.""" + row = dict(raw) + annotation = { + key: row.pop(key) for key in ("priority", "queue-token", "node-count") if key in row + } + if scheduling is not None: + if any(key in annotation and annotation[key] != value for key, value in scheduling.items()): + raise ValueError("conflicting workflow scheduling envelopes") + annotation.update(scheduling) + if "execution" not in row: + raise ValueError("missing explicit native execution reference") + row["execution"] = resolve_reference(row["execution"], root) + return JobSpec.model_validate( + {"schema_version": 1, "row": row, "scheduling": annotation}, strict=True + ) + + +def intent_id(repository: str, run_id: str, attempt: str, point_id: str) -> str: + """A repeated workflow launch recovers one intent instead of allocating again.""" + if not repository or not run_id.isdecimal() or not attempt.isdecimal(): + raise ValueError("explicit GitHub execution identity is required") + return digest({"repository": repository, "run": run_id, "attempt": attempt, "point": point_id}) + + +def read_json(path: Path) -> dict[str, Any]: + value = decode_json(path.read_text()) + if not isinstance(value, dict): + raise ValueError("expected JSON object") + return value + + +def file_digest(path: Path) -> str: + with path.open("rb") as stream: + return hashlib.file_digest(stream, "sha256").hexdigest() diff --git a/infx/srt_slurm/launch.py b/infx/srt_slurm/launch.py new file mode 100644 index 0000000000..ad93d901f0 --- /dev/null +++ b/infx/srt_slurm/launch.py @@ -0,0 +1,524 @@ +"""Prepare an immutable one-point bundle and delegate allocation to native srtctl.""" + +from __future__ import annotations + +import argparse +import fcntl +import json +import os +import shutil +import signal +import subprocess +import time +from pathlib import Path +from typing import Any, Literal + +import yaml +from pydantic import BaseModel, ConfigDict, Field + +from infx.benchmarks.common import child_failed, verify_file, write_json +from infx.benchmarks.identity import capture_identity, verify_runtime +from infx.benchmarks.spec import RuntimeSpec +from infx.srt_slurm.contracts import digest, load_mapping +from infx.srt_slurm.job import JobSpec, file_digest, intent_id, parse_job, read_json +from infx.srt_slurm.render import ClientPolicy, PilotSite, client_spec, render_recipe + + +class RuntimeLock(BaseModel): + model_config = ConfigDict(extra="forbid", strict=True) + + schema_version: Literal[1] + repository: str + revision: str = Field(pattern=r"^[0-9a-f]{40}$") + uv_lock_sha256: str = Field(pattern=r"^[0-9a-f]{64}$") + capabilities: list[str] + + +class NativeCommandError(RuntimeError): + def __init__(self, output: dict[str, Any], detail: str) -> None: + self.output = output + super().__init__(detail) + + +def checked_json(argv: list[str], *, timeout: int = 600) -> dict[str, Any]: + with subprocess.Popen( + argv, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, start_new_session=True + ) as process: + try: + stdout, stderr = process.communicate(timeout=timeout) + except BaseException: + from contextlib import suppress + + with suppress(ProcessLookupError): + os.killpg(process.pid, signal.SIGTERM) + with suppress(subprocess.TimeoutExpired): + process.wait(timeout=3) + with suppress(ProcessLookupError): + os.killpg(process.pid, signal.SIGKILL) + process.wait(timeout=5) + raise + result = subprocess.CompletedProcess(argv, process.returncode, stdout, stderr) + try: + value = json.loads(result.stdout) + except json.JSONDecodeError as error: + raise RuntimeError(f"native command did not return JSON: {result.stderr}") from error + if not isinstance(value, dict): + raise ValueError("native command must return a JSON object") + if result.returncode: + raise NativeCommandError( + value, f"native command failed ({result.returncode}): {value}\n{result.stderr}" + ) + return value + + +def native(site: PilotSite, *args: str, timeout: int = 600) -> dict[str, Any]: + return checked_json( + [site.native_python, "-I", "-m", "srtctl.cli.submit", *args, "--json"], timeout=timeout + ) + + +def verify_site(job: JobSpec, site: PilotSite, root: Path) -> dict[str, Any]: + reference = job.row.execution + if reference is None: + raise ValueError("missing native execution reference") + lock = RuntimeLock.model_validate(read_json(root / reference.runtime_lock)) + source = Path(site.native_source) + revision = subprocess.run( + ["git", "-C", str(source), "rev-parse", "HEAD"], check=True, capture_output=True, text=True + ).stdout.strip() + dirty = subprocess.run( + ["git", "-C", str(source), "status", "--porcelain", "--untracked-files=all"], + check=True, + capture_output=True, + text=True, + ).stdout + if revision != lock.revision or dirty or file_digest(source / "uv.lock") != lock.uv_lock_sha256: + raise ValueError("native runtime checkout or dependency lock differs from pilot pin") + verify_file(site.image) + if ( + not Path(site.model_snapshot).is_dir() + or Path(site.model_snapshot).name != site.model_revision + ): + raise ValueError("model must be a prepared immutable Hugging Face snapshot") + for value in ( + site.native_python, + site.native_source, + site.wrapper_python, + site.shared_root, + site.model_snapshot, + ): + site.require_visible(value) + if set(site.client_sites) != {"agentx", "eval"}: + raise ValueError("both throughput and real eval client environments must be provisioned") + wrapper_identity = capture_identity(site.wrapper_python, ["infx"], dataset_loader=None) + site.require_interpreter(wrapper_identity, python_minor="3.12") + verify_wrapper_source(wrapper_identity, root) + native_identity = capture_identity(site.native_python, ["srtctl"], dataset_loader=None) + site.require_interpreter(native_identity, python_minor="3.12") + return { + "runtime_lock": lock.model_dump(), + "wrapper_identity": wrapper_identity, + "native_identity": native_identity, + } + + +def verify_wrapper_source(identity: dict[str, Any], root: Path) -> None: + distribution = identity["distributions"]["infx"] + if (distribution.get("direct_url") or {}).get("dir_info", {}).get("editable"): + raise ValueError("prepared wrapper must be installed noneditable") + installed = { + name: value for name, value in distribution["files"].items() if name.startswith("infx/") + } + expected_python = { + str(path.relative_to(root)) + for path in (root / "infx").rglob("*.py") + if "__pycache__" not in path.parts + } + if not expected_python <= installed.keys(): + raise ValueError("installed wrapper is missing candidate Python modules") + for name, value in installed.items(): + path = root / name + if not path.is_file() or file_digest(path) != value: + raise ValueError(f"installed wrapper differs from candidate checkout: {name}") + + +def effective_identity( + job: JobSpec, + site: PilotSite, + identity: dict[str, Any], + runtime: RuntimeSpec, + resources: dict[str, Any], +) -> tuple[str, str]: + semantics = job.semantic_inputs() + inputs = { + "requested": semantics, + "installed": identity, + "client_identity": read_json(Path(runtime.identity.path)), + "client_assets": {asset.path: asset.sha256 for asset in runtime.assets}, + "client_env": runtime.env, + "client_env_unset": sorted(runtime.env_unset), + "model_revision": site.model_revision, + "image_sha256": site.image.sha256, + "dataset_revision": resources.get("dataset_revision"), + "task_sha256": resources.get("task", {}).get("sha256"), + "document_identities_sha256": resources.get("document_identities", {}).get("sha256"), + } + point = digest(inputs) + semantics["row"].pop("conc", None) + return point, digest(inputs) + + +def prepare( + job: JobSpec, + site: PilotSite, + root: Path, + source: dict[str, Any], +) -> dict[str, Any]: + """A local preparation lock protects files; only native srtctl may claim/submit Slurm.""" + from infx.benchmarks.prepare import ClientSite, prepare as prepare_client + + execution = intent_id( + source["repository"], str(source["run_id"]), str(source["attempt"]), job.point_id + ) + directory = Path(site.shared_root) / "runs" / execution + directory.mkdir(parents=True, exist_ok=True) + with (directory / "prepare.lock").open("a") as stream: + fcntl.flock(stream, fcntl.LOCK_EX) + index = directory / "bundle.json" + if index.exists(): + bundle = read_json(index) + old_job = JobSpec.model_validate(bundle["job"]) + if old_job.semantic_inputs() != job.semantic_inputs() or bundle["source"] != source: + raise ValueError("existing execution bundle belongs to different inputs") + verify_bundle(bundle) + return bundle + if (directory / "client").exists() or (directory / "native").exists(): + raise ValueError( + "incomplete preparation requires inspection; refusing to overwrite or allocate" + ) + identity = verify_site(job, site, root) + kind = "eval" if job.mode == "eval" else "agentx" + client_site = ClientSite.model_validate(read_json(Path(site.client_sites[kind]))) + if client_site.model_path != site.model_snapshot: + raise ValueError("serving and client model snapshots disagree") + prepare_client(client_site, kind=kind, output=directory / "client") + runtime = RuntimeSpec.model_validate(read_json(directory / "client" / "runtime.json")) + site.require_interpreter( + read_json(Path(runtime.identity.path)), + python_minor="3.11" if kind == "agentx" else None, + ) + for path in ( + runtime.python, + runtime.identity.path, + *(asset.path for asset in runtime.assets), + ): + site.require_visible(path) + for key in ( + "HF_HUB_CACHE", + "HF_DATASETS_CACHE", + "HF_MODULES_CACHE", + "AIPERF_DATASET_MMAP_CACHE_DIR", + ): + if runtime.env.get(key): + site.require_visible(runtime.env[key], writable=True) + resources = read_json(directory / "client" / "prepared-resources.json") + reference = job.row.execution + if reference is None: + raise ValueError("missing execution reference") + policy = ClientPolicy.model_validate(load_mapping(root / reference.client_policy)) + point_id, curve_id = effective_identity(job, site, identity, runtime, resources) + spec = client_spec(job, policy, runtime, resources, point_id) + spec_path = directory / "client.json" + write_json(spec_path, spec.model_dump(mode="json")) + recipe, profile = render_recipe( + job, root, site, policy, spec_path, directory / "client-output" + ) + recipe["benchmark"]["argv"][3:4] = [ + "infx.srt_slurm.client_guard", + "--bundle", + str(index), + "--client", + kind, + ] + # Retain --spec/--artifact-root after the guarded module's explicit options. + for name, data in (("recipe.yaml", recipe), ("profile.yaml", profile)): + (directory / name).write_text(yaml.safe_dump(data, sort_keys=False)) + prepared = native( + site, + "prepare", + "--recipe", + str(directory / "recipe.yaml"), + "--profile", + str(directory / "profile.yaml"), + "--output", + str(directory / "native"), + "--expected-nodes", + str(job.scheduling.node_count), + "--runtime-python", + site.native_python, + ) + if prepared.get("state") != "prepared" or prepared["resources"] != { + "nodes": 1, + "gpus_per_node": 8, + "serving_gpus": 8, + "workers": 1, + "cardinality": 1, + }: + raise ValueError( + "native resolved allocation differs from the queued TP8 aggregate point" + ) + if not set(identity["runtime_lock"]["capabilities"]) <= set( + prepared.get("capabilities", []) + ): + raise ValueError("native runtime lacks required Phase 1 capabilities") + files = { + str(path): file_digest(path) + for path in directory.rglob("*") + if path.is_file() and path.name != "prepare.lock" + } + bundle = { + "schema_version": 1, + "point_id": point_id, + "requested_point_id": job.point_id, + "effective_curve_id": curve_id, + "execution_id": execution, + "job": job.model_dump(mode="json", by_alias=True), + "source": source, + "site": site.model_dump(mode="json"), + "identity": identity, + "prepared": prepared, + "files": files, + "directory": str(directory), + "telemetry": policy.telemetry, + } + bundle["bundle_digest"] = digest(bundle) + write_json(index, bundle) + for path in (*files, str(index)): + Path(path).chmod(0o444) + return bundle + + +def verify_bundle(bundle: dict[str, Any]) -> None: + if bundle["bundle_digest"] != digest( + {key: value for key, value in bundle.items() if key != "bundle_digest"} + ): + raise ValueError("prepared bundle identity changed") + for name, expected in bundle["files"].items(): + if file_digest(Path(name)) != expected: + raise ValueError(f"prepared input changed: {name}") + + +def verify_execution_clients(bundle: dict[str, Any], site: PilotSite) -> None: + verify_file(site.image) + actual = capture_identity(site.wrapper_python, ["infx"], dataset_loader=None) + if actual != bundle["identity"]["wrapper_identity"]: + raise ValueError("installed wrapper changed before allocation") + spec = read_json(Path(bundle["directory"]) / "client.json") + verify_runtime( + RuntimeSpec.model_validate(spec["runtime"]), dataset_loader=spec.get("dataset_loader") + ) + + +def copy_diagnostics(bundle: dict[str, Any], workspace: Path, *, complete: bool) -> None: + directory = Path(bundle["directory"]) + diagnostics = workspace / "native-execution" + diagnostics.mkdir(exist_ok=True) + for name in bundle["files"]: + source = Path(name) + relative = source.relative_to(directory) + target = diagnostics / "prepared" / relative + if source.is_symlink(): + raise ValueError("prepared diagnostic input is a symlink") + target.parent.mkdir(parents=True, exist_ok=True) + shutil.copyfile(source, target) + shutil.copyfile(directory / "bundle.json", diagnostics / "bundle.json") + client = directory / "client-output" + if client.exists(): + if any(path.is_symlink() for path in client.rglob("*")): + raise ValueError("client diagnostics contain a symlink") + shutil.copytree(client, workspace / "results", dirs_exist_ok=True) + logs = directory / "native-output" + for path in logs.rglob("*"): + if path.is_file() and not path.is_symlink() and path.suffix in {".log", ".json", ".yaml"}: + target = workspace / "results" / "native" / path.relative_to(logs) + target.parent.mkdir(parents=True, exist_ok=True) + shutil.copyfile(path, target) + write_json(diagnostics / "output-state.json", {"schema_version": 1, "complete": complete}) + + +def publish_outputs(bundle: dict[str, Any], receipt: dict[str, Any], workspace: Path) -> None: + directory = Path(bundle["directory"]) + client = directory / "client-output" + audit = read_json(client / "diagnostics" / "client-audit.json") + if audit["errors"] or child_failed(audit["status"]): + raise ValueError("client did not produce a successful closed result") + if bundle["job"]["row"].get("eval-only"): + names = [ + *client.glob("results*.json"), + *client.glob("samples*.jsonl"), + client / "meta_env.json", + ] + else: + names = [client / f"{bundle['point_id']}.json"] + for source_path in names: + destination = workspace / source_path.name + if destination.exists() and file_digest(destination) != file_digest(source_path): + raise ValueError(f"refusing to overwrite different result: {destination}") + shutil.copyfile(source_path, destination) + diagnostics = workspace / "native-execution" + diagnostics.mkdir(exist_ok=True) + shutil.copyfile(directory / "bundle.json", diagnostics / "bundle.json") + write_json( + diagnostics / "execution.json", + { + "schema_version": 1, + "point_id": bundle["point_id"], + "execution_id": bundle["execution_id"], + "bundle_digest": bundle["bundle_digest"], + "source": bundle["source"], + "mode": "eval" if bundle["job"]["row"].get("eval-only") else "throughput", + "native_receipt": { + "job_id": receipt["job_id"], + "state": "COMPLETED", + "manifest_sha256": bundle["prepared"]["manifest_sha256"], + }, + "client_exit_code": 0, + }, + ) + + +def execute(bundle: dict[str, Any], workspace: Path, *, reconcile_timeout: int = 120) -> None: + verify_bundle(bundle) + site = PilotSite.model_validate(bundle["site"]) + verify_execution_clients(bundle, site) + journal = Path(site.shared_root) / "journal" + receipt_path = Path( + native( + site, + "intent-path", + "--intent", + bundle["execution_id"], + "--cluster", + site.cluster, + "--journal-dir", + str(journal), + )["receipt_path"] + ) + receipt: dict[str, Any] | None = None + completed = False + previous = {} + + def interrupted(signum: int, _frame: Any) -> None: + raise InterruptedError(f"workflow interrupted by signal {signum}") + + for signum in (signal.SIGINT, signal.SIGTERM): + previous[signum] = signal.signal(signum, interrupted) + try: + try: + receipt = native( + site, + "submit-prepared", + "--prepared-dir", + bundle["prepared"]["prepared_dir"], + "--intent", + bundle["execution_id"], + "--cluster", + site.cluster, + "--journal-dir", + str(journal), + ) + except NativeCommandError as error: + if error.output.get("receipt_path"): + receipt_path = Path(error.output["receipt_path"]) + raise + receipt_path = Path(receipt["receipt_path"]) + if receipt.get("state") != "accepted" or len(receipt["accepted_ids"]) != 1: + raise ValueError( + "submission was not uniquely accepted; reconcile the native journal before retry" + ) + state = native( + site, + "wait", + "--receipt", + str(receipt_path), + "--timeout", + "28800", + "--poll", + "10", + timeout=28920, + ) + if state.get("state") != "completed" or state.get("terminal") is not True: + raise ValueError(f"native job has not completed successfully: {state}") + publish_outputs(bundle, receipt, workspace) + completed = True + finally: + for signum, handler in previous.items(): + signal.signal(signum, handler) + try: + if not completed and receipt_path is not None and receipt_path.exists(): + deadline = time.monotonic() + reconcile_timeout + while True: + try: + recovery = native( + site, "reconcile", "--receipt", str(receipt_path), timeout=180 + ) + except NativeCommandError as error: + recovery = error.output + if recovery.get("accepted_ids") or time.monotonic() >= deadline: + break + time.sleep(min(2, max(0, deadline - time.monotonic()))) + if recovery.get("accepted_ids"): + native(site, "cancel-known", "--receipt", str(receipt_path), timeout=180) + try: + closure = native( + site, + "wait-known", + "--receipt", + str(receipt_path), + "--timeout", + "120", + "--poll", + "5", + timeout=150, + ) + except NativeCommandError as error: + closure = error.output + if closure.get("terminal") is not True: + raise RuntimeError( + f"owned allocation cleanup is unresolved: {receipt_path}" + ) + else: + raise RuntimeError( + f"submission ownership remains unresolved; intent stays fenced: {receipt_path}" + ) + finally: + copy_diagnostics(bundle, workspace, complete=completed) + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--job", type=Path, required=True) + parser.add_argument("--site", type=Path, required=True) + parser.add_argument("--root", type=Path, required=True) + parser.add_argument("--source", type=Path, required=True) + parser.add_argument("--prepare-only", action="store_true") + args = parser.parse_args() + job = parse_job(read_json(args.job), args.root) + site = PilotSite.model_validate(read_json(args.site)) + bundle = prepare(job, site, args.root, read_json(args.source)) + print( + json.dumps( + { + "point_id": bundle["point_id"], + "bundle_digest": bundle["bundle_digest"], + "bundle": str(Path(bundle["directory"]) / "bundle.json"), + } + ) + ) + if not args.prepare_only: + execute(bundle, args.root) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/infx/srt_slurm/render.py b/infx/srt_slurm/render.py new file mode 100644 index 0000000000..d7de84bc73 --- /dev/null +++ b/infx/srt_slurm/render.py @@ -0,0 +1,258 @@ +"""Render the explicitly selected aggregate recipe and prepared Python client.""" + +from __future__ import annotations + +import copy +import math +from pathlib import Path +from typing import Any, Literal + +from pydantic import BaseModel, ConfigDict, Field, field_validator + +from infx.benchmarks.spec import AgentXSpec, EvalSpec, PreparedFile, ResultMetadata, RuntimeSpec +from infx.srt_slurm.contracts import load_mapping +from infx.srt_slurm.job import JobSpec + + +class ClientPolicy(BaseModel): + model_config = ConfigDict(extra="forbid", strict=True) + + schema_version: Literal[1] + golden_curve: str + golden_model: str + thinking_mode: Literal["thinking_on"] + dataset_repository: Literal["semianalysisai/cc-traces-weka-062126"] + dataset_loader: Literal["semianalysis_cc_traces_weka_062126"] + dataset_entries: Literal[393] + duration_seconds: Literal[3600] + warmup_requests_per_lane: Literal[10] + warmup_grace_seconds: Literal[1800] + trace_idle_gap_cap_seconds: Literal[300] + live_failed_request_threshold: Literal[0.1] + failed_request_threshold: Literal[0.1] + random_seed: Literal[42] + required_server_metric_prefix: Literal["vllm:"] + eval_task: Literal["gsm8k"] + eval_documents: Literal[1319] + eval_max_length: Literal[16384] + eval_max_tokens: Literal[12288] + eval_minimum_score: Literal[0.9] + telemetry: Literal["temporary-parity-exception-no-native-power"] + + +class PilotSite(BaseModel): + """Provisioned shared paths; no implicit host/environment fallback.""" + + model_config = ConfigDict(extra="forbid", strict=True) + + schema_version: Literal[1] + cluster: Literal["h100-dgxc"] + native_python: str + native_source: str + wrapper_python: str + shared_root: str + model_snapshot: str + model_revision: str = Field(pattern=r"^[0-9a-f]{40}$") + image: PreparedFile + image_reference: str + client_sites: dict[Literal["agentx", "eval"], str] + mounts: dict[str, str] + # Receipt reader deployment is a release prerequisite, not inferred from code presence. + reader_revision: str = Field(pattern=r"^[0-9a-f]{40}$") + collector_revision: str = Field(pattern=r"^[0-9a-f]{40}$") + + @field_validator( + "native_python", "native_source", "wrapper_python", "shared_root", "model_snapshot" + ) + @classmethod + def absolute(cls, value: str) -> str: + if not Path(value).is_absolute(): + raise ValueError("site paths must be absolute and shared with compute nodes") + return value + + @field_validator("shared_root") + @classmethod + def writable_root(cls, value: str) -> str: + if Path(value).resolve().is_relative_to("/workspace"): + raise ValueError("pilot output must not create directories under /workspace") + return value + + @field_validator("mounts") + @classmethod + def identity_mounts(cls, value: dict[str, str]) -> dict[str, str]: + if not value or any(not Path(k).is_absolute() or k != v for k, v in value.items()): + raise ValueError("prepared interpreters and assets require explicit same-path mounts") + if any(str(Path(path).resolve()) != path for path in value): + raise ValueError( + "same-path mount roots must be canonical paths without symlink aliases" + ) + return value + + def require_visible(self, value: str, *, writable: bool = False) -> None: + path = Path(value) + if not path.is_absolute() or not all( + any(candidate.is_relative_to(Path(mount)) for mount in self.mounts) + for candidate in (path, path.resolve()) + ): + raise ValueError(f"prepared path is not mounted at its absolute location: {value}") + if writable and path.resolve().is_relative_to("/workspace"): + raise ValueError("pilot caches must not create directories under /workspace") + + def require_interpreter( + self, identity: dict[str, Any], *, python_minor: str | None = None + ) -> None: + paths = identity.get("python_paths", {}) + for name in ("executable", "executable_resolved", "prefix", "base_prefix"): + value = paths.get(name) + if not isinstance(value, str): + raise ValueError(f"prepared interpreter identity is missing {name}") + self.require_visible(value) + if python_minor is not None and not identity.get("python_version", "").startswith( + python_minor + "." + ): + raise ValueError(f"pilot interpreter must use Python {python_minor}") + + +def golden_acceptance(root: Path, policy: ClientPolicy, draft_tokens: int) -> float: + path = (root / policy.golden_curve).resolve(strict=True) + if not path.is_relative_to(root.resolve()): + raise ValueError("golden curve must be a committed checkout resource") + curve = load_mapping(path) + value = curve[policy.golden_model][policy.thinking_mode][draft_tokens] + if isinstance(value, bool) or not isinstance(value, int | float): + raise ValueError("golden acceptance must be numeric") + if not math.isfinite(value) or not 1 <= value <= draft_tokens + 1: + raise ValueError("golden acceptance is outside the draft/target token range") + return float(value) + + +def client_spec( + job: JobSpec, + policy: ClientPolicy, + runtime: RuntimeSpec, + resources: dict[str, Any], + result_filename: str, +) -> AgentXSpec | EvalSpec: + row = job.row + metadata = ResultMetadata( + hw="h100", + model=row.model, + model_prefix=row.model_prefix, + image=row.image, + framework=row.framework, + precision=row.precision, + spec_decoding=row.spec_decoding, + tp=row.tp, + pp=row.pp, + dcp_size=row.dcp_size, + pcp_size=row.pcp_size, + ep=row.ep, + dp_attention=row.dp_attn, + total_cpu_dram_gb=row.total_cpu_dram_gb, + recipe_fingerprint=row.recipe_fingerprint or job.point_id, + ) + shared = { + "schema_version": 1, + "runtime": runtime, + "metadata": metadata, + "concurrency": row.conc, + } + if job.mode == "eval": + return EvalSpec( + **shared, + task=PreparedFile.model_validate(resources["task"]), + document_identities=PreparedFile.model_validate(resources["document_identities"]), + task_name=policy.eval_task, + expected_documents=policy.eval_documents, + max_length=policy.eval_max_length, + max_tokens=policy.eval_max_tokens, + minimum_score=policy.eval_minimum_score, + ) + names = ( + "dataset_repository", + "dataset_loader", + "dataset_entries", + "duration_seconds", + "warmup_requests_per_lane", + "warmup_grace_seconds", + "trace_idle_gap_cap_seconds", + "live_failed_request_threshold", + "failed_request_threshold", + "random_seed", + "required_server_metric_prefix", + ) + return AgentXSpec( + **shared, + **{key: getattr(policy, key) for key in names}, + result_filename=result_filename, + tokenizer=row.model, + dataset_revision=resources["dataset_revision"], + ) + + +def render_recipe( + job: JobSpec, + root: Path, + site: PilotSite, + policy: ClientPolicy, + spec_path: Path, + output: Path, +) -> tuple[dict[str, Any], dict[str, Any]]: + reference = job.row.execution + if reference is None: + raise ValueError("explicit execution reference is required") + recipe = copy.deepcopy(load_mapping(root / reference.recipe)) + if recipe["model"]["path"] != job.row.model or recipe["model"]["container"] != job.row.image: + raise ValueError("master and recipe model/image identities disagree") + if site.image_reference != job.row.image: + raise ValueError("prepared image identity does not match the selected image") + if set(recipe["roles"]) != {"agg"} or recipe["frontend"]["type"] != "vllm": + raise ValueError("pilot requires one aggregate direct vLLM frontend") + role = recipe["roles"]["agg"] + if (role["nodes"], role["workers"], role["gpus"]) != (1, 1, 8): + raise ValueError("pilot requires one physical node and one TP8 worker") + args = role["args"] + args["max-num-seqs"] = 2 * job.row.conc + args["max-cudagraph-capture-size"] = min(2048, 1 << (12 * job.row.conc - 1).bit_length()) + spec = args["speculative-config"] + if spec.get("synthetic_acceptance_length") is not None: + raise ValueError("recipe must not hard-code a synthetic acceptance value") + spec["enable_adaptive_verification"] = job.mode == "eval" + spec["rejection_sample_method"] = "block" if job.mode == "eval" else "synthetic" + if job.mode != "eval": + spec["synthetic_acceptance_length"] = golden_acceptance( + root, policy, spec["num_speculative_tokens"] + ) + recipe["model"]["path"] = site.model_snapshot + recipe["model"]["container"] = site.image.path + recipe["identity"] = { + "model": {"repo": job.row.model, "revision": site.model_revision}, + "container": {"image": job.row.image}, + } + module = "eval" if job.mode == "eval" else "agentx" + recipe["benchmark"] = { + "type": "custom", + "argv": [ + site.wrapper_python, + "-I", + "-m", + f"infx.benchmarks.{module}", + "--spec", + str(spec_path), + "--artifact-root", + str(output), + ], + "cwd": str(spec_path.parent), + "env": {"HF_HUB_OFFLINE": "1", "HF_DATASETS_OFFLINE": "1"}, + "env_unset": ["PYTHONPATH", "PYTHONHOME", "BASH_ENV", "ENV"], + "container_image": site.image.path, + } + profile = load_mapping(root / reference.profile) + if profile.get("use_exclusive_sbatch_directive") is not True: + raise ValueError("the direct port 8000 policy requires an exclusive node") + profile.update( + srtctl_root=site.native_source, + output_dir=str(spec_path.parent / "native-output"), + default_mounts=site.mounts, + ) + return recipe, profile diff --git a/infx/srt_slurm/workflow.py b/infx/srt_slurm/workflow.py new file mode 100644 index 0000000000..5bc4bb75ec --- /dev/null +++ b/infx/srt_slurm/workflow.py @@ -0,0 +1,80 @@ +"""GitHub Actions boundary for the isolated native H100 pilot.""" + +from __future__ import annotations + +import json +import os +import subprocess +from pathlib import Path + +from infx.benchmarks.common import write_json +from infx.srt_slurm.job import parse_job +from infx.srt_slurm.launch import execute, prepare +from infx.srt_slurm.render import PilotSite + + +def main() -> int: + root = Path(os.environ["GITHUB_WORKSPACE"]).resolve() + raw = json.loads(os.environ["NATIVE_CONFIG_JSON"]) + if os.environ["NATIVE_AGENTX_FAST"] != "false" or os.environ["NATIVE_EVAL_LIMIT"] not in ( + "", + "full", + ): + raise ValueError("Phase 1 requires the full-duration/full-dataset qualification policy") + if os.environ["NATIVE_REQUIRE_POWER"] != "false": + raise ValueError("the Phase 1 telemetry exception cannot satisfy require-power") + for key, variable in (("run-eval", "NATIVE_RUN_EVAL"), ("eval-only", "NATIVE_EVAL_ONLY")): + value = json.loads(os.environ[variable]) + if key in raw and raw[key] != value: + raise ValueError(f"workflow and matrix disagree on {key}") + raw[key] = value + if raw["eval-only"]: + raw["eval-framework"] = os.environ["NATIVE_EVAL_FRAMEWORK"] + raw["eval-suite"] = os.environ["NATIVE_EVAL_SUITE"] + job = parse_job( + raw, + root, + { + "priority": os.environ["NATIVE_PRIORITY"], + "queue-token": os.environ["NATIVE_QUEUE_TOKEN"], + "node-count": 1, + }, + ) + site = PilotSite.model_validate_json(os.environ["NATIVE_SITE_JSON"]) + if ( + os.environ["NATIVE_READER_REVISION"] != site.reader_revision + or os.environ["NATIVE_COLLECTOR_REVISION"] != site.collector_revision + ): + raise ValueError("reader-first deployment and trusted collector pins are not enabled") + source = { + "repository": os.environ["GITHUB_REPOSITORY"], + "run_id": int(os.environ["GITHUB_RUN_ID"]), + "attempt": int(os.environ["GITHUB_RUN_ATTEMPT"]), + "head_sha": subprocess.run( + ["git", "rev-parse", "HEAD"], check=True, text=True, capture_output=True + ).stdout.strip(), + } + bundle = prepare(job, site, root, source) + with Path(os.environ["GITHUB_ENV"]).open("a") as stream: + stream.write( + f"RESULT_FILENAME={bundle['point_id']}\nNATIVE_POINT_ID={bundle['point_id']}\nGPU_COUNT=8\n" + ) + diagnostics = root / "native-execution" + diagnostics.mkdir(exist_ok=True) + write_json( + diagnostics / "prepared.json", + { + "schema_version": 1, + "point_id": bundle["point_id"], + "execution_id": bundle["execution_id"], + "bundle_digest": bundle["bundle_digest"], + "native_manifest_sha256": bundle["prepared"]["manifest_sha256"], + "source": source, + }, + ) + execute(bundle, root) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/infx/workflows/merge_source.py b/infx/workflows/merge_source.py new file mode 100644 index 0000000000..ca6c19c7cb --- /dev/null +++ b/infx/workflows/merge_source.py @@ -0,0 +1,102 @@ +"""Select reuse evidence without replacing an already authorized source run.""" + +from __future__ import annotations + +import argparse +import json +import subprocess +from typing import Any + +from infx.workflows.reuse import DEFAULT_ALLOWED_AUTHOR_ASSOCIATIONS, parse_reuse_command + + +def gh(path: str, *, paginated: bool = False) -> Any: + args = ["gh", "api", path] + if paginated: + args.extend(("--paginate", "--slurp")) + return json.loads(subprocess.run(args, check=True, text=True, capture_output=True).stdout) + + +def select_source(repo: str, pr_number: int, branch: str, explicit: int | None = None) -> int: + comments = [ + item + for page in gh(f"repos/{repo}/issues/{pr_number}/comments?per_page=100", paginated=True) + for item in page + ] + comments.sort(key=lambda value: (value["created_at"], value["id"]), reverse=True) + authorized = None + for comment in comments: + if comment.get("author_association") not in DEFAULT_ALLOWED_AUTHOR_ASSOCIATIONS: + continue + matches, pinned = parse_reuse_command(comment.get("body", "")) + if matches: + authorized = pinned + break + if explicit is not None and authorized is not None and explicit != authorized: + raise ValueError("explicit source conflicts with the existing maintainer authorization") + selected = authorized if authorized is not None else explicit + shas = { + item["sha"] + for page in gh(f"repos/{repo}/pulls/{pr_number}/commits?per_page=100", paginated=True) + for item in page + } + if selected is None: + from urllib.parse import quote + + runs = [ + item + for page in gh( + f"repos/{repo}/actions/workflows/run-sweep.yml/runs?event=pull_request&branch={quote(branch, safe='')}&status=completed&per_page=100", + paginated=True, + ) + for item in page["workflow_runs"] + ] + else: + runs = [gh(f"repos/{repo}/actions/runs/{selected}")] + for run in runs: + allowed = {"success", "failure", "cancelled"} if authorized is not None else {"success"} + valid = ( + run.get("event") == "pull_request" + and run.get("status") == "completed" + and run.get("conclusion") in allowed + and run.get("head_sha") in shas + and run.get("path", "").split("@", 1)[0] == ".github/workflows/run-sweep.yml" + ) + if not valid: + if selected is not None: + raise ValueError( + "authorized source is no longer eligible; refusing to substitute another run" + ) + continue + artifacts = [ + item + for page in gh( + f"repos/{repo}/actions/runs/{run['id']}/artifacts?per_page=100", paginated=True + ) + for item in page["artifacts"] + ] + if any( + not item["expired"] + and item["name"].startswith(("results_bmk", "eval_results_all", "bmk_agentic_")) + for item in artifacts + ): + return int(run["id"]) + if selected is not None: + raise ValueError( + "authorized source artifacts are unavailable; refusing to substitute another run" + ) + raise ValueError("no eligible successful sweep exists for this pull request") + + +def main() -> None: + parser = argparse.ArgumentParser() + parser.add_argument("--repo", required=True) + parser.add_argument("--pr", type=int, required=True) + parser.add_argument("--branch", required=True) + parser.add_argument("--source-run", type=int) + args = parser.parse_args() + print(select_source(args.repo, args.pr, args.branch, args.source_run)) + + +if __name__ == "__main__": + main() diff --git a/perf-changelog.yaml b/perf-changelog.yaml index de15076d41..b8df115714 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8447,3 +8447,12 @@ - "将 GLM-5.2 MI355X SGLang AgentX 镜像更新至 lmsysorg/sglang-rocm:v0.5.19-rocm720-mi35x-20260917(2026-09-17 ROCm 7.2.0 mi35x 日构建,digest sha256:21c1cc9a…,rocm720-mi35x 系列最新 tag;2026-09-18 构建仅提供 rocm724/rocm10 版本):glm5.2-fp4-mi355x-sglang-agentic-mtp 自 20260916,glm5.2-fp8-mi355x-sglang-agentic-mtp 自 20260908" - "在 glm5.2_fp8_mi355x_sglang_mtp.sh 中改用 --cuda-graph-max-bs-decode 传递 decode graph batch,替换已移除的 --cuda-graph-max-bs 别名,与 MXFP4 姊妹配方一致;其余服务参数、搜索空间与 golden acceptance 不变" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3277 + +- config-keys: + - dsv41flash-fp4-h100-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Migrate the H100 aggregate TP8 vLLM DSV4.1-Flash AgentX lane to an isolated pinned native srt-slurm runtime with prepared Python clients, immutable measurement receipts and a separate real c28 GSM8K evaluation; preserve the image, eight concurrency points and 480-minute exclusive allocation. Power telemetry remains an explicit temporary parity exception." + - "将 H100 聚合 TP8 vLLM DSV4.1-Flash AgentX 路径迁移至独立固定的原生 srt-slurm 运行时,使用预备式 Python 客户端、不可变测量回执及独立真实 c28 GSM8K eval;保留镜像、八个并发点及 480 分钟独占分配。功耗遥测保留明确临时一致性例外。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXX diff --git a/runners/srt-slurm/h100-phase1.yaml b/runners/srt-slurm/h100-phase1.yaml new file mode 100644 index 0000000000..55f3d5558c --- /dev/null +++ b/runners/srt-slurm/h100-phase1.yaml @@ -0,0 +1,11 @@ +cluster: h100-dgxc +default_account: customer +default_partition: hpc-gpu-1 +default_time_limit: '08:00:00' +gpus_per_node: 8 +default_gpu_type: h100 +use_gpus_per_node_directive: true +use_segment_sbatch_directive: false +use_exclusive_sbatch_directive: true +use_het_jobs: false +preflight: true diff --git a/utils/changelog_gate_tests/test_merge_with_reuse.py b/utils/changelog_gate_tests/test_merge_with_reuse.py index 198fdd7d00..586d76ca64 100644 --- a/utils/changelog_gate_tests/test_merge_with_reuse.py +++ b/utils/changelog_gate_tests/test_merge_with_reuse.py @@ -42,12 +42,14 @@ def test_merge_preflight_uses_artifacts_without_requiring_a_sweep_label( sys.exit(73) # Preflight reached the first write; do not perform it. elif args[0] == "api": path = args[1] - if path.endswith("/commits"): - print("tested-sha") + if "/comments?" in path: + print("[[]]") + elif "/commits?" in path: + print(json.dumps([[{"sha": "tested-sha"}]])) elif "/workflows/run-sweep.yml/runs?" in path: - print("123\\ttested-sha") + print(json.dumps([{"workflow_runs": [{"id": 123, "head_sha": "tested-sha", "event": "pull_request", "status": "completed", "conclusion": "success", "path": ".github/workflows/run-sweep.yml"}]}])) elif "/runs/123/artifacts?" in path: - print(os.environ["TEST_REUSE_ARTIFACTS"]) + print(json.dumps([{"artifacts": [{"name": os.environ["TEST_REUSE_ARTIFACTS"], "expired": False}]}])) else: raise AssertionError(args) else: diff --git a/utils/fixtures/native_pilot/client-policy.json b/utils/fixtures/native_pilot/client-policy.json new file mode 100644 index 0000000000..a8d42281b2 --- /dev/null +++ b/utils/fixtures/native_pilot/client-policy.json @@ -0,0 +1,23 @@ +{ + "schema_version": 1, + "golden_curve": "golden_al_distribution/dsv41flash_dspark.yaml", + "golden_model": "deepseek-v4.1-flash", + "thinking_mode": "thinking_on", + "dataset_repository": "semianalysisai/cc-traces-weka-062126", + "dataset_loader": "semianalysis_cc_traces_weka_062126", + "dataset_entries": 393, + "duration_seconds": 3600, + "warmup_requests_per_lane": 10, + "warmup_grace_seconds": 1800, + "trace_idle_gap_cap_seconds": 300, + "live_failed_request_threshold": 0.1, + "failed_request_threshold": 0.1, + "random_seed": 42, + "required_server_metric_prefix": "vllm:", + "eval_task": "gsm8k", + "eval_documents": 1319, + "eval_max_length": 16384, + "eval_max_tokens": 12288, + "eval_minimum_score": 0.9, + "telemetry": "temporary-parity-exception-no-native-power" +} diff --git a/utils/fixtures/native_pilot/golden.yaml b/utils/fixtures/native_pilot/golden.yaml new file mode 100644 index 0000000000..4a2af28fa3 --- /dev/null +++ b/utils/fixtures/native_pilot/golden.yaml @@ -0,0 +1,23 @@ +# Source GitHub Actions run: https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34493175056 +# Draft lengths 6-8 rejected by the image (no measured AL): https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34494319147 +# Acceptance Length (AL) measured with SPEED-Bench Qualitative coding. +# model: deepseek-ai/DeepSeek-V4.1-Flash | image: vllm/vllm-openai:deepseekv41-flash-0909 | TP: 4 +# temperature: 1.0 | output_len: 4096 +# thinking_on chat_template_kwargs: {"thinking":true,"reasoning_effort":"high"} +# thinking_off chat_template_kwargs: {"thinking":false} +# method: dspark | draft_sample_method: probabilistic | rejection_sample_method: block +# enable_adaptive_verification: false | engram cpu_offload: true +# key = num_speculative_tokens; AL includes the target verification token. +deepseek-v4.1-flash: + thinking_off: + 1: 1.88 + 2: 2.62 + 3: 3.27 + 4: 3.73 + 5: 4.07 + thinking_on: + 1: 1.81 + 2: 2.43 + 3: 2.90 + 4: 3.26 + 5: 3.51 diff --git a/utils/fixtures/native_pilot/profile.yaml b/utils/fixtures/native_pilot/profile.yaml new file mode 100644 index 0000000000..55f3d5558c --- /dev/null +++ b/utils/fixtures/native_pilot/profile.yaml @@ -0,0 +1,11 @@ +cluster: h100-dgxc +default_account: customer +default_partition: hpc-gpu-1 +default_time_limit: '08:00:00' +gpus_per_node: 8 +default_gpu_type: h100 +use_gpus_per_node_directive: true +use_segment_sbatch_directive: false +use_exclusive_sbatch_directive: true +use_het_jobs: false +preflight: true diff --git a/utils/fixtures/native_pilot/recipe.yaml b/utils/fixtures/native_pilot/recipe.yaml new file mode 100644 index 0000000000..9c0df8c853 --- /dev/null +++ b/utils/fixtures/native_pilot/recipe.yaml @@ -0,0 +1,54 @@ +schema: 2 +name: inferencex-h100-dsv41flash-agentx +slurm: + account: customer + partition: hpc-gpu-1 + time_limit: '08:00:00' +model: + path: deepseek-ai/DeepSeek-V4.1-Flash + container: example.invalid/pilot:v1 + precision: fp4 +resources: + gpu_type: h100 + gpus_per_node: 8 +engine: vllm +frontend: + type: vllm + enable_multiple_frontends: false +roles: + agg: + nodes: 1 + workers: 1 + gpus: 8 + env: + VLLM_ENGINE_READY_TIMEOUT_S: '3600' + VLLM_USE_RUST_FRONTEND: '1' + VLLM_USE_V2_MODEL_RUNNER: '1' + PYTHONUNBUFFERED: '1' + PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True + HF_HUB_OFFLINE: '1' + HF_DATASETS_OFFLINE: '1' + args: + served-model-name: deepseek-ai/DeepSeek-V4.1-Flash + tensor-parallel-size: 8 + language-model-only: true + tokenizer-mode: deepseek_v41 + tool-call-parser: deepseek_v41 + enable-auto-tool-choice: true + reasoning-parser: deepseek_v41 + engram-config: {cpu_offload: true} + speculative-config: + method: dspark + num_speculative_tokens: 5 + draft_sample_method: probabilistic + rejection_sample_method: block + enable_adaptive_verification: true + max-model-len: 1048576 + max-num-batched-tokens: 4096 + gpu-memory-utilization: 0.92 + disable-uvicorn-access-log: true +health_check: + max_attempts: 360 + interval_seconds: 10 +benchmark: + type: custom diff --git a/utils/fixtures/native_pilot_probe.py b/utils/fixtures/native_pilot_probe.py new file mode 100644 index 0000000000..9d599e6b65 --- /dev/null +++ b/utils/fixtures/native_pilot_probe.py @@ -0,0 +1,93 @@ +"""Inspect an installed native prepared job without running a scheduler or client. + +Only physical node discovery and the final scheduler process boundary are +simulated. Schema loading, mounts, topology, worker commands, benchmark context, +and srun command construction use the installed runtime's real implementations. +""" + +import json +import sys +import threading +from pathlib import Path +from types import SimpleNamespace +from unittest.mock import patch + +import yaml +from srtctl.benchmarks.base import get_runner +from srtctl.cli.do_sweep import SweepOrchestrator +from srtctl.core.config import cluster_config_scope, load_config +from srtctl.core.prepared import load_prepared +from srtctl.core.runtime import Nodes, RuntimeContext + + +def inspect(prepared): + prepared = Path(prepared).resolve() + load_prepared(prepared, verify_runtime=True) + profile = yaml.safe_load((prepared / "profile.yaml").read_text()) + nodes = Nodes( + head="simulated-h100", + bench="simulated-h100", + infra="simulated-h100", + worker=("simulated-h100",), + ) + with ( + cluster_config_scope(profile), + patch.object(Nodes, "from_slurm", return_value=nodes), + patch("srtctl.core.runtime.get_hostname_ip", return_value="127.0.0.1"), + patch("srtctl.core.slurm.get_hostname_ip", return_value="127.0.0.1"), + patch( + "srtctl.cli.mixins.benchmark_stage.get_hostname_ip", + return_value="127.0.0.1", + ), + ): + config = load_config(prepared / "config.yaml", frozen=True) + runtime = RuntimeContext.from_config( + config, "offline-inspection", log_dir_base=prepared.parent / "inspection" + ) + orchestrator = SweepOrchestrator(config, runtime) + processes = orchestrator.backend_processes + server = config.backend.build_worker_command( + processes[0], processes, runtime, frontend_type=config.frontend.type + ) + runner = get_runner(config.benchmark.type) + captured = [] + + def intercept_scheduler(argv, **_kwargs): + captured.append(argv) + return SimpleNamespace(poll=lambda: 0, returncode=0) + + with patch( + "srtctl.core.slurm.subprocess.Popen", side_effect=intercept_scheduler + ): + exit_code = orchestrator._run_benchmark_script( + runner, runtime.log_dir / "inspection.log", threading.Event() + ) + return { + "server_argv": server, + "client_argv": runner.build_command(config, runtime), + "worker_env": config.backend.get_environment_for_mode("agg"), + "client_env": { + **orchestrator._get_benchmark_env(runner), + **runner.get_environment(config, runtime), + }, + "client_env_unset": runner.get_environment_unset(config, runtime), + "client_cwd": runner.get_working_directory(config, runtime), + "mounts": { + str(host): str(container) + for host, container in runtime.container_mounts.items() + }, + "srun_argv": captured, + "exit_code": exit_code, + "processes": [ + {"node": p.node, "mode": p.endpoint_mode, "gpus": sorted(p.gpu_indices)} + for p in processes + ], + "time_limit": config.slurm.time_limit, + "frontend_type": config.frontend.type, + "discovery_services": config.services, + "profiling": config.profiling.enabled, + } + + +if __name__ == "__main__": + print(json.dumps(inspect(sys.argv[1]), default=str)) diff --git a/utils/merge_with_reuse.sh b/utils/merge_with_reuse.sh index df2681a371..50c17a2a61 100755 --- a/utils/merge_with_reuse.sh +++ b/utils/merge_with_reuse.sh @@ -103,31 +103,12 @@ REUSE_INCOMPATIBLE_LABELS="$( [ -z "$REUSE_INCOMPATIBLE_LABELS" ] \ || die "PR #${PR} uses ${REUSE_INCOMPATIBLE_LABELS}, which is not eligible for artifact reuse" -# Fail early unless a successful run with reusable artifacts exists on a -# current PR commit. This excludes reuse-gate-only success runs. -PR_SHAS="$(gh api "repos/${REPO}/pulls/${PR}/commits" --paginate --jq '.[].sha')" -ELIGIBLE_RUN="" -while IFS=$'\t' read -r run_id run_sha; do - grep -qxF "$run_sha" <<<"$PR_SHAS" || continue - artifact_names="$( - gh api "repos/${REPO}/actions/runs/${run_id}/artifacts?per_page=100" \ - --paginate \ - --jq '.artifacts[] | select(.expired == false) | .name' - )" - if grep -Eq '^(results_bmk|eval_results_all|bmk_agentic_)' \ - <<<"$artifact_names"; then - ELIGIBLE_RUN="$run_id" - break - fi -done < <( - gh api \ - "repos/${REPO}/actions/workflows/run-sweep.yml/runs?event=pull_request&branch=${HEAD_BRANCH}&status=completed&per_page=100" \ - --paginate \ - --jq '.workflow_runs[] | select(.conclusion == "success") | [.id, .head_sha] | @tsv' -) -if [ -z "$ELIGIBLE_RUN" ]; then - die "PR #${PR} has no successful reusable run-sweep.yml run on a current commit" -fi +# Preserve an existing explicit maintainer source choice, including failed runs +# accepted via /use. Never let a newer diagnostic sweep silently replace it. +ELIGIBLE_RUN="$( + PYTHONPATH="$SCRIPT_DIR/..${PYTHONPATH:+:$PYTHONPATH}" python3 -m infx.workflows.merge_source \ + --repo "$REPO" --pr "$PR" --branch "$HEAD_BRANCH" +)" log "Posting /reuse-sweep-run ${ELIGIBLE_RUN} on PR #${PR}" gh pr comment "$PR" --repo "$REPO" --body "/reuse-sweep-run ${ELIGIBLE_RUN}" >/dev/null diff --git a/utils/test_merge_source.py b/utils/test_merge_source.py new file mode 100644 index 0000000000..d980675f98 --- /dev/null +++ b/utils/test_merge_source.py @@ -0,0 +1,37 @@ +from infx.workflows import merge_source + + +def test_approved_older_run_survives_newer_diagnostic(monkeypatch): + paths = [] + + def api(path, **kwargs): + paths.append(path) + if "/comments?" in path: + return [ + [ + { + "id": 1, + "created_at": "2026-09-01", + "author_association": "MEMBER", + "body": "/use 10", + } + ] + ] + if "/commits?" in path: + return [[{"sha": "approved"}, {"sha": "newer"}]] + if path.endswith("/runs/10"): + return { + "id": 10, + "head_sha": "approved", + "event": "pull_request", + "status": "completed", + "conclusion": "failure", + "path": ".github/workflows/run-sweep.yml", + } + if "/runs/10/artifacts" in path: + return [{"artifacts": [{"name": "bmk_agentic_result", "expired": False}]}] + raise AssertionError(f"unexpected API request: {path}") + + monkeypatch.setattr(merge_source, "gh", api) + assert merge_source.select_source("owner/repo", 2, "branch") == 10 + assert not any("workflows/run-sweep.yml/runs?" in path for path in paths) diff --git a/utils/test_native_phase1_matrix_evals.py b/utils/test_native_phase1_matrix_evals.py new file mode 100644 index 0000000000..0bba01d690 --- /dev/null +++ b/utils/test_native_phase1_matrix_evals.py @@ -0,0 +1,67 @@ +"""The native pilot eval boundary is narrower than legacy all-evals selection.""" + +import pytest + +from infx.matrix.generate import select_matrix_evals + + +def agentic_row(name, concurrency, execution=None): + row = { + "exp-name": name, + "model": "fixture/model", + "model-prefix": "fixture", + "runner": "cluster:fixture", + "framework": "vllm", + "precision": "fp4", + "scenario-type": "agentic-coding", + "conc": concurrency, + "tp": 8, + "run-eval": False, + } + if execution is not None: + row["execution"] = execution + return row + + +@pytest.mark.parametrize( + "native_concurrencies, expected_native", + [([1, 16, 28], [("native-28", 28)]), ([1, 16], [])], +) +def test_all_evals_keeps_native_c28_only_and_all_legacy_points( + native_concurrencies, expected_native +): + rows = [ + agentic_row( + f"native-{concurrency}", + concurrency, + {"runtime": "srt-slurm", "contract-version": 1}, + ) + for concurrency in native_concurrencies + ] + rows.extend( + agentic_row(f"legacy-{concurrency}", concurrency) for concurrency in (1, 16, 28) + ) + result = select_matrix_evals(rows, mode="all") + assert [(row["exp-name"], row["conc"]) for row in result] == expected_native + [ + ("legacy-1", 1), + ("legacy-16", 16), + ("legacy-28", 28), + ] + assert [ + (row["run-eval"], row["eval-only"], row["eval-framework"]) for row in result + ] == [(True, True, "lm-eval")] * len(result) + + +def test_all_evals_clears_representative_selection_when_native_c28_is_absent(): + result = select_matrix_evals( + [ + agentic_row("native-1", 1, {"runtime": "srt-slurm", "contract-version": 1}), + agentic_row( + "native-16", 16, {"runtime": "srt-slurm", "contract-version": 1} + ), + ], + mode="all", + ) + # Default selection initially chooses c16. Expansion must remove that + # selection rather than submitting an eval the native adapter will reject. + assert result == [] diff --git a/utils/test_native_pilot.py b/utils/test_native_pilot.py new file mode 100644 index 0000000000..cd8e5a11f7 --- /dev/null +++ b/utils/test_native_pilot.py @@ -0,0 +1,535 @@ +"""Behavior at the matrix, recipe, preparation, and scheduler-client boundaries.""" + +import json +import os +import shlex +import shutil +import subprocess +from pathlib import Path + +import pytest + +from infx.srt_slurm.contracts import digest, load_mapping, resolve_reference +from infx.srt_slurm.job import intent_id, parse_job +from infx.srt_slurm.launch import execute, publish_outputs, verify_bundle +from infx.srt_slurm.render import ClientPolicy, PilotSite, render_recipe + +ROOT = Path(__file__).resolve().parents[1] + + +@pytest.fixture +def inputs(tmp_path): + tmp_path = tmp_path.resolve() + root = tmp_path / "checkout" + paths = ( + "benchmarks/srt-slurm/phase1/h100-dsv41flash.yaml", + "runners/srt-slurm/h100-phase1.yaml", + "benchmarks/srt-slurm/phase1/client-policy.json", + "golden_al_distribution/dsv41flash_dspark.yaml", + ) + fixture_names = ("recipe.yaml", "profile.yaml", "client-policy.json", "golden.yaml") + for name, fixture_name in zip(paths, fixture_names, strict=True): + target = root / name + target.parent.mkdir(parents=True, exist_ok=True) + shutil.copyfile(ROOT / "utils/fixtures/native_pilot" / fixture_name, target) + (root / "runtime.json").write_text("{}") + reference = { + "runtime": "srt-slurm", + "contract-version": 1, + "recipe": paths[0], + "profile": paths[1], + "client-policy": paths[2], + "runtime-lock": "runtime.json", + } + row = { + "image": "example.invalid/pilot:v1", + "model": "deepseek-ai/DeepSeek-V4.1-Flash", + "model-prefix": "dsv41flash", + "precision": "fp4", + "framework": "vllm", + "runner": "cluster:h100-dgxc", + "tp": 8, + "pp": 1, + "dcp-size": 1, + "pcp-size": 1, + "ep": 1, + "dp-attn": False, + "spec-decoding": "mtp", + "conc": 1, + "kv-offloading": "none", + "total-cpu-dram-gb": 0, + "duration": 3600, + "exp-name": "pilot", + "scenario-type": "agentic-coding", + "execution": reference, + } + scheduling = {"priority": "-2.1", "queue-token": "explicit-queue", "node-count": 1} + site = PilotSite( + schema_version=1, + cluster="h100-dgxc", + native_python=str(tmp_path / "runtime/python"), + native_source=str(tmp_path / "runtime/src"), + wrapper_python=str(tmp_path / "wrapper/python"), + shared_root=str(tmp_path), + model_snapshot=str(tmp_path / ("d" * 40)), + model_revision="d" * 40, + image={"path": str(tmp_path / "model.sqsh"), "sha256": "e" * 64}, + image_reference=row["image"], + client_sites={"agentx": "/agentx.json", "eval": "/eval.json"}, + mounts={str(tmp_path): str(tmp_path)}, + reader_revision="a" * 40, + collector_revision="b" * 40, + ) + return root, row, scheduling, site + + +def test_mount_aliases_and_workspace_outputs_are_rejected(inputs, tmp_path): + _, _, _, site = inputs + alias = tmp_path / "alias" + target = tmp_path / "actual" + target.mkdir() + alias.symlink_to(target, target_is_directory=True) + with pytest.raises(ValueError, match="canonical paths"): + PilotSite.model_validate( + {**site.model_dump(), "mounts": {str(alias): str(alias)}} + ) + with pytest.raises(ValueError, match="under /workspace"): + PilotSite.model_validate( + {**site.model_dump(), "shared_root": "/workspace/pilot"} + ) + + +def test_interpreter_requires_mounted_base_and_expected_python(inputs, tmp_path): + _, _, _, site = inputs + shared = Path(site.shared_root) + identity = { + "python_version": "3.12.9", + "python_paths": { + "executable": str(shared / "venv/bin/python"), + "executable_resolved": "/unmounted/python/bin/python3.12", + "prefix": str(shared / "venv"), + "base_prefix": "/unmounted/python", + }, + } + with pytest.raises(ValueError, match="not mounted"): + site.require_interpreter(identity, python_minor="3.12") + identity["python_paths"].update( + executable_resolved=str(shared / "python/bin/python3.12"), + base_prefix=str(shared / "python"), + ) + with pytest.raises(ValueError, match="must use Python 3.11"): + site.require_interpreter(identity, python_minor="3.11") + link = shared / "external-cache" + link.symlink_to("/unmounted/cache", target_is_directory=True) + with pytest.raises(ValueError, match="not mounted"): + site.require_visible(str(link / "data")) + + +def test_policy_symlink_cannot_read_outside_checkout(inputs, tmp_path): + root, row, _, _ = inputs + outside = tmp_path / "not-a-policy" + outside.write_text("[malformed yaml") + policy = root / row["execution"]["client-policy"] + policy.unlink() + policy.symlink_to(outside) + with pytest.raises(ValueError, match="escapes checkout"): + resolve_reference(row["execution"], root) + + +def test_golden_resource_and_scheduling_have_distinct_identities(inputs): + root, row, scheduling, _ = inputs + first = parse_job(row, root, scheduling) + second = parse_job( + row, root, {**scheduling, "priority": "999", "queue-token": "other"} + ) + assert first.point_id == second.point_id + assert intent_id("owner/repo", "10", "1", first.point_id) != intent_id( + "owner/repo", "10", "2", first.point_id + ) + curve = root / "golden_al_distribution/dsv41flash_dspark.yaml" + curve.write_text(curve.read_text().replace("5: 3.51", "5: 3.52")) + assert parse_job(row, root, scheduling).point_id != first.point_id + with pytest.raises(ValueError, match="changed after matrix"): + resolve_reference(first.row.execution.model_dump(by_alias=True), root) + + +@pytest.mark.parametrize( + "concurrency,sequences,capture", + [ + (1, 2, 16), + (2, 4, 32), + (4, 8, 64), + (8, 16, 128), + (16, 32, 256), + (20, 40, 256), + (24, 48, 512), + (28, 56, 512), + ], +) +def test_h100_recipe_preserves_real_serving_parameters( + inputs, concurrency, sequences, capture +): + root, row, scheduling, site = inputs + job = parse_job({**row, "conc": concurrency}, root, scheduling) + policy = ClientPolicy.model_validate( + load_mapping(root / row["execution"]["client-policy"]) + ) + recipe, profile = render_recipe( + job, root, site, policy, root / "spec.json", root / "outputs" + ) + args = recipe["roles"]["agg"]["args"] + assert args["max-num-seqs"] == sequences + assert args["max-cudagraph-capture-size"] == capture + assert args["max-model-len"] == 1048576 + assert args["max-num-batched-tokens"] == 4096 + assert args["speculative-config"]["synthetic_acceptance_length"] == 3.51 + assert args["speculative-config"]["enable_adaptive_verification"] is False + assert profile["default_time_limit"] == "08:00:00" + assert profile["use_exclusive_sbatch_directive"] is True + assert recipe["benchmark"]["argv"][-4:] == [ + "--spec", + str(root / "spec.json"), + "--artifact-root", + str(root / "outputs"), + ] + + +def test_real_eval_has_no_synthetic_acceptance(inputs): + root, row, scheduling, site = inputs + job = parse_job( + { + **row, + "conc": 28, + "run-eval": True, + "eval-only": True, + "eval-framework": "lm-eval", + }, + root, + scheduling, + ) + policy = ClientPolicy.model_validate( + load_mapping(root / row["execution"]["client-policy"]) + ) + recipe, _ = render_recipe( + job, root, site, policy, root / "spec.json", root / "outputs" + ) + spec = recipe["roles"]["agg"]["args"]["speculative-config"] + assert spec == { + "method": "dspark", + "num_speculative_tokens": 5, + "draft_sample_method": "probabilistic", + "rejection_sample_method": "block", + "enable_adaptive_verification": True, + } + + +@pytest.mark.parametrize( + "changes", + [{"tp": 4}, {"conc": 32}, {"run-eval": True}, {"rogue": 1}, {"duration": 1200}], +) +def test_unqualified_inputs_are_rejected_before_runtime(inputs, changes): + root, row, scheduling, _ = inputs + with pytest.raises(ValueError): + parse_job({**row, **changes}, root, scheduling) + + +def test_missing_queue_demand_and_duplicate_yaml_fail(inputs): + root, row, scheduling, _ = inputs + del scheduling["node-count"] + with pytest.raises(ValueError): + parse_job(row, root, scheduling) + path = root / "duplicate.yaml" + path.write_text("nodes: 1\nnodes: 2\n") + with pytest.raises(ValueError, match="Duplicate YAML key"): + load_mapping(path) + + +def test_bundle_mutation_fails_before_scheduler(inputs, tmp_path, monkeypatch): + from infx.srt_slurm import launch + + file = tmp_path / "prepared.json" + file.write_text('{"approved":true}') + from infx.srt_slurm.job import file_digest + + bundle = {"files": {str(file): file_digest(file)}} + bundle["bundle_digest"] = digest(bundle) + verify_bundle(bundle) + file.write_text('{"approved":false}') + calls = [] + monkeypatch.setattr(launch, "native", lambda *args, **kwargs: calls.append(args)) + with pytest.raises(ValueError, match="prepared input changed"): + execute(bundle, tmp_path) + assert calls == [] + + +def test_client_descendants_cannot_be_published_as_success(inputs, tmp_path): + root, row, scheduling, _ = inputs + directory = tmp_path / "run" + audit = directory / "client-output/diagnostics/client-audit.json" + audit.parent.mkdir(parents=True) + audit.write_text( + json.dumps( + { + "errors": [], + "status": { + "returncode": 0, + "cancelled_by_signal": None, + "timed_out": False, + "orphaned_descendants": True, + }, + } + ) + ) + job = parse_job(row, root, scheduling) + bundle = { + "directory": str(directory), + "point_id": job.point_id, + "job": job.model_dump(by_alias=True), + } + with pytest.raises(ValueError, match="closed result"): + publish_outputs(bundle, {}, tmp_path / "publish") + assert not (tmp_path / "publish").exists() + + +def test_native_unknown_acceptance_is_not_retried(inputs, tmp_path, monkeypatch): + from infx.srt_slurm import launch + + _, _, _, site = inputs + receipt_path = tmp_path / "receipt.json" + receipt_path.write_text("{}") + bundle = { + "files": {}, + "site": site.model_dump(), + "execution_id": "unique", + "prepared": {"prepared_dir": str(tmp_path)}, + } + bundle["bundle_digest"] = digest(bundle) + calls = [] + + def external(_site, command, *args, **kwargs): + calls.append(command) + if command == "intent-path": + return {"state": "intent", "receipt_path": str(receipt_path)} + if command == "submit-prepared": + raise launch.NativeCommandError( + {"state": "unknown", "receipt_path": str(receipt_path)}, "unknown" + ) + return {"state": "unknown"} + + monkeypatch.setattr(launch, "native", external) + monkeypatch.setattr(launch, "verify_execution_clients", lambda *_args: None) + monkeypatch.setattr(launch, "copy_diagnostics", lambda *_args, **_kwargs: None) + with pytest.raises(RuntimeError, match="intent stays fenced"): + execute(bundle, tmp_path, reconcile_timeout=0) + assert calls == ["intent-path", "submit-prepared", "reconcile"] + + +def test_installed_native_runtime_prepares_and_renders_the_entire_pilot( + inputs, tmp_path +): + """Exercise the real cross-repository boundary; never allocate or run clients. + + Opt in with INFX_NATIVE_PHASE1_PYTHON and INFX_NATIVE_PHASE1_SOURCE pointing + to an installed, source-matching native runtime and its clean source tree. + """ + import yaml + + from infx.srt_slurm.launch import native + + python = os.environ.get("INFX_NATIVE_PHASE1_PYTHON") + source = os.environ.get("INFX_NATIVE_PHASE1_SOURCE") + if not python or not source: + pytest.skip("requires explicitly provisioned native runtime Python and source") + root, row, scheduling, original_site = inputs + root = root.resolve() + shared = tmp_path.resolve() + model = shared / ("d" * 40) + model.mkdir() + image = shared / "model.sqsh" + image.write_text("Inert test placeholder: no container is started.\n") + site = PilotSite.model_validate( + { + **original_site.model_dump(), + "native_python": python, + "native_source": source, + "wrapper_python": str(shared / "wrapper/python"), + "shared_root": str(shared), + "model_snapshot": str(model), + "image": {"path": str(image), "sha256": "e" * 64}, + "mounts": {str(shared): str(shared)}, + } + ) + policy = ClientPolicy.model_validate( + load_mapping(root / row["execution"]["client-policy"]) + ) + probe = ROOT / "utils/fixtures/native_pilot_probe.py" + inspections = [] + cases = [ + (False, 1, 2, 16), + (False, 2, 4, 32), + (False, 4, 8, 64), + (False, 8, 16, 128), + (False, 16, 32, 256), + (False, 20, 40, 256), + (False, 24, 48, 512), + (False, 28, 56, 512), + (True, 28, 56, 512), + ] + for evaluation, concurrency, sequences, capture in cases: + kind = "eval" if evaluation else "agentx" + current_row = {**row, "conc": concurrency} + if evaluation: + current_row.update( + {"run-eval": True, "eval-only": True, "eval-framework": "lm-eval"} + ) + job = parse_job(current_row, root, scheduling) + point = shared / f"{kind}-c{concurrency}" + point.mkdir() + spec_path = point / "client.json" + output = point / "client-output" + recipe, profile = render_recipe(job, root, site, policy, spec_path, output) + # This is the guarded argv installed by launch.prepare; no guard or + # benchmark executes in this test. + recipe["benchmark"]["argv"][3:4] = [ + "infx.srt_slurm.client_guard", + "--bundle", + str(point / "bundle.json"), + "--client", + kind, + ] + for name, data in (("recipe.yaml", recipe), ("profile.yaml", profile)): + (point / name).write_text(yaml.safe_dump(data, sort_keys=False)) + prepared = native( + site, + "prepare", + "--recipe", + str(point / "recipe.yaml"), + "--profile", + str(point / "profile.yaml"), + "--output", + str(point / "native"), + "--expected-nodes", + "1", + "--runtime-python", + python, + ) + assert prepared["state"] == "prepared" + assert prepared["resources"] == { + "nodes": 1, + "gpus_per_node": 8, + "serving_gpus": 8, + "workers": 1, + "cardinality": 1, + } + result = subprocess.run( + [python, "-I", str(probe), prepared["prepared_dir"]], + capture_output=True, + text=True, + timeout=60, + check=False, + ) + assert result.returncode == 0, result.stderr + inspected = json.loads(result.stdout) + inspections.append({"kind": kind, "concurrency": concurrency, **inspected}) + argv = inspected["server_argv"] + + def value(flag, command=argv): + assert command.count(flag) == 1 + return command[command.index(flag) + 1] + + assert argv[:3] == ["vllm", "serve", str(model)] + assert value("--tensor-parallel-size") == "8" + assert value("--device-ids") == "0,1,2,3,4,5,6,7" + assert value("--max-num-seqs") == str(sequences) + assert value("--max-cudagraph-capture-size") == str(capture) + assert value("--max-model-len") == "1048576" + assert value("--max-num-batched-tokens") == "4096" + assert value("--gpu-memory-utilization") == "0.92" + assert value("--served-model-name") == row["model"] + assert value("--tokenizer-mode") == "deepseek_v41" + assert value("--tool-call-parser") == "deepseek_v41" + assert value("--reasoning-parser") == "deepseek_v41" + assert "--language-model-only" in argv + assert "--enable-auto-tool-choice" in argv + assert "--disable-uvicorn-access-log" in argv + assert json.loads(value("--engram-config")) == {"cpu_offload": True} + speculative = json.loads(value("--speculative-config")) + assert speculative == { + "method": "dspark", + "num_speculative_tokens": 5, + "draft_sample_method": "probabilistic", + "rejection_sample_method": "block" if evaluation else "synthetic", + "enable_adaptive_verification": evaluation, + **({} if evaluation else {"synthetic_acceptance_length": 3.51}), + } + assert inspected["processes"] == [ + {"node": "simulated-h100", "mode": "agg", "gpus": list(range(8))} + ] + assert inspected["frontend_type"] == "vllm" + assert inspected["discovery_services"] == [] + assert inspected["profiling"] is False + assert inspected["worker_env"] == recipe["roles"]["agg"]["env"] + assert inspected["client_argv"] == recipe["benchmark"]["argv"] + assert inspected["client_env_unset"] == [ + "PYTHONPATH", + "PYTHONHOME", + "BASH_ENV", + "ENV", + ] + assert inspected["client_cwd"] == str(point) + environment = inspected["client_env"] + assert environment["SRT_ENDPOINT"] == f"http://127.0.0.1:{value('--port')}" + assert ( + environment["AIPERF_SERVER_METRICS_URLS"] + == environment["SRT_ENDPOINT"] + "/metrics" + ) + assert environment["SRT_MODEL_NAME"] == row["model"] + assert environment["SRT_LOG_DIR"] == "/logs" + assert ( + environment["HF_HUB_OFFLINE"] == environment["HF_DATASETS_OFFLINE"] == "1" + ) + assert inspected["mounts"][str(shared)] == str(shared) + assert inspected["exit_code"] == 0 + assert len(inspected["srun_argv"]) == 1 + srun = inspected["srun_argv"][0] + assert srun[0] == "srun" + assert srun[srun.index("--container-image") + 1] == str(image) + assert f"{shared}:{shared}" in srun[srun.index("--container-mounts") + 1].split( + "," + ) + assert srun[-3:-1] == ["bash", "-c"] + assert ( + shlex.split(srun[-1].rsplit(" && exec ", 1)[1]) == inspected["client_argv"] + ) + assert inspected["time_limit"] == "08:00:00" + directives = {} + for line in (point / "native/job.slurm").read_text().splitlines(): + if line.startswith("#SBATCH --"): + key, _, val = line.removeprefix("#SBATCH --").partition("=") + assert key not in directives + directives[key] = val + assert { + name: directives[name] + for name in ("nodes", "ntasks", "gpus-per-node", "time", "exclusive") + } == { + "nodes": "1", + "ntasks": "1", + "gpus-per-node": "8", + "time": "08:00:00", + "exclusive": "", + } + (shared / "cross-repo-inspection.json").write_text( + json.dumps(inspections, indent=2) + ) + frozen = point / "native/config.yaml" + frozen.chmod(0o644) + frozen.write_text(frozen.read_text() + "# changed after preparation\n") + tampered = subprocess.run( + [python, "-I", str(probe), str(point / "native")], + capture_output=True, + text=True, + timeout=60, + check=False, + ) + assert tampered.returncode != 0 + assert "Prepared input changed: config.yaml" in tampered.stderr From 8cc15c43d16ec08c7d2fb6419e65d5308dcf50c6 Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sat, 19 Sep 2026 19:25:32 -0400 Subject: [PATCH 03/16] chore: link Phase 1 qualification to PR 3299 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 中文:仅在新增 changelog 尾部记录实际 PR #3299,保留全部历史字节。 --- perf-changelog.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/perf-changelog.yaml b/perf-changelog.yaml index b8df115714..e136cfc6db 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8455,4 +8455,4 @@ description: - "Migrate the H100 aggregate TP8 vLLM DSV4.1-Flash AgentX lane to an isolated pinned native srt-slurm runtime with prepared Python clients, immutable measurement receipts and a separate real c28 GSM8K evaluation; preserve the image, eight concurrency points and 480-minute exclusive allocation. Power telemetry remains an explicit temporary parity exception." - "将 H100 聚合 TP8 vLLM DSV4.1-Flash AgentX 路径迁移至独立固定的原生 srt-slurm 运行时,使用预备式 Python 客户端、不可变测量回执及独立真实 c28 GSM8K eval;保留镜像、八个并发点及 480 分钟独占分配。功耗遥测保留明确临时一致性例外。" - pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/XXX + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3299 From a12eb850b0c10a86fd6aa228e0b08822873860b0 Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sat, 19 Sep 2026 19:35:09 -0400 Subject: [PATCH 04/16] fix: bind tokenizer snapshots and verify the native runtime in CI MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 中文:在分配前验证离线模型 refs/main、资源绑定及规范 snapshot 路径与服务端完全一致;保留 tokenizer 名称兼容性。新增 Linux CI 任务,核实运行时 pin、依赖锁与 NVIDIA 版本谱系,在独立非 editable 环境中运行原生单元测试及九点已安装运行时边界检查,并同步双语文档。 --- .github/workflows/ci.yml | 140 ++++++++++++++++++++ docs/srt-slurm-phase1.md | 2 +- docs/srt-slurm-phase1_zh.md | 2 +- docs/testing.md | 6 +- docs/testing_zh.md | 6 +- infx/benchmarks/common.py | 49 +++++-- infx/srt_slurm/launch.py | 13 +- utils/test_model_cache_binding.py | 209 ++++++++++++++++++++++++++++++ 8 files changed, 409 insertions(+), 18 deletions(-) create mode 100644 utils/test_model_cache_binding.py diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index b8e7cb0d67..ea7910fffc 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -15,6 +15,9 @@ on: - 'infx/ruff.toml' - '**/pytest.ini' - 'utils/srt-slurm' + - 'benchmarks/srt-slurm/phase1/**' + - 'runners/srt-slurm/h100-phase1.yaml' + - 'utils/fixtures/native_pilot/**' push: branches: [main] paths: *python-paths @@ -82,3 +85,140 @@ jobs: python -c "import torch; assert torch.version.cuda is None and torch.version.hip is None" # SRT is initialized for our connector tests; its upstream suite runs in its own CI. python -m pytest utils/ runners/ experimental/CollectiveX/tests/ experimental/operatorx/tests/ --ignore=utils/srt-slurm -n 4 + + native-pilot: + name: Native pilot contract + runs-on: ubuntu-latest + timeout-minutes: 20 + steps: + - name: Checkout code + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + fetch-depth: 1 + persist-credentials: false + + - name: Validate native runtime lock + id: native-lock + shell: python + run: | + import json + import os + import re + from pathlib import Path + + lock = json.loads(Path("benchmarks/srt-slurm/phase1/runtime-lock.json").read_text()) + if type(lock.get("schema_version")) is not int or lock["schema_version"] != 1: + raise ValueError("Unsupported native runtime lock schema") + if lock.get("repository") != "https://github.com/SemiAnalysisAI/srt-slurm.git": + raise ValueError("Native runtime repository is not allowlisted") + for field, length in (("revision", 40), ("uv_lock_sha256", 64)): + value = lock.get(field) + if not isinstance(value, str) or re.fullmatch(r"[0-9a-f]{%d}" % length, value) is None: + raise ValueError(f"Invalid native runtime {field}") + with Path(os.environ["GITHUB_OUTPUT"]).open("a") as output: + output.write("repository=SemiAnalysisAI/srt-slurm\n") + output.write(f"revision={lock['revision']}\n") + + - name: Checkout pinned native runtime + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + repository: ${{ steps.native-lock.outputs.repository }} + ref: ${{ steps.native-lock.outputs.revision }} + path: .native-phase1 + fetch-depth: 0 + persist-credentials: false + + - name: Verify native source and version lineage + shell: python + run: | + import hashlib + import json + import subprocess + from pathlib import Path + + source = Path(".native-phase1").resolve() + lock = json.loads(Path("benchmarks/srt-slurm/phase1/runtime-lock.json").read_text()) + def git(*args): + return subprocess.run( + ["git", "-C", str(source), *args], check=True, + capture_output=True, text=True, timeout=120, + ).stdout.strip() + if git("rev-parse", "HEAD") != lock["revision"]: + raise ValueError("Native checkout differs from its pinned revision") + if hashlib.sha256((source / "uv.lock").read_bytes()).hexdigest() != lock["uv_lock_sha256"]: + raise ValueError("Native dependency lock differs from its pinned digest") + # hatch-vcs needs the original NVIDIA release tag, which the fork may not carry. + git("fetch", "--no-tags", "--force", "https://github.com/NVIDIA/srt-slurm.git", + "refs/tags/v2.2.1:refs/tags/v2.2.1") + upstream = "984180e5b8755aef85e9995048b5a16cb5336bce" + if git("rev-parse", "refs/tags/v2.2.1^{commit}") != upstream: + raise ValueError("NVIDIA v2.2.1 tag no longer matches the reviewed lineage") + git("merge-base", "--is-ancestor", upstream, "HEAD") + if git("describe", "--tags", "--match", "v[0-9]*", "--abbrev=0") != "v2.2.1": + raise ValueError("Unexpected native hatch-vcs release lineage") + if git("status", "--porcelain", "--untracked-files=all"): + raise ValueError("Native source checkout must remain clean") + + - name: Set up uv + uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1 + with: + cache-suffix: Native-pilot-contract + cache-dependency-glob: | + uv.lock + .native-phase1/uv.lock + + - name: Prepare independent locked test environments + shell: python + run: | + import os + import subprocess + from pathlib import Path + + workspace = Path(os.environ["GITHUB_WORKSPACE"]) + native_env = Path(os.environ["RUNNER_TEMP"]) / "native-phase1-venv" + subprocess.run( + ["uv", "sync", "--locked", "--all-extras", "--group", "test", + "--no-editable", "--python", "3.12"], + cwd=workspace, check=True, timeout=600, + env={**os.environ, "UV_PROJECT_ENVIRONMENT": str(workspace / ".venv")}, + ) + subprocess.run( + ["uv", "sync", "--frozen", "--dev", "--no-editable", "--python", "3.12"], + cwd=workspace / ".native-phase1", check=True, timeout=600, + env={**os.environ, "UV_PROJECT_ENVIRONMENT": str(native_env)}, + ) + + - name: Run pinned native Linux tests + shell: python + run: | + import os + import subprocess + from pathlib import Path + + python = Path(os.environ["RUNNER_TEMP"]) / "native-phase1-venv/bin/python" + source = Path(os.environ["GITHUB_WORKSPACE"]) / ".native-phase1" + subprocess.run( + [str(python), "-m", "pytest", "tests/", "-m", "not integration", "-q"], + cwd=source, check=True, timeout=600, + env={**os.environ, "PYTHONPATH": str(source / "src"), + "PATH": str(python.parent) + os.pathsep + os.environ["PATH"]}, + ) + + - name: Run nine-point installed native boundary test + shell: python + run: | + import os + import subprocess + from pathlib import Path + + workspace = Path(os.environ["GITHUB_WORKSPACE"]) + native_env = Path(os.environ["RUNNER_TEMP"]) / "native-phase1-venv" + subprocess.run( + [str(workspace / ".venv/bin/python"), "-I", "-m", "pytest", "-q", + "utils/test_native_pilot.py::test_installed_native_runtime_prepares_and_renders_the_entire_pilot"], + cwd=workspace, check=True, timeout=180, + env={**os.environ, + "PATH": str(workspace / ".venv/bin") + os.pathsep + os.environ["PATH"], + "INFX_NATIVE_PHASE1_PYTHON": str(native_env / "bin/python"), + "INFX_NATIVE_PHASE1_SOURCE": str(workspace / ".native-phase1")}, + ) diff --git a/docs/srt-slurm-phase1.md b/docs/srt-slurm-phase1.md index 4076e3c184..709797196a 100644 --- a/docs/srt-slurm-phase1.md +++ b/docs/srt-slurm-phase1.md @@ -61,7 +61,7 @@ Provision on shared Linux storage visible to the H100 login host and compute con 1. Install the pinned native source using its committed `uv.lock` and a noneditable environment (`uv sync --frozen --no-editable --no-dev --python 3.12`). Keep that source checkout clean. Preserve the hashed Linux-built wheel and its build-tool constraints: `uv.lock` freezes runtime dependencies but does not pin the upstream Hatch build dependencies. If rebuilding, fetch and verify NVIDIA’s `v2.2.1` tag at `984180e5b8755aef85e9995048b5a16cb5336bce` to retain the same hatch-vcs version lineage. 2. Install a noneditable InferenceX wheel from the exact measured checkout into a shared Python 3.12 environment. Its installed package bytes are compared against the checkout before allocation. 3. Materialize separate client environments and retain their resolved package artifacts/locks. AgentX must come from `754356e9a39acc6cc6afb242d123bb57c3fb6f75`; lm-eval must come from `b315ef3b05176acc9732bb7fdec116abe1ecc476`. Editable and wrong-source installations are rejected. Preparation captures every installed distribution, not just the named entry point. -4. Materialize the complete model/tokenizer snapshot, exact `semianalysisai/cc-traces-weka-062126` snapshot and GSM8K cache. Capture their real revisions; do not invent or substitute a revision. Prepare the unchanged serving image as a verified squash file and record its provenance/hash. +4. Materialize the complete model/tokenizer snapshot, exact `semianalysisai/cc-traces-weka-062126` snapshot and GSM8K cache. Capture their real revisions; do not invent or substitute a revision. The client’s offline model `refs/main` and snapshot files must be bound assets, and its resolved model snapshot must be the exact canonical serving snapshot. This preserves nominal tokenizer names without permitting a different cached revision. Prepare the unchanged serving image as a verified squash file and record its provenance/hash. 5. Write one `ClientSite` JSON for AgentX and one for eval. These explicitly provide the interpreter, distributions, offline cache environment, environment removals, asset roots/files, model snapshot, timeout and termination grace. `RuntimeSpec` rejects credentials; execution strips ambient credentials and unqualified AIPerf overrides. The packaged task and 1,319 independent document hashes are included in installed wheels. 6. Write the `PilotSite` JSON with these two client-site paths, source/interpreter/model/image paths, mounts and actual deployed reader/collector revisions. The Pydantic models in [`render.py`](../infx/srt_slurm/render.py) and [`prepare.py`](../infx/benchmarks/prepare.py) are the exact schemas. diff --git a/docs/srt-slurm-phase1_zh.md b/docs/srt-slurm-phase1_zh.md index 072ca0d729..6399192239 100644 --- a/docs/srt-slurm-phase1_zh.md +++ b/docs/srt-slurm-phase1_zh.md @@ -61,7 +61,7 @@ flowchart TD 1. 使用原生源代码提交中的 `uv.lock` 安装非 editable 环境:`uv sync --frozen --no-editable --no-dev --python 3.12`。保持该源码 checkout 干净。保留带哈希的 Linux wheel 及构建工具约束:`uv.lock` 固定运行依赖,但未固定上游 Hatch 构建依赖。重新构建时,获取并核实 NVIDIA 的 `v2.2.1` tag 指向 `984180e5b8755aef85e9995048b5a16cb5336bce`,保留相同 hatch-vcs 版本谱系。 2. 从实际测量 checkout 构建并安装非 editable InferenceX wheel,使用共享 Python 3.12 环境。分配前逐文件比较已安装包与 checkout。 3. 准备独立客户端环境并保留实际解析的包产物与锁。AgentX 必须来自 `754356e9a39acc6cc6afb242d123bb57c3fb6f75`;lm-eval 必须来自 `b315ef3b05176acc9732bb7fdec116abe1ecc476`。拒绝 editable 或错误来源。准备阶段记录所有已安装 distribution,而非仅入口包。 -4. 完整准备模型/tokenizer snapshot、准确的 `semianalysisai/cc-traces-weka-062126` snapshot 与 GSM8K 缓存。记录真实 revision,不编造或替换。准备原样服务镜像的已校验 squash 文件,记录来源和哈希。 +4. 完整准备模型/tokenizer snapshot、准确的 `semianalysisai/cc-traces-weka-062126` snapshot 与 GSM8K 缓存。记录真实 revision,不编造或替换。准备原样服务镜像的已校验 squash 文件,记录来源和哈希。客户端离线模型的 `refs/main` 与 snapshot 文件必须纳入资源绑定,解析后的模型 snapshot 必须与服务端的规范路径完全一致,保留 tokenizer 的模型名称同时禁止解析到其他缓存版本。 5. 分别编写 AgentX、eval 的 `ClientSite` JSON:解释器、distribution、离线缓存环境、移除变量、资源根目录/文件、模型 snapshot、超时与终止宽限。`RuntimeSpec` 拒绝凭证;执行时移除继承凭证及未验收的 AIPerf 覆盖项。wheel 包含 eval task 与 1,319 个独立文档哈希。 6. 编写 `PilotSite` JSON,包含两个客户端配置路径、源码/解释器/模型/镜像路径、挂载以及实际部署的 reader/collector revision。准确 schema 见 [`render.py`](../infx/srt_slurm/render.py) 与 [`prepare.py`](../infx/benchmarks/prepare.py)。 diff --git a/docs/testing.md b/docs/testing.md index 131ca1fd49..2d713efb44 100644 --- a/docs/testing.md +++ b/docs/testing.md @@ -31,9 +31,11 @@ These sources outrank this guide when behavior changes. Update the English page ## Testing layers -[`CI`](../.github/workflows/ci.yml) runs **Lint** and **Tests** in parallel for PRs (including forks) and pushes to `main` that change Python files, `.github/scripts/` helpers, `ci.yml`, `pyproject.toml`, `uv.lock`, `.python-version`, MCP configuration, Ruff configuration, or `pytest.ini`. [`Workflow security`](../.github/workflows/zizmor.yml) runs **Zizmor** for changes to workflows, action definitions, Dependabot, pre-commit, or zizmor configuration. Python-only changes do not trigger Zizmor; other workflow-only changes do not trigger Lint or Tests. Editing `ci.yml` triggers all three jobs. Each workflow can be dispatched manually. Changes only to other docs, shell scripts, or benchmark YAML do not trigger either workflow; run the applicable checks locally or dispatch them manually. +[`CI`](../.github/workflows/ci.yml) runs **Lint**, **Tests** and **Native pilot contract** in parallel for PRs (including forks) and pushes to `main` that change Python files, `.github/scripts/` helpers, `ci.yml`, `pyproject.toml`, `uv.lock`, `.python-version`, MCP configuration, Ruff configuration, or `pytest.ini`; Phase 1 runtime/recipe/profile/fixture changes also trigger it. [`Workflow security`](../.github/workflows/zizmor.yml) runs **Zizmor** for changes to workflows, action definitions, Dependabot, pre-commit, or zizmor configuration. Python-only changes do not trigger Zizmor; other workflow-only changes do not trigger Lint or Tests. Editing `ci.yml` triggers both workflows. Each workflow can be dispatched manually. Changes only to other docs, shell scripts, or other benchmark YAML do not trigger either workflow; run the applicable checks locally or dispatch them manually. -Tests runs every suite under `utils/`, `runners/`, and `experimental/CollectiveX/tests/` with four pytest workers, plus MCP compatibility. New tests in those directories are discovered automatically. CI installs `infx` as a wheel with `uv sync --locked --all-extras --group test --no-editable`, using Python 3.12 and CPU-only PyTorch. A failing job does not cancel the other; a newer PR update cancels the superseded CI run. Branch pushes without a PR no longer start a separate changelog-test run. +Tests runs every suite under `utils/`, `runners/`, and `experimental/CollectiveX/tests/` and `experimental/operatorx/tests/` with four pytest workers, plus MCP compatibility. New tests in those directories are discovered automatically. CI installs `infx` as a wheel with `uv sync --locked --all-extras --group test --no-editable`, using Python 3.12 and CPU-only PyTorch. A failing job does not cancel the other; a newer PR update cancels the superseded CI run. Branch pushes without a PR no longer start a separate changelog-test run. + +Native pilot contract validates the Phase 1 runtime repository, commit, dependency-lock digest and NVIDIA version lineage, then provisions a separate noneditable Python 3.12 environment. It runs the pinned native Linux unit suite and the installed-runtime boundary test for all eight throughput points plus c28 eval. The latter exercises preparation, command construction and mounts without allocating Slurm or starting a model. Hardware qualification remains the full PR sweep and real eval. | Layer | What it can prove | What it cannot prove | | --- | --- | --- | diff --git a/docs/testing_zh.md b/docs/testing_zh.md index d6173ab3c7..e243c1a9b2 100644 --- a/docs/testing_zh.md +++ b/docs/testing_zh.md @@ -31,9 +31,11 @@ ## 测试层级 -[`CI`](../.github/workflows/ci.yml) 在 PR(包括 fork)或向 `main` 的推送修改 Python 文件、`.github/scripts/` 辅助脚本、`ci.yml`、`pyproject.toml`、`uv.lock`、`.python-version`、MCP 配置、Ruff 配置或 `pytest.ini` 时,并行运行 **Lint** 和 **Tests**。[`Workflow security`](../.github/workflows/zizmor.yml) 在工作流、action 定义、Dependabot、pre-commit 或 zizmor 配置变更时运行 **Zizmor**。仅修改 Python 文件不会触发 Zizmor;仅修改其他工作流不会触发 Lint 或 Tests。修改 `ci.yml` 会触发全部三项任务。两个工作流均可手动分发。仅修改其他文档、Shell 脚本或基准测试 YAML 不会触发这两个工作流;请在本地执行相应检查,或手动分发。 +[`CI`](../.github/workflows/ci.yml) 在 PR(包括 fork)或向 `main` 的推送修改 Python 文件、`.github/scripts/` 辅助脚本、`ci.yml`、`pyproject.toml`、`uv.lock`、`.python-version`、MCP 配置、Ruff 配置或 `pytest.ini` 时,并行运行 **Lint**、**Tests** 和 **Native pilot contract**。阶段 1 的 runtime、recipe、profile 与 fixture 变更也会触发 CI。[`Workflow security`](../.github/workflows/zizmor.yml) 在工作流、action 定义、Dependabot、pre-commit 或 zizmor 配置变更时运行 **Zizmor**。仅修改 Python 文件不会触发 Zizmor;仅修改其他工作流不会触发 Lint 或 Tests。修改 `ci.yml` 会触发 CI 与 Workflow security 两个工作流。两个工作流均可手动分发。仅修改其他文档、Shell 脚本或其他基准测试 YAML 不会触发这两个工作流;请在本地执行相应检查,或手动分发。 -Tests 使用四个 pytest worker 运行 `utils/`、`runners/` 和 `experimental/CollectiveX/tests/` 下的全部测试,并检查 MCP 兼容性。这些目录中的新增测试会自动发现。CI 通过 `uv sync --locked --all-extras --group test --no-editable` 将 `infx` 安装为 wheel,使用 Python 3.12 和仅支持 CPU 的 PyTorch。一项任务失败不会取消另一项;PR 更新会取消旧提交的 CI。尚未创建 PR 的分支推送不再单独触发变更日志测试。 +Tests 使用四个 pytest worker 运行 `utils/`、`runners/` 、`experimental/CollectiveX/tests/` 和 `experimental/operatorx/tests/` 下的全部测试,并检查 MCP 兼容性。这些目录中的新增测试会自动发现。CI 通过 `uv sync --locked --all-extras --group test --no-editable` 将 `infx` 安装为 wheel,使用 Python 3.12 和仅支持 CPU 的 PyTorch。一项任务失败不会取消另一项;PR 更新会取消旧提交的 CI。尚未创建 PR 的分支推送不再单独触发变更日志测试。 + +Native pilot contract 验证阶段 1 的运行时仓库、提交、依赖锁摘要与 NVIDIA 版本谱系,然后准备独立的非 editable Python 3.12 环境。该任务运行固定原生版本的 Linux 单元测试,以及覆盖八个吞吐点和 c28 eval 的已安装运行时边界测试。后者实际执行准备、命令构建与挂载解析,但不分配 Slurm 资源或启动模型。硬件验收仍须完成完整 PR sweep 与真实 eval。 | 层级 | 能够证明 | 不能证明 | | --- | --- | --- | diff --git a/infx/benchmarks/common.py b/infx/benchmarks/common.py index 9506785eb8..4a8ac32aa0 100644 --- a/infx/benchmarks/common.py +++ b/infx/benchmarks/common.py @@ -12,7 +12,7 @@ import time from collections.abc import Mapping, Sequence from pathlib import Path -from typing import Any +from typing import Any, Literal from urllib.parse import urlsplit from .spec import PreparedFile, RuntimeSpec, secret_environment_key @@ -31,33 +31,60 @@ def verify_file(file: PreparedFile) -> Path: def verify_snapshot_assets( - runtime: RuntimeSpec, repository: str, *, expected_revision: str | None, only_snapshot: bool + runtime: RuntimeSpec, + repository: str, + *, + expected_revision: str | None, + only_snapshot: bool, + repo_type: Literal["dataset", "model"] = "dataset", ) -> str: - dataset = Path(runtime.env["HF_HUB_CACHE"]) / ("datasets--" + repository.replace("/", "--")) - reference = dataset / "refs" / "main" + cache = Path(runtime.env["HF_HUB_CACHE"]) / (f"{repo_type}s--" + repository.replace("/", "--")) + reference = cache / "refs" / "main" revision = reference.read_text().strip() if re.fullmatch(r"[0-9a-f]{40}", revision) is None: - raise ValueError("prepared dataset revision must be an immutable snapshot SHA") + raise ValueError(f"prepared {repo_type} revision must be an immutable snapshot SHA") if expected_revision is not None and revision != expected_revision: - raise ValueError("offline dataset main ref does not match the prepared snapshot") - snapshot = dataset / "snapshots" / revision + raise ValueError(f"offline {repo_type} main ref does not match the prepared snapshot") + snapshot = cache / "snapshots" / revision if not snapshot.is_dir(): - raise ValueError("prepared dataset snapshot is unavailable") + raise ValueError(f"prepared {repo_type} snapshot is unavailable") files = [path for path in snapshot.rglob("*") if path.is_file()] if not files: - raise ValueError("prepared dataset snapshot is empty") + raise ValueError(f"prepared {repo_type} snapshot is empty") bound = {Path(asset.path).resolve() for asset in runtime.assets} if any(path.resolve() not in bound for path in (reference, *files)): raise ValueError( - "dataset snapshot/ref contains content absent from the prepared asset list" + f"{repo_type} snapshot/ref contains content absent from the prepared asset list" ) if only_snapshot and [path for path in snapshot.parent.iterdir() if path.is_dir()] != [ snapshot ]: - raise ValueError("pilot dataset cache must contain only the prepared snapshot") + raise ValueError(f"pilot {repo_type} cache must contain only the prepared snapshot") return revision +def verify_model_snapshot_assets( + runtime: RuntimeSpec, + repository: str, + *, + expected_revision: str, + expected_snapshot: Path, +) -> Path: + """Bind nominal tokenizer lookup to the exact canonical serving snapshot.""" + revision = verify_snapshot_assets( + runtime, + repository, + expected_revision=expected_revision, + only_snapshot=False, + repo_type="model", + ) + cache = Path(runtime.env["HF_HUB_CACHE"]) / ("models--" + repository.replace("/", "--")) + snapshot = (cache / "snapshots" / revision).resolve(strict=True) + if snapshot != expected_snapshot.resolve(strict=True): + raise ValueError("offline model cache snapshot differs from the serving model snapshot") + return snapshot + + def _unique_object(pairs: list[tuple[str, Any]]) -> dict[str, Any]: result: dict[str, Any] = {} for key, value in pairs: diff --git a/infx/srt_slurm/launch.py b/infx/srt_slurm/launch.py index ad93d901f0..5744fcc778 100644 --- a/infx/srt_slurm/launch.py +++ b/infx/srt_slurm/launch.py @@ -16,7 +16,12 @@ import yaml from pydantic import BaseModel, ConfigDict, Field -from infx.benchmarks.common import child_failed, verify_file, write_json +from infx.benchmarks.common import ( + child_failed, + verify_file, + verify_model_snapshot_assets, + write_json, +) from infx.benchmarks.identity import capture_identity, verify_runtime from infx.benchmarks.spec import RuntimeSpec from infx.srt_slurm.contracts import digest, load_mapping @@ -203,6 +208,12 @@ def prepare( raise ValueError("serving and client model snapshots disagree") prepare_client(client_site, kind=kind, output=directory / "client") runtime = RuntimeSpec.model_validate(read_json(directory / "client" / "runtime.json")) + verify_model_snapshot_assets( + runtime, + job.row.model, + expected_revision=site.model_revision, + expected_snapshot=Path(site.model_snapshot), + ) site.require_interpreter( read_json(Path(runtime.identity.path)), python_minor="3.11" if kind == "agentx" else None, diff --git a/utils/test_model_cache_binding.py b/utils/test_model_cache_binding.py new file mode 100644 index 0000000000..00ba58efe7 --- /dev/null +++ b/utils/test_model_cache_binding.py @@ -0,0 +1,209 @@ +"""Nominal tokenizer lookup must resolve to the prepared serving snapshot.""" + +import json +import shutil +import subprocess +from pathlib import Path + +import pytest +from test_benchmark_preparation import client_site, installed_child # noqa: F401 +from test_native_pilot import inputs as pilot_inputs # noqa: F401 + +from infx.benchmarks.common import verify_model_snapshot_assets +from infx.benchmarks.prepare import bind_file +from infx.benchmarks.spec import RuntimeSpec +from infx.srt_slurm import launch +from infx.srt_slurm.job import parse_job + + +@pytest.fixture +def model_cache(tmp_path): + cache = tmp_path / "hub/models--example--model" + reference = cache / "refs/main" + reference.parent.mkdir(parents=True) + reference.write_text("a" * 40) + snapshot = cache / "snapshots" / ("a" * 40) + snapshot.mkdir(parents=True) + blob = cache / "blobs/tokenizer" + blob.parent.mkdir() + blob.write_bytes(b'[{"id": 7, "piece": "prepared"}]') + tokenizer = snapshot / "tokenizer.json" + tokenizer.symlink_to("../../blobs/tokenizer") + config = snapshot / "config.json" + config.write_text('{"model_type":"fixture"}') + identity = tmp_path / "identity.json" + identity.write_text("{}") + runtime = RuntimeSpec( + python=str(tmp_path / "python"), + identity=bind_file(identity), + distributions=["fixture-client"], + env={ + "HF_HUB_OFFLINE": "1", + "HF_DATASETS_OFFLINE": "1", + "HF_HUB_CACHE": str(tmp_path / "hub"), + "HF_DATASETS_CACHE": str(tmp_path / "datasets"), + }, + env_unset=[], + assets=[bind_file(path) for path in (reference, config, tokenizer)], + timeout_seconds=10, + terminate_grace_seconds=1, + ) + return runtime, reference, snapshot, tokenizer + + +def test_bound_model_cache_resolves_to_serving_snapshot(model_cache): + runtime, _, snapshot, _ = model_cache + resolved = verify_model_snapshot_assets( + runtime, + "example/model", + expected_revision="a" * 40, + expected_snapshot=snapshot, + ) + assert resolved == snapshot.resolve() + assert ( + resolved / "tokenizer.json" + ).read_bytes() == b'[{"id": 7, "piece": "prepared"}]' + + +def test_model_cache_requires_bound_tokenizer(model_cache): + runtime, _, snapshot, tokenizer = model_cache + runtime = runtime.model_copy( + update={ + "assets": [ + asset for asset in runtime.assets if asset.path != str(tokenizer) + ] + } + ) + with pytest.raises(ValueError, match="model snapshot/ref contains content absent"): + verify_model_snapshot_assets( + runtime, + "example/model", + expected_revision="a" * 40, + expected_snapshot=snapshot, + ) + + +def test_model_cache_rejects_distinct_serving_path_with_same_revision( + model_cache, tmp_path +): + runtime, _, snapshot, _ = model_cache + other_snapshot = tmp_path / "other" / snapshot.name + other_snapshot.mkdir(parents=True) + with pytest.raises( + ValueError, match="cache snapshot differs from the serving model" + ): + verify_model_snapshot_assets( + runtime, + "example/model", + expected_revision="a" * 40, + expected_snapshot=other_snapshot, + ) + + +@pytest.mark.parametrize( + "reference_revision,bind_reference,error", + [ + ("b" * 40, True, "offline model main ref does not match"), + ("a" * 40, False, "model snapshot/ref contains content absent"), + ], +) +def test_bad_model_cache_fails_preparation_before_native_boundary( + pilot_inputs, client_site, monkeypatch, reference_revision, bind_reference, error +): + root, row, scheduling, site = pilot_inputs + cache = ( + Path(client_site.env["HF_HUB_CACHE"]) + / "models--deepseek-ai--DeepSeek-V4.1-Flash" + ) + reference = cache / "refs/main" + reference.parent.mkdir(parents=True) + reference.write_text(reference_revision) + snapshot = cache / "snapshots" / ("a" * 40) + snapshot.parent.mkdir() + shutil.move(client_site.model_path, snapshot) + client_site = client_site.model_copy( + update={ + "model_path": str(snapshot), + "asset_roots": [str(snapshot), client_site.asset_roots[1]], + "asset_files": [str(reference)] if bind_reference else [], + } + ) + client_file = Path(site.shared_root) / "client-site.json" + client_file.write_text(client_site.model_dump_json()) + native_source = Path(site.native_source) + native_source.mkdir(parents=True) + (native_source / "uv.lock").write_text("controlled dependency lock") + Path(site.image.path).write_bytes(b"controlled image") + site = site.model_copy( + update={ + "model_snapshot": str(snapshot), + "model_revision": "a" * 40, + "image": bind_file(Path(site.image.path)), + "client_sites": {"agentx": str(client_file), "eval": str(client_file)}, + } + ) + (root / "runtime.json").write_text( + json.dumps( + { + "schema_version": 1, + "repository": "https://example.invalid/native.git", + "revision": "c" * 40, + "uv_lock_sha256": bind_file(native_source / "uv.lock").sha256, + "capabilities": [], + } + ) + ) + actual_run = subprocess.run + native_calls = [] + + def external_process(argv, **kwargs): + if argv[0] == "git": + return subprocess.CompletedProcess( + argv, 0, "c" * 40 if "rev-parse" in argv else "", "" + ) + if argv[0] in {site.wrapper_python, site.native_python}: + if "-m" in argv: + native_calls.append(argv) + raise AssertionError("invalid cache reached native execution") + distribution = argv[argv.index("--distribution") + 1] + identity = { + "python_version": "3.12.0", + "python_paths": { + key: site.shared_root + for key in ( + "executable", + "executable_resolved", + "prefix", + "base_prefix", + ) + }, + "distributions": {distribution: {"files": {}}}, + } + return subprocess.CompletedProcess(argv, 0, json.dumps(identity), "") + return actual_run(argv, **kwargs) + + monkeypatch.setattr(subprocess, "run", external_process) + job = parse_job( + { + **row, + "conc": 28, + "run-eval": True, + "eval-only": True, + "eval-framework": "lm-eval", + }, + root, + scheduling, + ) + with pytest.raises(ValueError, match=error): + launch.prepare( + job, + site, + root, + { + "repository": "example/pilot", + "run_id": 10, + "attempt": 1, + "head_sha": "d" * 40, + }, + ) + assert native_calls == [] From 4e44348ee1bc46b23297f88e1343137597cb011d Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sat, 19 Sep 2026 19:40:27 -0400 Subject: [PATCH 05/16] fix(ci): launch native pilot with managed Python MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 使用 uv 管理的 Python 3.12 启动原生试点,修复 H100 runner 缺少 python 命令的问题;在准备和 Slurm 申请前明确报告缺失或不匹配的站点与部署变量,并增加行为回归测试。 --- .github/workflows/benchmark-tmpl.yml | 7 +-- docs/srt-slurm-phase1.md | 2 + docs/srt-slurm-phase1_zh.md | 2 + infx/srt_slurm/workflow.py | 52 ++++++++++++++++++--- utils/test_native_workflow_site.py | 70 ++++++++++++++++++++++++++++ 5 files changed, 124 insertions(+), 9 deletions(-) create mode 100644 utils/test_native_workflow_site.py diff --git a/.github/workflows/benchmark-tmpl.yml b/.github/workflows/benchmark-tmpl.yml index f33d79b52e..231fa5b91b 100644 --- a/.github/workflows/benchmark-tmpl.yml +++ b/.github/workflows/benchmark-tmpl.yml @@ -301,10 +301,11 @@ jobs: NATIVE_SITE_JSON: ${{ vars.INFX_H100_PHASE1_SITE_JSON }} NATIVE_READER_REVISION: ${{ vars.INFX_PHASE1_READER_REVISION }} NATIVE_COLLECTOR_REVISION: ${{ vars.INFX_PHASE1_COLLECTOR_REVISION }} - shell: python + # H100 runners lack ambient python; uv supplies the explicit managed interpreter. + shell: uv run --locked --python 3.12 python {0} # zizmor: ignore[misfeature] run: | - import subprocess - subprocess.run(['uv', 'run', '--locked', '--python', '3.12', 'python', '-m', 'infx.srt_slurm.workflow'], check=True) + from infx.srt_slurm.workflow import main + raise SystemExit(main()) - name: Upload native execution provenance if: ${{ always() && fromJSON(inputs.config).execution.runtime == 'srt-slurm' && env.NATIVE_POINT_ID != '' }} diff --git a/docs/srt-slurm-phase1.md b/docs/srt-slurm-phase1.md index 709797196a..edd9d28179 100644 --- a/docs/srt-slurm-phase1.md +++ b/docs/srt-slurm-phase1.md @@ -69,6 +69,8 @@ Preparation validates existing assets; it does not install packages, download mo Before enabling sweeps, deploy the app reader and migration `016_measurement_snapshots.sql`, then land/deploy the trusted collector. Configure `INFX_H100_PHASE1_SITE_JSON`, `INFX_PHASE1_READER_REVISION` and `INFX_PHASE1_COLLECTOR_REVISION` in InferenceX. Configure `INFX_RECEIPT_ISSUER_SHAS` and `INFX_RECEIPT_ISSUER_WORKFLOW` in both repositories; the workflow is `.github/workflows/phase1-receipt.yml`. These values are absent in the inspected repository configuration. A source branch containing the code alone is not a deployed reader. +The GitHub native launch uses uv-managed Python 3.12 and does not require an ambient `python` command. The workflow checks the three site/deployment variables before preparation or Slurm allocation. Its error names missing variables, identifies invalid fields in the site JSON without echoing their values, and names reader/collector revision variables that disagree with the site configuration. This check requires explicit configuration and does not provision assets or deploy services. + ## Preparation, execution and recovery The standalone adapter accepts explicit files: diff --git a/docs/srt-slurm-phase1_zh.md b/docs/srt-slurm-phase1_zh.md index 6399192239..15357674e3 100644 --- a/docs/srt-slurm-phase1_zh.md +++ b/docs/srt-slurm-phase1_zh.md @@ -69,6 +69,8 @@ flowchart TD 先部署 app reader 与 `016_measurement_snapshots.sql` migration,再合入/部署受信任 collector。InferenceX 需配置 `INFX_H100_PHASE1_SITE_JSON`、`INFX_PHASE1_READER_REVISION`、`INFX_PHASE1_COLLECTOR_REVISION`。两个仓库均需配置 `INFX_RECEIPT_ISSUER_SHAS`、`INFX_RECEIPT_ISSUER_WORKFLOW`,workflow 路径为 `.github/workflows/phase1-receipt.yml`。检查时这些变量尚不存在。分支中有代码不等于 reader 已部署。 +GitHub 原生启动步骤使用 uv 管理的 Python 3.12,不依赖环境中已有的 `python` 命令。workflow 在准备或申请 Slurm 资源前校验这三个站点/部署变量。错误会列出缺失变量,指出站点 JSON 的无效字段而不回显字段值,并列出与站点配置不一致的 reader/collector revision 变量。该校验要求显式配置,不会准备资源或部署服务。 + ## 准备、执行与恢复 独立适配器接收明确的文件参数: diff --git a/infx/srt_slurm/workflow.py b/infx/srt_slurm/workflow.py index 5bc4bb75ec..43f4d4b18c 100644 --- a/infx/srt_slurm/workflow.py +++ b/infx/srt_slurm/workflow.py @@ -5,15 +5,61 @@ import json import os import subprocess +from collections.abc import Mapping from pathlib import Path +from pydantic import ValidationError + from infx.benchmarks.common import write_json from infx.srt_slurm.job import parse_job from infx.srt_slurm.launch import execute, prepare from infx.srt_slurm.render import PilotSite +def load_site(environment: Mapping[str, str]) -> PilotSite: + """Explain missing deployment configuration before touching preparation or Slurm.""" + variables = { + "NATIVE_SITE_JSON": "INFX_H100_PHASE1_SITE_JSON", + "NATIVE_READER_REVISION": "INFX_PHASE1_READER_REVISION", + "NATIVE_COLLECTOR_REVISION": "INFX_PHASE1_COLLECTOR_REVISION", + } + missing = [ + repository_name + for name, repository_name in variables.items() + if not environment.get(name, "").strip() + ] + if missing: + raise ValueError("Native H100 pilot requires repository variables: " + ", ".join(missing)) + try: + site = PilotSite.model_validate_json(environment["NATIVE_SITE_JSON"]) + except ValidationError as error: + problems = "; ".join( + f"{'.'.join(str(part) for part in item['loc']) or 'JSON'}: {item['msg']}" + for item in error.errors(include_input=False, include_context=False, include_url=False) + ) + raise ValueError( + "Repository variable INFX_H100_PHASE1_SITE_JSON must contain valid PilotSite JSON: " + + problems + ) from None + mismatched = [ + variables[name] + for name, expected in ( + ("NATIVE_READER_REVISION", site.reader_revision), + ("NATIVE_COLLECTOR_REVISION", site.collector_revision), + ) + if environment[name] != expected + ] + if mismatched: + raise ValueError( + "Repository variables " + + ", ".join(mismatched) + + " must match the deployed revisions recorded in INFX_H100_PHASE1_SITE_JSON" + ) + return site + + def main() -> int: + site = load_site(os.environ) root = Path(os.environ["GITHUB_WORKSPACE"]).resolve() raw = json.loads(os.environ["NATIVE_CONFIG_JSON"]) if os.environ["NATIVE_AGENTX_FAST"] != "false" or os.environ["NATIVE_EVAL_LIMIT"] not in ( @@ -40,12 +86,6 @@ def main() -> int: "node-count": 1, }, ) - site = PilotSite.model_validate_json(os.environ["NATIVE_SITE_JSON"]) - if ( - os.environ["NATIVE_READER_REVISION"] != site.reader_revision - or os.environ["NATIVE_COLLECTOR_REVISION"] != site.collector_revision - ): - raise ValueError("reader-first deployment and trusted collector pins are not enabled") source = { "repository": os.environ["GITHUB_REPOSITORY"], "run_id": int(os.environ["GITHUB_RUN_ID"]), diff --git a/utils/test_native_workflow_site.py b/utils/test_native_workflow_site.py new file mode 100644 index 0000000000..cc726801a4 --- /dev/null +++ b/utils/test_native_workflow_site.py @@ -0,0 +1,70 @@ +"""Actionable deployment failures precede any native preparation or scheduler work.""" + +import json + +import pytest +from test_native_pilot import inputs as pilot_inputs # noqa: F401 + +from infx.srt_slurm import workflow + + +def test_missing_site_variables_fail_before_workflow_or_scheduler_inputs(monkeypatch): + monkeypatch.setattr(workflow.os, "environ", {"NATIVE_SITE_JSON": " "}) + with pytest.raises(ValueError) as error: + workflow.main() + assert str(error.value) == ( + "Native H100 pilot requires repository variables: INFX_H100_PHASE1_SITE_JSON, " + "INFX_PHASE1_READER_REVISION, INFX_PHASE1_COLLECTOR_REVISION" + ) + + +@pytest.fixture +def environment(pilot_inputs): + *_, site = pilot_inputs + return { + "NATIVE_SITE_JSON": site.model_dump_json(), + "NATIVE_READER_REVISION": "a" * 40, + "NATIVE_COLLECTOR_REVISION": "b" * 40, + } + + +def test_explicit_matching_deployed_revisions_load_site(environment): + site = workflow.load_site(environment) + assert site.cluster == "h100-dgxc" + assert site.reader_revision == "a" * 40 + assert site.collector_revision == "b" * 40 + + +@pytest.mark.parametrize( + "value,diagnostic", + [("{", "JSON: Invalid JSON"), ('{"schema_version":2}', "schema_version:")], +) +def test_invalid_site_names_repository_variable(environment, value, diagnostic): + environment["NATIVE_SITE_JSON"] = value + with pytest.raises(ValueError) as error: + workflow.load_site(environment) + assert str(error.value).startswith( + "Repository variable INFX_H100_PHASE1_SITE_JSON must contain valid PilotSite JSON: " + ) + assert diagnostic in str(error.value) + + +@pytest.mark.parametrize("role", ["READER", "COLLECTOR"]) +def test_mismatched_deployment_names_the_specific_variable(environment, role): + environment[f"NATIVE_{role}_REVISION"] = "c" * 40 + with pytest.raises(ValueError) as error: + workflow.load_site(environment) + assert str(error.value) == ( + f"Repository variables INFX_PHASE1_{role}_REVISION must match the deployed revisions " + "recorded in INFX_H100_PHASE1_SITE_JSON" + ) + + +def test_invalid_site_diagnostic_excludes_input_values(environment): + site = json.loads(environment["NATIVE_SITE_JSON"]) + site["schema_version"] = "not-a-version-sensitive-value" + environment["NATIVE_SITE_JSON"] = json.dumps(site) + with pytest.raises(ValueError) as error: + workflow.load_site(environment) + assert "schema_version:" in str(error.value) + assert "not-a-version-sensitive-value" not in str(error.value) From c4af40de63cb5352e1bf1a725e7d8f07752c3441 Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sat, 19 Sep 2026 19:44:30 -0400 Subject: [PATCH 06/16] fix(results): stream receipt archive verification MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Hash ZIP members in bounded chunks, retain CRC and measured-size validation, and parse evaluation samples one line at a time. Cover real large archives, bounded memory, corrupted CRCs, and false member sizes. 分块计算回执 ZIP 成员哈希,保留 CRC 与实际大小校验,并逐行解析评估样本。新增真实大归档、内存上限、CRC 损坏及成员大小不符的行为测试,并同步中英文文档。 --- docs/measurement-receipts.md | 2 + docs/measurement-receipts_zh.md | 2 + infx/results/publication_receipt.py | 86 +++++++++++++++++------------ utils/test_publication_receipt.py | 81 +++++++++++++++++++++++++-- 4 files changed, 130 insertions(+), 41 deletions(-) diff --git a/docs/measurement-receipts.md b/docs/measurement-receipts.md index b7999debac..85d530ca00 100644 --- a/docs/measurement-receipts.md +++ b/docs/measurement-receipts.md @@ -19,6 +19,8 @@ Maintainers review `qualification/phase1/*.json` inputs on trusted `main` before The [receipt validator](../infx/results/publication_receipt.py) checks exact GitHub artifact IDs and ownership, API and ZIP digests, contained members, expected execution identities, physical topology, canonical configuration, required metrics, dataset metadata, and complete evaluation sample/filter coverage. Per-job raw lm-eval results and metadata remain valid inputs; aggregate deployments retain explicit zero split-worker counts. The compact version-1 receipt preserves original measurements when staging later becomes production. +Archive member digests are computed with 1 MiB reads, so a large AgentX trace is not loaded into memory at once. The validator reads each member to EOF to verify its CRC and measured size, while retaining the 10 GiB per-member and 20 GiB per-artifact limits and existing path checks. Evaluation JSONL is decoded one line at a time; its memory use depends on the largest sample line and the tracked document/filter identities, with the same coverage and score checks. Summary and execution JSON remain structured validation inputs. + ## Staging, publication, and recovery The [transport resolver](../infx/workflows/receipt_transport.py) uses read-only APIs to locate a unique accepted receipt from a successful `workflow_dispatch` issuer on `main` at an allowed revision. It verifies the original source attempt rather than substituting the latest rerun. An API inventory containing `native-execution-*` requires a receipt; missing or invalid evidence never enters legacy native ingestion. Ordinary legacy inventories retain their existing path. diff --git a/docs/measurement-receipts_zh.md b/docs/measurement-receipts_zh.md index 0f5e0f92d4..7c00ad2a46 100644 --- a/docs/measurement-receipts_zh.md +++ b/docs/measurement-receipts_zh.md @@ -19,6 +19,8 @@ [回执验证器](../infx/results/publication_receipt.py) 检查准确的 GitHub 产物 ID 及归属、API 与 ZIP digest、安全成员路径、预期执行身份、物理拓扑、规范化配置、必需指标、数据集元数据,以及完整的评估样本和过滤器覆盖。输入支持每个任务的原始 lm-eval 结果与元数据;聚合部署保留明确为零的拆分 worker 计数。紧凑的 version 1 回执在结果从 staging 进入生产时保留原始测量身份。 +归档成员的 digest 通过每次读取 1 MiB 数据计算,避免将大型 AgentX trace 一次性载入内存。验证器将每个成员读取到 EOF,以校验 CRC 和实际字节数,同时保留单成员 10 GiB、单产物 20 GiB 的限制及原有路径检查。评估 JSONL 逐行解码,内存占用取决于最大的样本行和已记录的文档/过滤器身份;覆盖率和分数检查保持不变。摘要及执行 JSON 仍作为结构化输入进行验证。 + ## Staging、发布与恢复 [传输解析器](../infx/workflows/receipt_transport.py) 使用只读 API,从允许版本在 `main` 上成功完成的 `workflow_dispatch` 签发运行中查找唯一的已接受回执。它验证源运行原始 attempt,不用最近一次 rerun 替换。只要 API 清单包含 `native-execution-*`,就必须提供回执;证据缺失或无效时,不能回退到旧 native 导入方式。普通旧产物清单继续使用现有路径。 diff --git a/infx/results/publication_receipt.py b/infx/results/publication_receipt.py index fb5cd37376..91551fb797 100644 --- a/infx/results/publication_receipt.py +++ b/infx/results/publication_receipt.py @@ -9,6 +9,7 @@ import argparse import hashlib +import io import json import math import re @@ -195,10 +196,17 @@ def inspect_archive(archive: Path) -> list[Member]: total += item.file_size if total > 20 * 1024**3 or item.file_size > 10 * 1024**3: raise ValueError("Artifact exceeds extraction budget") - payload = source.read(item) - members.append( - Member(path=name, size=len(payload), sha256=hashlib.sha256(payload).hexdigest()) - ) + digest = hashlib.sha256() + size = 0 + with source.open(item) as payload: + while chunk := payload.read(1024**2): + size += len(chunk) + if size > item.file_size: + raise ValueError(f"Archive member exceeds declared size: {name}") + digest.update(chunk) + if size != item.file_size: + raise ValueError(f"Archive member differs from declared size: {name}") + members.append(Member(path=name, size=size, sha256=digest.hexdigest())) files.add(name) for name in files: if any(str(parent) in files for parent in PurePosixPath(name).parents): @@ -306,8 +314,6 @@ def validate_point_content(point: Point, archives: Path) -> None: if point.dataset and row.get("dataset") != point.dataset: raise ValueError("Dataset identity differs from independent expectation") if point.kind == "eval": - with zipfile.ZipFile(archives / f"{point.samples_artifact_id}.zip") as archive: - text = archive.read(safe_member(point.samples_path or "")).decode() identities = None if point.task == "gsm8k" and point.sample_count == 1319: identities = decode_json( @@ -317,38 +323,46 @@ def validate_point_content(point: Point, archives: Path) -> None: ) observed: set[tuple[int, str]] = set() strict_passed = 0 - for line in text.splitlines(): - if not line.strip(): - continue - sample = decode_json(line) - require_finite(sample) - doc_id, filter_name = sample.get("doc_id"), sample.get("filter") - if ( - type(doc_id) is not int - or doc_id < 0 - or filter_name not in point.filters - or sample.get("task_name", point.task) != point.task - or (doc_id, filter_name) in observed - ): - raise ValueError("Invalid/duplicate evaluation sample identity") - if identities is not None: - document_hash = hashlib.sha256( - json.dumps(sample.get("doc"), indent=2, ensure_ascii=False).encode() - ).hexdigest() - target = sample.get("target") + with ( + zipfile.ZipFile(archives / f"{point.samples_artifact_id}.zip") as archive, + archive.open(safe_member(point.samples_path or "")) as source, + io.TextIOWrapper(source, encoding="utf-8") as samples, + ): + for line in samples: + if not line.strip(): + continue + sample = decode_json(line) + require_finite(sample) + doc_id, filter_name = sample.get("doc_id"), sample.get("filter") if ( - identities.get(str(doc_id)) != document_hash - or sample.get("doc_hash") != document_hash - or target != sample.get("doc", {}).get("answer") - or sample.get("target_hash") != hashlib.sha256(str(target).encode()).hexdigest() + type(doc_id) is not int + or doc_id < 0 + or filter_name not in point.filters + or sample.get("task_name", point.task) != point.task + or (doc_id, filter_name) in observed ): - raise ValueError("Pilot eval document/target differs from prepared full split") - observed.add((doc_id, filter_name)) - if filter_name == "strict-match": - score = sample.get("exact_match,strict-match", sample.get("exact_match")) - if type(score) not in (int, float) or score not in (0, 1): - raise ValueError("GSM8K strict sample requires a binary score") - strict_passed += int(score) + raise ValueError("Invalid/duplicate evaluation sample identity") + if identities is not None: + document_hash = hashlib.sha256( + json.dumps(sample.get("doc"), indent=2, ensure_ascii=False).encode() + ).hexdigest() + target = sample.get("target") + if ( + identities.get(str(doc_id)) != document_hash + or sample.get("doc_hash") != document_hash + or target != sample.get("doc", {}).get("answer") + or sample.get("target_hash") + != hashlib.sha256(str(target).encode()).hexdigest() + ): + raise ValueError( + "Pilot eval document/target differs from prepared full split" + ) + observed.add((doc_id, filter_name)) + if filter_name == "strict-match": + score = sample.get("exact_match,strict-match", sample.get("exact_match")) + if type(score) not in (int, float) or score not in (0, 1): + raise ValueError("GSM8K strict sample requires a binary score") + strict_passed += int(score) documents = {doc for doc, _ in observed} if ( documents != set(range(point.sample_count)) diff --git a/utils/test_publication_receipt.py b/utils/test_publication_receipt.py index 6d8dbfd929..0b3b16fc32 100644 --- a/utils/test_publication_receipt.py +++ b/utils/test_publication_receipt.py @@ -3,7 +3,9 @@ import hashlib import json import stat +import struct import tempfile +import tracemalloc import unittest import zipfile from pathlib import Path @@ -150,6 +152,62 @@ def test_rejects_archive_traversal_duplicate_and_link_members(self): with self.assertRaisesRegex(ValueError, "link"): inspect_archive(self.root / "link.zip") + def test_hashes_large_compressed_member_with_bounded_memory(self): + archive_path = self.root / "large.zip" + with zipfile.ZipFile( + archive_path, "w", compression=zipfile.ZIP_DEFLATED + ) as archive: + with archive.open("trace.jsonl", "w") as trace: + for _ in range(64): + trace.write(b"x" * 1024**2) + archive.writestr("empty", b"") + tracemalloc.start() + try: + members = inspect_archive(archive_path) + _, peak = tracemalloc.get_traced_memory() + finally: + tracemalloc.stop() + self.assertEqual( + [member.model_dump() for member in members], + [ + { + "path": "empty", + "size": 0, + "sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855", + }, + { + "path": "trace.jsonl", + "size": 64 * 1024**2, + "sha256": "e20a69eca39368572e90b9135738a613838f954987a0b44b6220889c171cbb76", + }, + ], + ) + self.assertLess(peak, 8 * 1024**2) + + def test_streamed_member_rejects_crc_corruption_at_end(self): + archive_path = self.root / "bad-crc.zip" + with zipfile.ZipFile(archive_path, "w") as archive: + archive.writestr("trace.jsonl", b"x" * (1024**2 + 3)) + item = archive.getinfo("trace.jsonl") + payload_offset = item.header_offset + 30 + len(item.filename.encode()) + with archive_path.open("r+b") as source: + source.seek(payload_offset + item.file_size - 1) + source.write(b"y") + with self.assertRaisesRegex(zipfile.BadZipFile, "CRC"): + inspect_archive(archive_path) + + def test_streamed_member_rejects_incomplete_declared_size(self): + archive_path = self.root / "bad-size.zip" + with zipfile.ZipFile(archive_path, "w") as archive: + archive.writestr("trace.jsonl", b"abc") + data = bytearray(archive_path.read_bytes()) + central_header = data.index(b"PK\x01\x02") + # ZIP central-directory uncompressed size; payload and its CRC remain intact. + struct.pack_into(" Date: Sat, 19 Sep 2026 19:44:57 -0400 Subject: [PATCH 07/16] docs: record Phase 1 GitHub qualification evidence MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 记录真实 GitHub CI 通过结果及 H100 sweep 在 Slurm 提交前因站点和部署变量缺失而停止的证据,不将本地或 CPU 检查等同于硬件验收。 --- docs/srt-slurm-phase1.md | 2 ++ docs/srt-slurm-phase1_zh.md | 2 ++ 2 files changed, 4 insertions(+) diff --git a/docs/srt-slurm-phase1.md b/docs/srt-slurm-phase1.md index edd9d28179..c112abd6d9 100644 --- a/docs/srt-slurm-phase1.md +++ b/docs/srt-slurm-phase1.md @@ -114,4 +114,6 @@ The merge helper preserves the latest explicit authorized `/use RUN_ID` (or `/re | Power | Explicit temporary parity exception; measured power not claimed | | Retirement | Legacy H100 script retained pending all exit evidence | +GitHub [CI run 35476764022](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35476764022) verified commit `4e44348ee1bc46b23297f88e1343137597cb011d`: 1,871 Python tests and 2,458 native Linux tests passed, together with the installed-runtime check covering all nine points. [Sweep 35476764181](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35476764181) reached the H100 runners using managed Python 3.12, then stopped before Slurm submission because `INFX_H100_PHASE1_SITE_JSON`, `INFX_PHASE1_READER_REVISION` and `INFX_PHASE1_COLLECTOR_REVISION` were unset. This is a verified provisioning/deployment prerequisite failure, not H100 throughput or eval qualification. + Record actual InferenceX/native/collector/app commits, source run/attempt, prepared expectation, nine artifact bindings, source receipt, publication record and app verification report in this ledger when available. No fabricated IDs or placeholder success entries may close a gate. diff --git a/docs/srt-slurm-phase1_zh.md b/docs/srt-slurm-phase1_zh.md index 15357674e3..175ef68f4f 100644 --- a/docs/srt-slurm-phase1_zh.md +++ b/docs/srt-slurm-phase1_zh.md @@ -114,4 +114,6 @@ Merge helper 保留最近明确授权的 `/use RUN_ID` 或 `/reuse-sweep-run RUN | 功耗 | 明确临时一致性例外;不声称实测功耗 | | 退休 | 保留旧 H100 script,等待全部出口证据 | +GitHub [CI run 35476764022](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35476764022) 已验证提交 `4e44348ee1bc46b23297f88e1343137597cb011d`:1,871 项 Python 测试、2,458 项原生 Linux 测试以及覆盖全部九点的已安装运行时检查均通过。[Sweep 35476764181](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35476764181) 使用受管理的 Python 3.12 到达 H100 runner,但因 `INFX_H100_PHASE1_SITE_JSON`、`INFX_PHASE1_READER_REVISION` 和 `INFX_PHASE1_COLLECTOR_REVISION` 未设置,在提交 Slurm 任务前停止。这是已验证的资源准备/部署前置条件失败,不代表 H100 吞吐或 eval 已完成验收。 + 证据就绪后记录实际 InferenceX/native/collector/app 提交、source run/attempt、准备期预期、九点 artifact 绑定、源回执、发布记录与 app 验证报告。不得用编造 ID 或占位成功条目关闭 gate。 From 12ef72231a7efae9b6bd1a011ccf5ddd3bde5fea Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sat, 19 Sep 2026 19:47:26 -0400 Subject: [PATCH 08/16] refactor(recipes): follow existing H100 recipe layout MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 将 H100 DSV4.1 Flash 配方移入现有 model/engine/GPU/workload 目录,将运行时锁和客户端策略移入同一配方树的 configs 目录;同步主配置、CI、测试夹具与中英文架构文档。配方字节、生成矩阵和历史性能记录保持不变,仅更新路径引用。 --- .github/workflows/ci.yml | 8 +++++--- .../multi_node/srt-slurm-recipes/RECIPES.md | 17 ++++++++++++++--- .../multi_node/srt-slurm-recipes/RECIPES_zh.md | 17 ++++++++++++++--- .../dsv41flash-agentx-client-policy.json} | 0 .../configs/prepared-runtime-lock.json} | 0 .../vllm/h100-fp4/agentx/agg-tp8-dspark5.yaml} | 0 configs/nvidia-master.yaml | 12 ++++++------ docs/srt-slurm-phase1.md | 4 +++- docs/srt-slurm-phase1_zh.md | 4 +++- utils/test_native_pilot.py | 4 ++-- 10 files changed, 47 insertions(+), 19 deletions(-) rename benchmarks/{srt-slurm/phase1/client-policy.json => multi_node/srt-slurm-recipes/configs/dsv41flash-agentx-client-policy.json} (100%) rename benchmarks/{srt-slurm/phase1/runtime-lock.json => multi_node/srt-slurm-recipes/configs/prepared-runtime-lock.json} (100%) rename benchmarks/{srt-slurm/phase1/h100-dsv41flash.yaml => multi_node/srt-slurm-recipes/dsv41flash/vllm/h100-fp4/agentx/agg-tp8-dspark5.yaml} (100%) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index ea7910fffc..bdff64bf78 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -15,7 +15,9 @@ on: - 'infx/ruff.toml' - '**/pytest.ini' - 'utils/srt-slurm' - - 'benchmarks/srt-slurm/phase1/**' + - 'benchmarks/multi_node/srt-slurm-recipes/dsv41flash/vllm/h100-fp4/**' + - 'benchmarks/multi_node/srt-slurm-recipes/configs/prepared-runtime-lock.json' + - 'benchmarks/multi_node/srt-slurm-recipes/configs/dsv41flash-agentx-client-policy.json' - 'runners/srt-slurm/h100-phase1.yaml' - 'utils/fixtures/native_pilot/**' push: @@ -106,7 +108,7 @@ jobs: import re from pathlib import Path - lock = json.loads(Path("benchmarks/srt-slurm/phase1/runtime-lock.json").read_text()) + lock = json.loads(Path("benchmarks/multi_node/srt-slurm-recipes/configs/prepared-runtime-lock.json").read_text()) if type(lock.get("schema_version")) is not int or lock["schema_version"] != 1: raise ValueError("Unsupported native runtime lock schema") if lock.get("repository") != "https://github.com/SemiAnalysisAI/srt-slurm.git": @@ -137,7 +139,7 @@ jobs: from pathlib import Path source = Path(".native-phase1").resolve() - lock = json.loads(Path("benchmarks/srt-slurm/phase1/runtime-lock.json").read_text()) + lock = json.loads(Path("benchmarks/multi_node/srt-slurm-recipes/configs/prepared-runtime-lock.json").read_text()) def git(*args): return subprocess.run( ["git", "-C", str(source), *args], check=True, diff --git a/benchmarks/multi_node/srt-slurm-recipes/RECIPES.md b/benchmarks/multi_node/srt-slurm-recipes/RECIPES.md index a731508e05..d16c75e4c3 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/RECIPES.md +++ b/benchmarks/multi_node/srt-slurm-recipes/RECIPES.md @@ -2,9 +2,9 @@ **English** | [中文](./RECIPES_zh.md) -InferenceX owns the recipes in this directory. Every NVIDIA srt-slurm launcher uses `setup_srt_slurm()` in [`runners/slurm_utils.sh`](../../../runners/slurm_utils.sh), makes a job-local Git clone of the pinned submodule, and copies this entire tree into `recipes/`. The shared helper records the actual revision in `srt-slurm-sha.txt`; power lanes copy that revision into `power-producer-sha.txt` for result validation. +InferenceX owns the recipes in this directory. The shared NVIDIA launcher path uses `setup_srt_slurm()` in [`runners/slurm_utils.sh`](../../../runners/slurm_utils.sh), makes a job-local Git clone of the pinned submodule, and copies this entire tree into `recipes/`. The shared helper records the actual revision in `srt-slurm-sha.txt`; power lanes copy that revision into `power-producer-sha.txt` for result validation. The prepared H100 execution path described below uses its own explicit runtime lock. -The shared version is the Git submodule pointer at [`utils/srt-slurm`](../../../utils/srt-slurm), currently [v2.2.1](https://github.com/NVIDIA/srt-slurm/releases/tag/v2.2.1) (`984180e5b8755aef85e9995048b5a16cb5336bce`). Update that submodule pointer when upgrading, then run the recipe and integration checks. Do not add model-specific checkout branches to launchers. +The shared helper's version is the Git submodule pointer at [`utils/srt-slurm`](../../../utils/srt-slurm), currently [v2.2.1](https://github.com/NVIDIA/srt-slurm/releases/tag/v2.2.1) (`984180e5b8755aef85e9995048b5a16cb5336bce`). Update that submodule pointer when upgrading, then run the recipe and integration checks. Do not add model-specific checkout branches to launchers. InferenceX requires srt-slurm 2.0 or newer and `schema: 2` recipes. Legacy recipe layouts are unsupported; migrate them before adding them to this tree. @@ -29,7 +29,18 @@ Shared runtime assets stay under `configs/` beside the model directories; they a ## TileRT exception -For `FRAMEWORK=tilert`, `setup_srt_slurm()` fetches the SemiAnalysisAI/srt-slurm fork directly at `6bc3f306bdafa1edfb5dded2fcda8f1ccede1bde` into the job checkout. This is the schema-2 TileRT port in [SemiAnalysisAI/srt-slurm#13](https://github.com/SemiAnalysisAI/srt-slurm/pull/13). It is the only alternate checkout; its pin lives in that helper because the TileRT backend and router are absent from the NVIDIA pin. TileRT uses the same schema-2 recipe layout and native post-eval dispatch as NVIDIA. TileRT jobs need network access to the fork at setup time. Remove the fork exception once those features are available upstream. +For `FRAMEWORK=tilert`, `setup_srt_slurm()` fetches the SemiAnalysisAI/srt-slurm fork directly at `6bc3f306bdafa1edfb5dded2fcda8f1ccede1bde` into the job checkout. This is the schema-2 TileRT port in [SemiAnalysisAI/srt-slurm#13](https://github.com/SemiAnalysisAI/srt-slurm/pull/13). It is the only alternate checkout selected by that helper; its pin lives in the helper because the TileRT backend and router are absent from the NVIDIA pin. TileRT uses the same schema-2 recipe layout and native post-eval dispatch as NVIDIA. TileRT jobs need network access to the fork at setup time. Remove the fork exception once those features are available upstream. + +## Prepared H100 execution + +The H100 DSV4.1 Flash pilot uses the same recipe hierarchy, including for its single-node aggregate topology. Its master entry selects the prepared Python adapter with `execution.runtime: srt-slurm` and binds these committed inputs: + +- Recipe: [`dsv41flash/vllm/h100-fp4/agentx/agg-tp8-dspark5.yaml`](./dsv41flash/vllm/h100-fp4/agentx/agg-tp8-dspark5.yaml). +- `execution.runtime-lock`: [`configs/prepared-runtime-lock.json`](./configs/prepared-runtime-lock.json), which pins the prepared native source revision, dependency-lock digest and required capabilities independently of the shared `utils/srt-slurm` gitlink. +- `execution.client-policy`: [`configs/dsv41flash-agentx-client-policy.json`](./configs/dsv41flash-agentx-client-policy.json), which binds the AgentX and GSM8K evaluation requirements and the measured golden-curve input. +- `execution.profile`: [`runners/srt-slurm/h100-phase1.yaml`](../../../runners/srt-slurm/h100-phase1.yaml), which supplies the H100 cluster profile. + +The adapter validates these input digests, the separately installed native runtime and explicitly prepared shared assets before allocation. This selection uses the prepared runtime contract rather than the shared helper's `CONFIG_FILE` submission path. Keep recipe moves synchronized with every `execution` reference, its input digests and workflow filters. Provisioning, deployment prerequisites, validation commands and hardware qualification limits are documented in the [Phase 1 guide](../../../docs/srt-slurm-phase1.md). ## Schema 2 and master configuration diff --git a/benchmarks/multi_node/srt-slurm-recipes/RECIPES_zh.md b/benchmarks/multi_node/srt-slurm-recipes/RECIPES_zh.md index ef3c0ed034..7a80f39963 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/RECIPES_zh.md +++ b/benchmarks/multi_node/srt-slurm-recipes/RECIPES_zh.md @@ -2,9 +2,9 @@ [English](./RECIPES.md) | **中文** -InferenceX 负责维护本目录中的配置。所有 NVIDIA srt-slurm 启动器均调用 [`runners/slurm_utils.sh`](../../../runners/slurm_utils.sh) 中的 `setup_srt_slurm()`,为作业创建固定版本子模块的本地 Git 克隆,并将整个目录复制到 `recipes/`。共享函数将实际提交记录到 `srt-slurm-sha.txt`;功耗测试路径还会将其复制到 `power-producer-sha.txt`,供结果校验使用。 +InferenceX 负责维护本目录中的配置。共享 NVIDIA 启动路径调用 [`runners/slurm_utils.sh`](../../../runners/slurm_utils.sh) 中的 `setup_srt_slurm()`,为作业创建固定版本子模块的本地 Git 克隆,并将整个目录复制到 `recipes/`。共享函数将实际提交记录到 `srt-slurm-sha.txt`;功耗测试路径还会将其复制到 `power-producer-sha.txt`,供结果校验使用。下文介绍的 H100 prepared 执行路径使用独立且显式指定的运行时锁文件。 -统一版本由 [`utils/srt-slurm`](../../../utils/srt-slurm) 的 Git 子模块指针指定,目前为 [v2.2.1](https://github.com/NVIDIA/srt-slurm/releases/tag/v2.2.1)(`984180e5b8755aef85e9995048b5a16cb5336bce`)。升级时更新该子模块指针,然后运行配置和集成检查。不要在启动器中新增按模型选择检出版本的分支。 +共享函数使用的版本由 [`utils/srt-slurm`](../../../utils/srt-slurm) 的 Git 子模块指针指定,目前为 [v2.2.1](https://github.com/NVIDIA/srt-slurm/releases/tag/v2.2.1)(`984180e5b8755aef85e9995048b5a16cb5336bce`)。升级时更新该子模块指针,然后运行配置和集成检查。不要在启动器中新增按模型选择检出版本的分支。 InferenceX 要求 srt-slurm 2.0 或更新版本,且配置必须声明 `schema: 2`。不支持旧版配置结构;加入本目录前必须先完成迁移。 @@ -29,7 +29,18 @@ qwen3.5/trtllm/gb300-fp4/agentx/disagg-1p7d-dep4-tep8-c7-b1-mtp-kvoffload.yaml ## TileRT 例外 -当 `FRAMEWORK=tilert` 时,`setup_srt_slurm()` 直接从 SemiAnalysisAI/srt-slurm 分支仓库获取提交 `6bc3f306bdafa1edfb5dded2fcda8f1ccede1bde`,检出到作业目录。该版本为 [SemiAnalysisAI/srt-slurm#13](https://github.com/SemiAnalysisAI/srt-slurm/pull/13) 中支持 schema 2 的 TileRT 移植。这是唯一的备用检出路径;由于统一的 NVIDIA 版本尚未包含 TileRT 后端和路由器,该例外的固定提交在共享函数中指定。TileRT 使用与 NVIDIA 相同的 schema 2 配置结构和原生评估调度。TileRT 作业在准备阶段需要通过网络访问分支仓库。上游支持这些功能后,应删除此分支仓库例外。 +当 `FRAMEWORK=tilert` 时,`setup_srt_slurm()` 直接从 SemiAnalysisAI/srt-slurm 分支仓库获取提交 `6bc3f306bdafa1edfb5dded2fcda8f1ccede1bde`,检出到作业目录。该版本为 [SemiAnalysisAI/srt-slurm#13](https://github.com/SemiAnalysisAI/srt-slurm/pull/13) 中支持 schema 2 的 TileRT 移植。这是该共享函数选择的唯一备用检出路径;由于共享 NVIDIA 版本尚未包含 TileRT 后端和路由器,该例外的固定提交在共享函数中指定。TileRT 使用与 NVIDIA 相同的 schema 2 配置结构和原生评估调度。TileRT 作业在准备阶段需要通过网络访问分支仓库。上游支持这些功能后,应删除此分支仓库例外。 + +## H100 prepared 执行路径 + +H100 DSV4.1 Flash 试点遵循相同的配置目录结构,其单节点聚合拓扑也使用此结构。主配置条目通过 `execution.runtime: srt-slurm` 选择 prepared Python 适配器,并绑定以下已纳入版本控制的输入: + +- 配置:[`dsv41flash/vllm/h100-fp4/agentx/agg-tp8-dspark5.yaml`](./dsv41flash/vllm/h100-fp4/agentx/agg-tp8-dspark5.yaml)。 +- `execution.runtime-lock`:[`configs/prepared-runtime-lock.json`](./configs/prepared-runtime-lock.json),独立固定 prepared 原生源码提交、依赖锁文件 digest 及必需能力,不依赖共享 `utils/srt-slurm` gitlink。 +- `execution.client-policy`:[`configs/dsv41flash-agentx-client-policy.json`](./configs/dsv41flash-agentx-client-policy.json),绑定 AgentX、GSM8K 评估要求及实测 golden-curve 输入。 +- `execution.profile`:[`runners/srt-slurm/h100-phase1.yaml`](../../../runners/srt-slurm/h100-phase1.yaml),提供 H100 集群配置。 + +适配器在申请资源前校验上述输入 digest、独立安装的原生运行时以及显式准备的共享资源。此路径采用 prepared 运行时契约,不经过共享函数的 `CONFIG_FILE` 提交流程。移动配置时,同步更新所有 `execution` 引用、对应输入 digest 及工作流过滤器。资源准备、部署前置条件、验证命令和硬件验收边界见 [Phase 1 指南](../../../docs/srt-slurm-phase1_zh.md)。 ## Schema 2 与主配置 diff --git a/benchmarks/srt-slurm/phase1/client-policy.json b/benchmarks/multi_node/srt-slurm-recipes/configs/dsv41flash-agentx-client-policy.json similarity index 100% rename from benchmarks/srt-slurm/phase1/client-policy.json rename to benchmarks/multi_node/srt-slurm-recipes/configs/dsv41flash-agentx-client-policy.json diff --git a/benchmarks/srt-slurm/phase1/runtime-lock.json b/benchmarks/multi_node/srt-slurm-recipes/configs/prepared-runtime-lock.json similarity index 100% rename from benchmarks/srt-slurm/phase1/runtime-lock.json rename to benchmarks/multi_node/srt-slurm-recipes/configs/prepared-runtime-lock.json diff --git a/benchmarks/srt-slurm/phase1/h100-dsv41flash.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/vllm/h100-fp4/agentx/agg-tp8-dspark5.yaml similarity index 100% rename from benchmarks/srt-slurm/phase1/h100-dsv41flash.yaml rename to benchmarks/multi_node/srt-slurm-recipes/dsv41flash/vllm/h100-fp4/agentx/agg-tp8-dspark5.yaml diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index 5bafdcb415..6d9fb8849c 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -8042,10 +8042,10 @@ glm5.1-fp8-b200-tilert-agentic: - "DECODE_NODES=1" # H100 AgentX arm for DeepSeek-V4.1-Flash. H100 is not in the upstream hardware -# table (h200, gb200, gb300, mi350x are). It needs its own script, not the -# shared symlink: the sparse attention indexer's +# table (h200, gb200, gb300, mi350x are). Its own serving recipe accounts +# for the sparse attention indexer's # [max-num-batched-tokens, max-model-len] buffer is 16 GiB at the shared flags' -# 8192 batched tokens and OOMs an 80 GB card at concurrency 1, so the script +# 8192 batched tokens and OOMs an 80 GB card at concurrency 1, so the recipe # caps batched tokens at 4096. Concurrency is bounded by the measured KV # ceiling rather than a guess; see the search-space comment. dsv41flash-fp4-h100-vllm-agentic-dspark: @@ -8059,10 +8059,10 @@ dsv41flash-fp4-h100-vllm-agentic-dspark: execution: runtime: srt-slurm contract-version: 1 - recipe: benchmarks/srt-slurm/phase1/h100-dsv41flash.yaml + recipe: benchmarks/multi_node/srt-slurm-recipes/dsv41flash/vllm/h100-fp4/agentx/agg-tp8-dspark5.yaml profile: runners/srt-slurm/h100-phase1.yaml - runtime-lock: benchmarks/srt-slurm/phase1/runtime-lock.json - client-policy: benchmarks/srt-slurm/phase1/client-policy.json + runtime-lock: benchmarks/multi_node/srt-slurm-recipes/configs/prepared-runtime-lock.json + client-policy: benchmarks/multi_node/srt-slurm-recipes/configs/dsv41flash-agentx-client-policy.json scenarios: agentic-coding: - dram-utilization: 0.80 diff --git a/docs/srt-slurm-phase1.md b/docs/srt-slurm-phase1.md index c112abd6d9..57d978e1a9 100644 --- a/docs/srt-slurm-phase1.md +++ b/docs/srt-slurm-phase1.md @@ -27,6 +27,8 @@ flowchart LR U --> I ``` +The recipe follows the existing YAML hierarchy at [`agg-tp8-dspark5.yaml`](../benchmarks/multi_node/srt-slurm-recipes/dsv41flash/vllm/h100-fp4/agentx/agg-tp8-dspark5.yaml). Its runtime pin and client policy live in the same tree’s `configs/` directory: [`prepared-runtime-lock.json`](../benchmarks/multi_node/srt-slurm-recipes/configs/prepared-runtime-lock.json) and [`dsv41flash-agentx-client-policy.json`](../benchmarks/multi_node/srt-slurm-recipes/configs/dsv41flash-agentx-client-policy.json). The retained Bash recipe remains at `benchmarks/single_node/agentic/dsv41flash_fp4_h100_vllm_mtp.sh`. + ## Call and file map ```mermaid @@ -36,7 +38,7 @@ flowchart TD F --> P[infx.srt_slurm.launch.prepare] P --> CP[infx.benchmarks.prepare.prepare] P --> R[infx.srt_slurm.render.render_recipe] - R --> Y[benchmarks/srt-slurm/phase1/h100-dsv41flash.yaml] + R --> Y["dsv41flash/vllm/h100-fp4/agentx/
agg-tp8-dspark5.yaml"] R --> H[runners/srt-slurm/h100-phase1.yaml] P --> NP[srtctl prepare] F --> X[infx.srt_slurm.launch.execute] diff --git a/docs/srt-slurm-phase1_zh.md b/docs/srt-slurm-phase1_zh.md index 175ef68f4f..f857c2cf2b 100644 --- a/docs/srt-slurm-phase1_zh.md +++ b/docs/srt-slurm-phase1_zh.md @@ -27,6 +27,8 @@ flowchart LR U --> I ``` +配方沿用现有 YAML 层级,位于 [`agg-tp8-dspark5.yaml`](../benchmarks/multi_node/srt-slurm-recipes/dsv41flash/vllm/h100-fp4/agentx/agg-tp8-dspark5.yaml)。运行时 pin 与客户端策略位于同一配方树的 `configs/` 下:[`prepared-runtime-lock.json`](../benchmarks/multi_node/srt-slurm-recipes/configs/prepared-runtime-lock.json) 和 [`dsv41flash-agentx-client-policy.json`](../benchmarks/multi_node/srt-slurm-recipes/configs/dsv41flash-agentx-client-policy.json)。旧 Bash 配方仍位于 `benchmarks/single_node/agentic/dsv41flash_fp4_h100_vllm_mtp.sh`。 + ## 调用与文件关系 ```mermaid @@ -36,7 +38,7 @@ flowchart TD F --> P[infx.srt_slurm.launch.prepare] P --> CP[infx.benchmarks.prepare.prepare] P --> R[infx.srt_slurm.render.render_recipe] - R --> Y[benchmarks/srt-slurm/phase1/h100-dsv41flash.yaml] + R --> Y["dsv41flash/vllm/h100-fp4/agentx/
agg-tp8-dspark5.yaml"] R --> H[runners/srt-slurm/h100-phase1.yaml] P --> NP[srtctl prepare] F --> X[infx.srt_slurm.launch.execute] diff --git a/utils/test_native_pilot.py b/utils/test_native_pilot.py index cd8e5a11f7..4f3139c54b 100644 --- a/utils/test_native_pilot.py +++ b/utils/test_native_pilot.py @@ -22,9 +22,9 @@ def inputs(tmp_path): tmp_path = tmp_path.resolve() root = tmp_path / "checkout" paths = ( - "benchmarks/srt-slurm/phase1/h100-dsv41flash.yaml", + "benchmarks/multi_node/srt-slurm-recipes/dsv41flash/vllm/h100-fp4/agentx/agg-tp8-dspark5.yaml", "runners/srt-slurm/h100-phase1.yaml", - "benchmarks/srt-slurm/phase1/client-policy.json", + "benchmarks/multi_node/srt-slurm-recipes/configs/dsv41flash-agentx-client-policy.json", "golden_al_distribution/dsv41flash_dspark.yaml", ) fixture_names = ("recipe.yaml", "profile.yaml", "client-policy.json", "golden.yaml") From 14d56f1bbf8f3c867ea79ae97a2f716f304aaaa2 Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sat, 19 Sep 2026 20:02:02 -0400 Subject: [PATCH 09/16] fix: bind pilot topology and inspect H100 assets through CI MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 绑定阶段 1 引擎实际并行拓扑,增加通过 CI 检查 H100 共享资源的显式入口,并同步中英文文档。 --- .github/workflows/e2e-tests.yml | 55 +++++++++ docs/srt-slurm-phase1.md | 4 + docs/srt-slurm-phase1_zh.md | 4 + infx/srt_slurm/provision.py | 118 +++++++++++++++++++ infx/srt_slurm/render.py | 16 +++ runners/srt-slurm/h100-phase1-provision.json | 11 ++ utils/test_native_pilot.py | 44 +++++++ utils/test_native_provision.py | 56 +++++++++ 8 files changed, 308 insertions(+) create mode 100644 infx/srt_slurm/provision.py create mode 100644 runners/srt-slurm/h100-phase1-provision.json create mode 100644 utils/test_native_provision.py diff --git a/.github/workflows/e2e-tests.yml b/.github/workflows/e2e-tests.yml index e9149bf9e0..08d563c7e9 100644 --- a/.github/workflows/e2e-tests.yml +++ b/.github/workflows/e2e-tests.yml @@ -8,6 +8,12 @@ permissions: on: # zizmor: ignore[concurrency-limits] workflow_dispatch: inputs: + phase1-site-operation: + description: "Inspect H100 shared assets without submitting a benchmark" + required: false + type: choice + options: [none, inspect] + default: none generate-cli-command: description: "Command passed to generate matrix script" required: false @@ -110,6 +116,11 @@ on: # zizmor: ignore[concurrency-limits] MODAL_TOKEN_SECRET: required: false inputs: + phase1-site-operation: + description: "Inspect H100 shared assets without submitting a benchmark" + required: false + type: string + default: none generate-cli-command: description: "Command passed to generate matrix script" required: false @@ -205,7 +216,51 @@ on: # zizmor: ignore[concurrency-limits] default: "[]" jobs: + phase1-site-route: + name: Queue Phase 1 site inspection + if: ${{ inputs.phase1-site-operation == 'inspect' }} + runs-on: ubuntu-latest + outputs: + queue-token: ${{ steps.token.outputs.result }} + steps: + - id: token + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + with: + result-encoding: string + script: | + return require('crypto').createHash('sha256') + .update(`${context.runId}:${process.env.GITHUB_RUN_ATTEMPT}:phase1-site`) + .digest('hex').slice(0, 32); + phase1-site: + if: ${{ inputs.phase1-site-operation == 'inspect' }} + needs: phase1-site-route + name: Phase 1 shared H100 asset inventory + # The shared Linux paths can only be inspected from this cluster's login runners. + runs-on: ${{ fromJSON(format('["self-hosted","cluster:h100-dgxc","nodes:1","ci-job-1.000-{0}","ci-attempt-{1}"]', needs.phase1-site-route.outputs.queue-token, github.run_attempt)) }} + timeout-minutes: 20 + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + ref: ${{ github.workflow_sha }} + persist-credentials: false + - uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1 + - name: Inspect explicit shared paths + # H100 login runners do not provide an ambient python executable. + shell: uv run --locked --no-dev --python 3.12 python {0} # zizmor: ignore[misfeature] + run: | + import sys + from infx.srt_slurm.provision import main + sys.argv = ["provision", "--config", "runners/srt-slurm/h100-phase1-provision.json", "--output", "phase1-provision-report"] + raise SystemExit(main()) + - name: Preserve provisioning evidence + if: ${{ always() }} + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: phase1-provision-${{ github.run_id }}-${{ github.run_attempt }} + path: phase1-provision-report/ + if-no-files-found: error get-jobs: + if: ${{ !inputs.phase1-site-operation || inputs.phase1-site-operation == 'none' }} name: get-jobs runs-on: ubuntu-latest outputs: diff --git a/docs/srt-slurm-phase1.md b/docs/srt-slurm-phase1.md index 57d978e1a9..59409ce157 100644 --- a/docs/srt-slurm-phase1.md +++ b/docs/srt-slurm-phase1.md @@ -58,6 +58,10 @@ Preparation records the actual installed native/wrapper/client files, interprete ## Provisioning before the first GPU run +The existing E2E dispatch accepts `phase1-site-operation: inspect`. It checks the explicit paths/revisions in `runners/srt-slurm/h100-phase1-provision.json` on an H100 login runner and preserves `inventory.json` as a run/attempt artifact. This operation enters the existing priority queue with `nodes:1`, submits no Slurm allocation, and does not modify the shared model or trace caches. The configuration comes from the retained H100 baseline; the report establishes which paths actually exist before runtime installation. Missing images, snapshots or weight shards fail inspection. An inventory is not hardware qualification or a deployed-reader declaration. + +The renderer also binds the actual engine TP/PP/context/data-parallel arguments to the requested topology before native preparation. Conflicting underscore/hyphen aliases, noninteger sizes and enabled expert parallelism fail closed. + Provision on shared Linux storage visible to the H100 login host and compute container. Do not reuse the macOS test environments. The native, wrapper and selected client interpreters, their standard libraries, installed distributions, native source, prepared bundles and client caches need explicit same-path mounts. Mount roots must be canonical paths; symlink aliases are rejected. Python 3.12 is required for native/wrapper execution and Python 3.11 for the pinned AgentX child. Outputs and writable caches must stay outside `/workspace`. The model's HF snapshot must retain access to its sibling blob directory. The native model argument preserves that full-cache mount. 1. Install the pinned native source using its committed `uv.lock` and a noneditable environment (`uv sync --frozen --no-editable --no-dev --python 3.12`). Keep that source checkout clean. Preserve the hashed Linux-built wheel and its build-tool constraints: `uv.lock` freezes runtime dependencies but does not pin the upstream Hatch build dependencies. If rebuilding, fetch and verify NVIDIA’s `v2.2.1` tag at `984180e5b8755aef85e9995048b5a16cb5336bce` to retain the same hatch-vcs version lineage. diff --git a/docs/srt-slurm-phase1_zh.md b/docs/srt-slurm-phase1_zh.md index f857c2cf2b..16af665b4b 100644 --- a/docs/srt-slurm-phase1_zh.md +++ b/docs/srt-slurm-phase1_zh.md @@ -58,6 +58,10 @@ flowchart TD ## 首次 GPU 运行前的部署 +现有 E2E 手动调度支持 `phase1-site-operation: inspect`,在 H100 登录 runner 上检查 `runners/srt-slurm/h100-phase1-provision.json` 显式提供的路径和 revision,并将 `inventory.json` 保存为绑定运行及 attempt 的 artifact。此操作以 `nodes:1` 进入现有优先级队列,不提交 Slurm allocation,也不修改共享模型或 trace 缓存。配置来自保留的 H100 基线;检查报告用于在安装运行环境之前确认实际存在的路径。镜像、快照或权重分片缺失会使检查失败。资源清单不代表硬件验收完成,也不代表 reader 已部署。 + +渲染器还会在原生准备步骤之前,将引擎实际 TP/PP/上下文/数据并行参数与请求的拓扑绑定。下划线与连字符别名冲突、非整数并行度或启用专家并行都会被拒绝。 + 在 H100 登录节点与计算容器均可访问的共享 Linux 存储上部署,不复用 macOS 测试环境。原生、wrapper 及所选客户端的 Python 解释器、标准库、已安装依赖、原生源代码、prepared bundle 与客户端缓存均需明确的同路径挂载。挂载根目录必须是规范路径,不接受符号链接别名。原生及 wrapper 使用 Python 3.12,固定的 AgentX 子进程使用 Python 3.11。输出与可写缓存不得放在 `/workspace` 下。HF 模型 snapshot 必须仍能访问相邻 blob 目录;原生模型参数会保留完整缓存路径。 1. 使用原生源代码提交中的 `uv.lock` 安装非 editable 环境:`uv sync --frozen --no-editable --no-dev --python 3.12`。保持该源码 checkout 干净。保留带哈希的 Linux wheel 及构建工具约束:`uv.lock` 固定运行依赖,但未固定上游 Hatch 构建依赖。重新构建时,获取并核实 NVIDIA 的 `v2.2.1` tag 指向 `984180e5b8755aef85e9995048b5a16cb5336bce`,保留相同 hatch-vcs 版本谱系。 diff --git a/infx/srt_slurm/provision.py b/infx/srt_slurm/provision.py new file mode 100644 index 0000000000..eab87f0c47 --- /dev/null +++ b/infx/srt_slurm/provision.py @@ -0,0 +1,118 @@ +"""Inspect and provision explicit shared H100 assets on a CI login runner.""" + +from __future__ import annotations + +import argparse +import platform +import shutil +import subprocess +from pathlib import Path +from typing import Any, Literal + +from pydantic import BaseModel, ConfigDict, Field, field_validator + +from infx.benchmarks.common import read_json, write_json + + +class ProvisionConfig(BaseModel): + model_config = ConfigDict(extra="forbid", strict=True) + + schema_version: Literal[1] + shared_root: str + hub_cache: str + image_path: str + image_reference: str + model_repository: str + model_revision: str = Field(pattern=r"^[0-9a-f]{40}$") + dataset_repository: str + dataset_revision: str = Field(pattern=r"^[0-9a-f]{40}$") + + @field_validator("shared_root", "hub_cache", "image_path") + @classmethod + def absolute_path(cls, value: str) -> str: + path = Path(value) + if not path.is_absolute() or path.resolve().is_relative_to("/workspace"): + raise ValueError("provisioning requires explicit shared paths outside /workspace") + return value + + @field_validator("model_repository", "dataset_repository") + @classmethod + def repository_name(cls, value: str) -> str: + if len(value.split("/")) != 2 or any(part in {"", ".", ".."} for part in value.split("/")): + raise ValueError("expected owner/name Hugging Face repository") + return value + + +def snapshot(config: ProvisionConfig, *, dataset: bool) -> Path: + repository = config.dataset_repository if dataset else config.model_repository + revision = config.dataset_revision if dataset else config.model_revision + prefix = "datasets--" if dataset else "models--" + return ( + Path(config.hub_cache) / (prefix + repository.replace("/", "--")) / "snapshots" / revision + ) + + +def inspect_assets(config: ProvisionConfig) -> dict[str, Any]: + """Report actual existing paths without altering shared caches or taking an allocation.""" + paths = { + "shared_parent": Path(config.shared_root).parent, + "hub_cache": Path(config.hub_cache), + "image": Path(config.image_path), + "model_snapshot": snapshot(config, dataset=False), + "dataset_snapshot": snapshot(config, dataset=True), + } + entries = {} + for name, path in paths.items(): + entries[name] = { + "path": str(path), + "canonical_path": str(path.resolve()), + "exists": path.exists(), + "is_directory": path.is_dir(), + "size": path.stat().st_size if path.is_file() else None, + } + model = paths["model_snapshot"] + indexes = sorted(model.glob("*.safetensors.index.json")) + sorted( + model.glob("*.bin.index.json") + ) + missing_shards = [] + for index in indexes: + for name in set(read_json(index).get("weight_map", {}).values()): + relative = Path(name) + if relative.is_absolute() or ".." in relative.parts: + raise ValueError("model index contains an unsafe shard path") + if not (model / relative).is_file(): + missing_shards.append(name) + ready = ( + all(value["exists"] for value in entries.values()) and bool(indexes) and not missing_shards + ) + return { + "schema_version": 1, + "platform": platform.platform(), + "assets_present": ready, + "paths": entries, + "model_indexes": [path.name for path in indexes], + "missing_model_shards": sorted(missing_shards), + "slurm_tools": { + name: shutil.which(name) for name in ("sbatch", "squeue", "sacct", "scancel") + }, + "qualification_complete": False, + } + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--config", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + config = ProvisionConfig.model_validate(read_json(args.config)) + report = inspect_assets(config) + report["head_sha"] = subprocess.run( + ["git", "rev-parse", "HEAD"], capture_output=True, text=True, check=True + ).stdout.strip() + write_json(args.output / "inventory.json", report) + print(f"Prepared asset inventory written; assets_present={report['assets_present']}") + return 0 if report["assets_present"] else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/infx/srt_slurm/render.py b/infx/srt_slurm/render.py index d7de84bc73..c56155f65e 100644 --- a/infx/srt_slurm/render.py +++ b/infx/srt_slurm/render.py @@ -212,6 +212,22 @@ def render_recipe( if (role["nodes"], role["workers"], role["gpus"]) != (1, 1, 8): raise ValueError("pilot requires one physical node and one TP8 worker") args = role["args"] + topology = { + "tensor-parallel-size": job.row.tp, + "pipeline-parallel-size": job.row.pp, + "decode-context-parallel-size": job.row.dcp_size, + "prefill-context-parallel-size": job.row.pcp_size, + "data-parallel-size": 1, + } + normalized = {name.replace("_", "-"): value for name, value in args.items()} + if len(normalized) != len(args): + raise ValueError("conflicting engine argument aliases") + for name, expected in topology.items(): + value = normalized.get(name, None if name == "tensor-parallel-size" else 1) + if type(value) is not int or value != expected: + raise ValueError(f"engine {name} differs from the requested aggregate topology") + if normalized.get("enable-expert-parallel", False) is not False: + raise ValueError("aggregate pilot requires expert parallelism disabled") args["max-num-seqs"] = 2 * job.row.conc args["max-cudagraph-capture-size"] = min(2048, 1 << (12 * job.row.conc - 1).bit_length()) spec = args["speculative-config"] diff --git a/runners/srt-slurm/h100-phase1-provision.json b/runners/srt-slurm/h100-phase1-provision.json new file mode 100644 index 0000000000..368f4eaef6 --- /dev/null +++ b/runners/srt-slurm/h100-phase1-provision.json @@ -0,0 +1,11 @@ +{ + "schema_version": 1, + "shared_root": "/mnt/nfs/sa-shared/gharunners/inferencex-native", + "hub_cache": "/mnt/nfs/sa-shared/gharunners/hf-hub-cache", + "image_path": "/mnt/nfs/lustre/containers/vllm_vllm-openai_nightly-cd10ed6f9f6b37a8ace9cf380007e66fe12ec0c3.sqsh", + "image_reference": "vllm/vllm-openai:nightly-cd10ed6f9f6b37a8ace9cf380007e66fe12ec0c3", + "model_repository": "deepseek-ai/DeepSeek-V4.1-Flash", + "model_revision": "dba1be0a40aa45a94ad051997016db3960a90277", + "dataset_repository": "semianalysisai/cc-traces-weka-062126", + "dataset_revision": "23f152f6f0f9399a85901b89a6458def0ef16729" +} diff --git a/utils/test_native_pilot.py b/utils/test_native_pilot.py index 4f3139c54b..1c9c534818 100644 --- a/utils/test_native_pilot.py +++ b/utils/test_native_pilot.py @@ -533,3 +533,47 @@ def value(flag, command=argv): ) assert tampered.returncode != 0 assert "Prepared input changed: config.yaml" in tampered.stderr + + +@pytest.mark.parametrize( + "flag,value", + [ + ("tensor-parallel-size", 4), + ("pipeline-parallel-size", 2), + ("decode-context-parallel-size", 2), + ("prefill-context-parallel-size", 2), + ("data-parallel-size", 2), + ("enable-expert-parallel", True), + ("tensor-parallel-size", "8"), + ], +) +def test_allocation_does_not_authorize_mislabeled_engine_topology(inputs, flag, value): + import yaml + + root, row, scheduling, site = inputs + path = root / row["execution"]["recipe"] + recipe = load_mapping(path) + recipe["roles"]["agg"]["args"][flag] = value + path.write_text(yaml.safe_dump(recipe)) + job = parse_job(row, root, scheduling) + policy = ClientPolicy.model_validate( + load_mapping(root / row["execution"]["client-policy"]) + ) + with pytest.raises(ValueError, match="topology|expert parallelism"): + render_recipe(job, root, site, policy, root / "spec.json", root / "output") + + +def test_conflicting_engine_argument_aliases_are_rejected(inputs): + import yaml + + root, row, scheduling, site = inputs + path = root / row["execution"]["recipe"] + recipe = load_mapping(path) + recipe["roles"]["agg"]["args"]["tensor_parallel_size"] = 4 + path.write_text(yaml.safe_dump(recipe)) + job = parse_job(row, root, scheduling) + policy = ClientPolicy.model_validate( + load_mapping(root / row["execution"]["client-policy"]) + ) + with pytest.raises(ValueError, match="conflicting engine argument aliases"): + render_recipe(job, root, site, policy, root / "spec.json", root / "output") diff --git a/utils/test_native_provision.py b/utils/test_native_provision.py new file mode 100644 index 0000000000..cacc54fbf6 --- /dev/null +++ b/utils/test_native_provision.py @@ -0,0 +1,56 @@ +"""Provisioning works from observed shared assets, without inventing site identities.""" + +import json +from pathlib import Path + +import pytest + +from infx.srt_slurm.provision import ProvisionConfig, inspect_assets + + +def asset_config(tmp_path: Path) -> ProvisionConfig: + hub = tmp_path / "hub" + model = hub / "models--example--model/snapshots" / ("a" * 40) + traces = hub / "datasets--example--traces/snapshots" / ("b" * 40) + model.mkdir(parents=True) + traces.mkdir(parents=True) + (model / "model.safetensors.index.json").write_text( + json.dumps({"weight_map": {"weight": "shard.safetensors"}}) + ) + (model / "shard.safetensors").write_bytes(b"weights") + (tmp_path / "image.sqsh").write_bytes(b"squash") + return ProvisionConfig( + schema_version=1, + shared_root=str(tmp_path / "prepared"), + hub_cache=str(hub), + image_path=str(tmp_path / "image.sqsh"), + image_reference="example/image:version", + model_repository="example/model", + model_revision="a" * 40, + dataset_repository="example/traces", + dataset_revision="b" * 40, + ) + + +def test_inspection_observes_missing_payload_and_creates_no_shared_root(tmp_path): + config = asset_config(tmp_path) + report = inspect_assets(config) + assert report["assets_present"] is True + assert report["qualification_complete"] is False + assert report["paths"]["image"]["size"] == 6 + model = Path(report["paths"]["model_snapshot"]["path"]) + (model / "shard.safetensors").unlink() + report = inspect_assets(config) + assert report["assets_present"] is False + assert report["missing_model_shards"] == ["shard.safetensors"] + assert not Path(config.shared_root).exists() + + +def test_inspection_rejects_shard_path_escape(tmp_path): + config = asset_config(tmp_path) + model = Path(config.hub_cache) / "models--example--model/snapshots" / ("a" * 40) + (model / "model.safetensors.index.json").write_text( + json.dumps({"weight_map": {"weight": "../escape"}}) + ) + with pytest.raises(ValueError, match="unsafe shard path"): + inspect_assets(config) From 3ebcabe65c378bcd78ae05b05ac9c8ce31f1305b Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sat, 19 Sep 2026 20:56:50 -0400 Subject: [PATCH 10/16] feat: provision and qualify the native H100 pilot without publication MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 准备并验收原生 H100 试点运行环境;PR 诊断与发布分离,保留完整九点测量契约,修复中断清理并验证监听端口归属。 Validation: 1920 producer tests and 207 subtests passed; installed native boundary covers all nine points. Ruff, MCP compatibility, workflow actionlint and authenticated Zizmor passed. Real H100 qualification remains pending. --- .github/workflows/benchmark-tmpl.yml | 130 +-- .github/workflows/e2e-tests.yml | 66 +- .github/workflows/run-sweep.yml | 63 +- .../configs/prepared-runtime-lock.json | 5 +- .../vllm/h100-fp4/agentx/agg-tp8-dspark5.yaml | 3 + docs/srt-slurm-phase1.md | 44 +- docs/srt-slurm-phase1_zh.md | 44 +- infx/benchmarks/agentx.py | 8 +- infx/results/publication_receipt.py | 6 + infx/srt_slurm/launch.py | 43 +- infx/srt_slurm/provision.py | 20 +- infx/srt_slurm/provision_runtime.py | 766 ++++++++++++++++++ infx/srt_slurm/qualification.py | 372 +++++++++ infx/srt_slurm/qualify_cancellation.py | 640 +++++++++++++++ infx/srt_slurm/render.py | 14 +- infx/srt_slurm/workflow.py | 69 +- infx/workflows/phase1_publication.py | 3 + infx/workflows/phase1_record.py | 7 + infx/workflows/sweep_runs.py | 3 + perf-changelog.yaml | 9 + utils/test_native_provision.py | 32 +- utils/test_native_qualification.py | 537 ++++++++++++ utils/test_phase1_receipt_control.py | 18 +- utils/test_provision_runtime.py | 411 ++++++++++ utils/test_publication_receipt.py | 20 + utils/test_qualify_cancellation.py | 496 ++++++++++++ 26 files changed, 3701 insertions(+), 128 deletions(-) create mode 100644 infx/srt_slurm/provision_runtime.py create mode 100644 infx/srt_slurm/qualification.py create mode 100644 infx/srt_slurm/qualify_cancellation.py create mode 100644 utils/test_native_qualification.py create mode 100644 utils/test_provision_runtime.py create mode 100644 utils/test_qualify_cancellation.py diff --git a/.github/workflows/benchmark-tmpl.yml b/.github/workflows/benchmark-tmpl.yml index 231fa5b91b..a65d315cb1 100644 --- a/.github/workflows/benchmark-tmpl.yml +++ b/.github/workflows/benchmark-tmpl.yml @@ -13,6 +13,11 @@ on: description: "Benchmark configuration as JSON" required: true type: string + native-qualification: + description: "Run native PR qualification with diagnostics only; never publish measurements" + required: false + type: boolean + default: false klaud-run: description: "Run Klaud Cold benchmarks at background priority" required: false @@ -95,6 +100,7 @@ on: type: string default: "" env: + NATIVE_CHECKOUT_ROOT: ${{ fromJSON(inputs.config).execution.runtime == 'srt-slurm' && format('{0}/native-candidate-{1}-{2}-{3}', github.workspace, github.run_id, github.run_attempt, inputs.queue-token) || github.workspace }} PORT: '8888' INFMAX_CONTAINER_WORKSPACE: '/workspace' DSV41_MIN_CUDAGRAPH_CAPTURE_SIZE: '1' @@ -203,7 +209,8 @@ jobs: ) || format('[{0}]', toJSON(inputs.runner)) ) }} - timeout-minutes: 500 + # Native preparation hashes immutable shared assets before the bounded eight-hour wait. + timeout-minutes: ${{ fromJSON(inputs.config).execution.runtime == 'srt-slurm' && 600 || 500 }} name: >- ${{ inputs.klaud-run && 'klaud | ' || '' }}p${{ inputs.priority }} | ${{ fromJSON(inputs.config).model-prefix }} ${{ fromJSON(inputs.config).precision }} ${{ inputs.runner }} ${{ fromJSON(inputs.config).framework == 'sglang' && 'sgl' || fromJSON(inputs.config).framework == 'dynamo-sglang' && 'dyn-sgl' || fromJSON(inputs.config).framework == 'sglang-disagg' && 'sgl-disagg' || fromJSON(inputs.config).framework }} TP${{ fromJSON(inputs.config).tp }}${{ format('{0}', fromJSON(inputs.config).pp) != '' && format('{0}', fromJSON(inputs.config).pp) != '1' && format('/PP{0}', fromJSON(inputs.config).pp) || '' }}${{ format('{0}', fromJSON(inputs.config).dcp-size) != '' && format('{0}', fromJSON(inputs.config).dcp-size) != '1' && format('/DCP{0}', fromJSON(inputs.config).dcp-size) || '' }}${{ format('{0}', fromJSON(inputs.config).pcp-size) != '' && format('{0}', fromJSON(inputs.config).pcp-size) != '1' && format('/PCP{0}', fromJSON(inputs.config).pcp-size) || '' }}${{ format('{0}', fromJSON(inputs.config).ep) != '' && format('{0}', fromJSON(inputs.config).ep) != '1' && format('/EP{0}', fromJSON(inputs.config).ep) || '' }}${{ inputs.dp-attn && '/DPA' || '' }} @@ -271,6 +278,8 @@ jobs: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: + # Each native attempt has a fresh Git directory; stale parent locks/submodules are never reused. + path: ${{ fromJSON(inputs.config).execution.runtime == 'srt-slurm' && format('native-candidate-{0}-{1}-{2}', github.run_id, github.run_attempt, inputs.queue-token) || '.' }} fetch-depth: 0 ref: ${{ inputs.ref || github.sha }} clean: true @@ -298,21 +307,32 @@ jobs: NATIVE_AGENTX_FAST: ${{ toJSON(inputs.agentx-fast) }} NATIVE_EVAL_LIMIT: ${{ inputs.eval-limit }} NATIVE_REQUIRE_POWER: ${{ toJSON(inputs.require-power) }} + NATIVE_PURPOSE: ${{ inputs.native-qualification && 'pr-qualification' || 'publication' }} + NATIVE_PREPARED_SITE_JSON: ${{ vars.INFX_H100_PHASE1_PREPARED_SITE_JSON }} NATIVE_SITE_JSON: ${{ vars.INFX_H100_PHASE1_SITE_JSON }} NATIVE_READER_REVISION: ${{ vars.INFX_PHASE1_READER_REVISION }} NATIVE_COLLECTOR_REVISION: ${{ vars.INFX_PHASE1_COLLECTOR_REVISION }} # H100 runners lack ambient python; uv supplies the explicit managed interpreter. + working-directory: ${{ env.NATIVE_CHECKOUT_ROOT }} shell: uv run --locked --python 3.12 python {0} # zizmor: ignore[misfeature] run: | from infx.srt_slurm.workflow import main raise SystemExit(main()) - name: Upload native execution provenance - if: ${{ always() && fromJSON(inputs.config).execution.runtime == 'srt-slurm' && env.NATIVE_POINT_ID != '' }} + if: ${{ always() && !inputs.native-qualification && fromJSON(inputs.config).execution.runtime == 'srt-slurm' && env.NATIVE_POINT_ID != '' }} uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 with: name: native-execution-${{ env.NATIVE_POINT_ID }} - path: native-execution/** + path: ${{ env.NATIVE_CHECKOUT_ROOT }}/native-execution/** + if-no-files-found: error + + - name: Upload nonpublishing native qualification evidence + if: ${{ always() && inputs.native-qualification && fromJSON(inputs.config).execution.runtime == 'srt-slurm' && env.NATIVE_POINT_ID != '' }} + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: native-qualification-${{ env.NATIVE_POINT_ID }} + path: ${{ env.NATIVE_CHECKOUT_ROOT }}/native-qualification/** if-no-files-found: error - name: Launch job script @@ -379,7 +399,7 @@ jobs: fi - name: Process result - if: ${{ always() && env.RESULT_FILENAME != '' && !inputs.eval-only && inputs.scenario-type != 'agentic-coding' }} + if: ${{ !inputs.native-qualification && always() && env.RESULT_FILENAME != '' && !inputs.eval-only && inputs.scenario-type != 'agentic-coding' }} env: RUNNER_TYPE: ${{ inputs.runner }} run: | @@ -390,101 +410,101 @@ jobs: python3 utils/process_result.py - name: Upload result - if: ${{ success() && env.RESULT_FILENAME != '' && !inputs.eval-only && inputs.scenario-type != 'agentic-coding' }} + if: ${{ !inputs.native-qualification && success() && env.RESULT_FILENAME != '' && !inputs.eval-only && inputs.scenario-type != 'agentic-coding' }} uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 with: name: bmk_${{ env.RESULT_FILENAME }} path: agg_${{ env.RESULT_FILENAME }}.json - name: Upload agentic aggregated result - if: ${{ always() && inputs.scenario-type == 'agentic-coding' }} + if: ${{ !inputs.native-qualification && always() && inputs.scenario-type == 'agentic-coding' }} uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 with: name: bmk_agentic_${{ env.RESULT_FILENAME }} - path: ${{ env.RESULT_FILENAME }}.json + path: ${{ env.NATIVE_CHECKOUT_ROOT }}/${{ env.RESULT_FILENAME }}.json - name: Upload agentic raw results - if: ${{ always() && inputs.scenario-type == 'agentic-coding' }} + if: ${{ !inputs.native-qualification && always() && inputs.scenario-type == 'agentic-coding' }} uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 with: name: agentic_${{ env.RESULT_FILENAME }} path: | - results/** - !results/aiperf_artifacts/inputs.json - !results/aiperf_artifacts/profile_export_raw.jsonl + ${{ env.NATIVE_CHECKOUT_ROOT }}/results/** + !${{ env.NATIVE_CHECKOUT_ROOT }}/results/aiperf_artifacts/inputs.json + !${{ env.NATIVE_CHECKOUT_ROOT }}/results/aiperf_artifacts/profile_export_raw.jsonl if-no-files-found: ignore - name: Upload server logs - if: always() + if: ${{ always() && !inputs.native-qualification }} uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 with: name: ${{ inputs.eval-only && 'eval_server_logs_' || 'server_logs_' }}${{ env.RESULT_FILENAME }} # fixed-seq writes server.log at root; agentic writes logs under results/. path: | - server.log - results/*.log - results/*_config.json - results/native/**/*.log - results/native/**/*.json - results/native/**/*.yaml + ${{ env.NATIVE_CHECKOUT_ROOT }}/server.log + ${{ env.NATIVE_CHECKOUT_ROOT }}/results/*.log + ${{ env.NATIVE_CHECKOUT_ROOT }}/results/*_config.json + ${{ env.NATIVE_CHECKOUT_ROOT }}/results/native/**/*.log + ${{ env.NATIVE_CHECKOUT_ROOT }}/results/native/**/*.json + ${{ env.NATIVE_CHECKOUT_ROOT }}/results/native/**/*.yaml if-no-files-found: ignore - name: Upload GPU metrics - if: always() + if: ${{ always() && !inputs.native-qualification }} uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 with: name: ${{ inputs.eval-only && 'eval_gpu_metrics_' || 'gpu_metrics_' }}${{ env.RESULT_FILENAME }} path: | - gpu_metrics.csv - gpu_metrics*_context.json - gpu_metrics_energy_start.csv - gpu_metrics_energy_end.csv - gpu_metrics_identity.json - gpu_metrics_identity.csv - results/gpu_metrics*.csv - results/gpu_metrics*_context.json - results/gpu_metrics_identity.json + ${{ env.NATIVE_CHECKOUT_ROOT }}/gpu_metrics.csv + ${{ env.NATIVE_CHECKOUT_ROOT }}/gpu_metrics*_context.json + ${{ env.NATIVE_CHECKOUT_ROOT }}/gpu_metrics_energy_start.csv + ${{ env.NATIVE_CHECKOUT_ROOT }}/gpu_metrics_energy_end.csv + ${{ env.NATIVE_CHECKOUT_ROOT }}/gpu_metrics_identity.json + ${{ env.NATIVE_CHECKOUT_ROOT }}/gpu_metrics_identity.csv + ${{ env.NATIVE_CHECKOUT_ROOT }}/results/gpu_metrics*.csv + ${{ env.NATIVE_CHECKOUT_ROOT }}/results/gpu_metrics*_context.json + ${{ env.NATIVE_CHECKOUT_ROOT }}/results/gpu_metrics_identity.json if-no-files-found: ignore - name: Upload power audit bundle - if: ${{ always() && !inputs.eval-only }} + if: ${{ !inputs.native-qualification && always() && !inputs.eval-only }} uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 with: name: power_audit_${{ env.RESULT_FILENAME }} path: | - ${{ env.RESULT_FILENAME }}.json - agg_${{ env.RESULT_FILENAME }}.json - gpu_metrics.csv - gpu_metrics*_context.json - gpu_metrics_energy_start.csv - gpu_metrics_energy_end.csv - gpu_metrics_identity.json - gpu_metrics_identity.csv - power_validation_${{ env.RESULT_FILENAME }}.json - results/gpu_metrics*.csv - results/gpu_metrics*_context.json - results/gpu_metrics_identity.json - results/agentic_power_window.json - results/agentic_power_timezone_offset.txt - results/power_validation.json + ${{ env.NATIVE_CHECKOUT_ROOT }}/${{ env.RESULT_FILENAME }}.json + ${{ env.NATIVE_CHECKOUT_ROOT }}/agg_${{ env.RESULT_FILENAME }}.json + ${{ env.NATIVE_CHECKOUT_ROOT }}/gpu_metrics.csv + ${{ env.NATIVE_CHECKOUT_ROOT }}/gpu_metrics*_context.json + ${{ env.NATIVE_CHECKOUT_ROOT }}/gpu_metrics_energy_start.csv + ${{ env.NATIVE_CHECKOUT_ROOT }}/gpu_metrics_energy_end.csv + ${{ env.NATIVE_CHECKOUT_ROOT }}/gpu_metrics_identity.json + ${{ env.NATIVE_CHECKOUT_ROOT }}/gpu_metrics_identity.csv + ${{ env.NATIVE_CHECKOUT_ROOT }}/power_validation_${{ env.RESULT_FILENAME }}.json + ${{ env.NATIVE_CHECKOUT_ROOT }}/results/gpu_metrics*.csv + ${{ env.NATIVE_CHECKOUT_ROOT }}/results/gpu_metrics*_context.json + ${{ env.NATIVE_CHECKOUT_ROOT }}/results/gpu_metrics_identity.json + ${{ env.NATIVE_CHECKOUT_ROOT }}/results/agentic_power_window.json + ${{ env.NATIVE_CHECKOUT_ROOT }}/results/agentic_power_timezone_offset.txt + ${{ env.NATIVE_CHECKOUT_ROOT }}/results/power_validation.json if-no-files-found: ignore - name: Upload eval results (if any) - if: ${{ always() && (env.RUN_EVAL == 'true' || inputs.eval-only) }} + if: ${{ !inputs.native-qualification && always() && (env.RUN_EVAL == 'true' || inputs.eval-only) }} uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 with: name: eval_${{ env.RESULT_FILENAME }}_${{ env.EVAL_FRAMEWORK }}_${{ env.EVAL_SUITE }}_${{ github.run_attempt }} path: | - meta_env.json - results*.json - *_report.json - *_results.jsonl - *_artifacts.tar.gz - sample*.jsonl - agent_preds.json - predictions.jsonl - swebench_report_*.json - *.traj* + ${{ env.NATIVE_CHECKOUT_ROOT }}/meta_env.json + ${{ env.NATIVE_CHECKOUT_ROOT }}/results*.json + ${{ env.NATIVE_CHECKOUT_ROOT }}/*_report.json + ${{ env.NATIVE_CHECKOUT_ROOT }}/*_results.jsonl + ${{ env.NATIVE_CHECKOUT_ROOT }}/*_artifacts.tar.gz + ${{ env.NATIVE_CHECKOUT_ROOT }}/sample*.jsonl + ${{ env.NATIVE_CHECKOUT_ROOT }}/agent_preds.json + ${{ env.NATIVE_CHECKOUT_ROOT }}/predictions.jsonl + ${{ env.NATIVE_CHECKOUT_ROOT }}/swebench_report_*.json + ${{ env.NATIVE_CHECKOUT_ROOT }}/*.traj* if-no-files-found: ${{ inputs.eval-only && 'error' || 'ignore' }} - name: Verify eval scores diff --git a/.github/workflows/e2e-tests.yml b/.github/workflows/e2e-tests.yml index 08d563c7e9..4cf0fe900c 100644 --- a/.github/workflows/e2e-tests.yml +++ b/.github/workflows/e2e-tests.yml @@ -9,11 +9,16 @@ on: # zizmor: ignore[concurrency-limits] workflow_dispatch: inputs: phase1-site-operation: - description: "Inspect H100 shared assets without submitting a benchmark" + description: "H100 site preparation or disposable native cancellation qualification" required: false type: choice - options: [none, inspect] + options: [none, inspect, provision, cancel-startup, cancel-client] default: none + phase1-site-draft: + description: "Explicit shared site-draft.json path for cancellation qualification" + required: false + type: string + default: "" generate-cli-command: description: "Command passed to generate matrix script" required: false @@ -117,10 +122,15 @@ on: # zizmor: ignore[concurrency-limits] required: false inputs: phase1-site-operation: - description: "Inspect H100 shared assets without submitting a benchmark" + description: "H100 site preparation or disposable native cancellation qualification" required: false type: string default: none + phase1-site-draft: + description: "Explicit shared site-draft.json path for cancellation qualification" + required: false + type: string + default: "" generate-cli-command: description: "Command passed to generate matrix script" required: false @@ -217,47 +227,71 @@ on: # zizmor: ignore[concurrency-limits] jobs: phase1-site-route: - name: Queue Phase 1 site inspection - if: ${{ inputs.phase1-site-operation == 'inspect' }} + name: Queue Phase 1 site preparation + if: ${{ inputs.phase1-site-operation && inputs.phase1-site-operation != 'none' }} runs-on: ubuntu-latest outputs: queue-token: ${{ steps.token.outputs.result }} steps: - id: token uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + env: + SITE_OPERATION: ${{ inputs.phase1-site-operation }} with: result-encoding: string script: | + if (!['inspect', 'provision', 'cancel-startup', 'cancel-client'].includes(process.env.SITE_OPERATION)) { + throw new Error('Unknown Phase 1 site operation'); + } return require('crypto').createHash('sha256') .update(`${context.runId}:${process.env.GITHUB_RUN_ATTEMPT}:phase1-site`) .digest('hex').slice(0, 32); phase1-site: - if: ${{ inputs.phase1-site-operation == 'inspect' }} + if: ${{ inputs.phase1-site-operation && inputs.phase1-site-operation != 'none' }} needs: phase1-site-route - name: Phase 1 shared H100 asset inventory + name: Phase 1 H100 site and lifecycle qualification # The shared Linux paths can only be inspected from this cluster's login runners. runs-on: ${{ fromJSON(format('["self-hosted","cluster:h100-dgxc","nodes:1","ci-job-1.000-{0}","ci-attempt-{1}"]', needs.phase1-site-route.outputs.queue-token, github.run_attempt)) }} - timeout-minutes: 20 + timeout-minutes: 120 steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: - ref: ${{ github.workflow_sha }} + # Provision the same immutable tree the sweep will measure (PR merge SHA). + ref: ${{ inputs.ref || github.workflow_sha }} + path: phase1-site-${{ github.run_id }}-${{ github.run_attempt }} persist-credentials: false - uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1 - - name: Inspect explicit shared paths + - name: Prepare explicit shared paths + working-directory: phase1-site-${{ github.run_id }}-${{ github.run_attempt }} + env: + SITE_OPERATION: ${{ inputs.phase1-site-operation }} + SITE_DRAFT: ${{ inputs.phase1-site-draft }} # H100 login runners do not provide an ambient python executable. shell: uv run --locked --no-dev --python 3.12 python {0} # zizmor: ignore[misfeature] run: | - import sys - from infx.srt_slurm.provision import main - sys.argv = ["provision", "--config", "runners/srt-slurm/h100-phase1-provision.json", "--output", "phase1-provision-report"] - raise SystemExit(main()) + import os, sys + from pathlib import Path + operation = os.environ["SITE_OPERATION"] + if operation in {"cancel-startup", "cancel-client"}: + from infx.srt_slurm.qualify_cancellation import qualify + if not os.environ["SITE_DRAFT"]: + raise ValueError("Cancellation qualification requires phase1-site-draft") + mode = operation.removeprefix("cancel-") + walltime, observation = (300, 240) if mode == "startup" else (3600, 3300) + qualify(Path.cwd(), Path(os.environ["SITE_DRAFT"]), Path("phase1-provision-report"), + f"{os.environ['GITHUB_RUN_ID']}-{os.environ['GITHUB_RUN_ATTEMPT']}-{mode}", + mode=mode, walltime_seconds=walltime, + observation_timeout_seconds=observation, cleanup_timeout_seconds=180) + else: + from infx.srt_slurm.provision import main + sys.argv = ["provision", "--config", "runners/srt-slurm/h100-phase1-provision.json", "--output", "phase1-provision-report", "--operation", operation] + raise SystemExit(main()) - name: Preserve provisioning evidence if: ${{ always() }} uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 with: - name: phase1-provision-${{ github.run_id }}-${{ github.run_attempt }} - path: phase1-provision-report/ + name: phase1-site-${{ inputs.phase1-site-operation }}-${{ github.run_id }}-${{ github.run_attempt }} + path: phase1-site-${{ github.run_id }}-${{ github.run_attempt }}/phase1-provision-report/ if-no-files-found: error get-jobs: if: ${{ !inputs.phase1-site-operation || inputs.phase1-site-operation == 'none' }} diff --git a/.github/workflows/run-sweep.yml b/.github/workflows/run-sweep.yml index ae7c35c038..696afe87d0 100644 --- a/.github/workflows/run-sweep.yml +++ b/.github/workflows/run-sweep.yml @@ -233,6 +233,7 @@ jobs: ) ) outputs: + native-qualification: ${{ steps.qualification.outputs.native-qualification }} tooling-ref: ${{ steps.tooling.outputs.result }} search-space-config: ${{ steps.setup.outputs.search-space-config }} reuse-enabled: ${{ steps.setup.outputs.reuse-enabled }} @@ -391,6 +392,20 @@ jobs: --ref "${GITHUB_REF}" \ --workflow-id "run-sweep.yml" + - name: Select nonpublishing native qualification + id: qualification + env: + SWEEP_MATRIX: ${{ steps.setup.outputs.search-space-config }} + run: uv run --no-project --python 3.12 --with pydantic --with pyyaml python -m infx.srt_slurm.qualification plan + + - name: Retain nonpublication intent + if: steps.qualification.outputs.native-qualification == 'true' + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: native-qualification-run + path: qualification-intent.json + if-no-files-found: error + - name: Summarize reused source if: steps.setup.outputs.reuse-source-head-sha != '' uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 @@ -603,6 +618,7 @@ jobs: MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: + native-qualification: ${{ needs.setup.outputs.native-qualification == 'true' }} config: ${{ toJSON(matrix.config) }} klaud-run: *klaud-run runner: ${{ matrix.config.runner }} @@ -710,6 +726,7 @@ jobs: MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} with: + native-qualification: ${{ needs.setup.outputs.native-qualification == 'true' }} config: ${{ toJSON(matrix.config) }} klaud-run: *klaud-run runner: ${{ matrix.config.runner }} @@ -806,6 +823,44 @@ jobs: eval-conc: ${{ matrix.config['eval-conc'] }} scenario-type: agentic-coding + native-qualification-summary: + name: Nonpublishing H100 qualification (8 throughput + real eval) + needs: [setup, sweep-agentic, sweep-agentic-evals] + if: ${{ always() && needs.setup.outputs.native-qualification == 'true' }} + runs-on: ubuntu-latest + permissions: + contents: read + actions: read # Download this run's diagnostic-only point artifacts. + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + - uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1 + - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + pattern: native-qualification-* + path: qualification-artifacts + - name: Verify full nonpublishing qualification contracts + env: + SWEEP_MATRIX: ${{ needs.setup.outputs.search-space-config }} + THROUGHPUT_RESULT: ${{ needs.sweep-agentic.result }} + EVAL_RESULT: ${{ needs.sweep-agentic-evals.result }} + shell: uv run --locked --python 3.12 python {0} # zizmor: ignore[misfeature] + run: | + import os + from infx.srt_slurm.qualification import main + if os.environ['THROUGHPUT_RESULT'] != 'success' or os.environ['EVAL_RESULT'] != 'success': + raise ValueError('All throughput and real eval jobs must succeed') + import sys + sys.argv = ['qualification', 'summary', '--artifacts', 'qualification-artifacts'] + main() + - name: Retain qualification summary + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: native-qualification-summary + path: qualification-summary.json + if-no-files-found: error + collect-results: name: collect-results needs: @@ -821,7 +876,7 @@ jobs: ] if: >- ${{ - always() && + always() && needs.setup.outputs.native-qualification != 'true' && needs.setup.result == 'success' && ( needs.canary-sweep.result == 'success' || @@ -839,7 +894,7 @@ jobs: collect-evals: name: collect-evals needs: [sweep-evals, sweep-agentic-evals, sweep-multi-node-evals, sweep-multi-node-agentic-evals, setup] - if: ${{ always() && needs.setup.result != 'skipped' && (needs.sweep-evals.result != 'skipped' || needs.sweep-agentic-evals.result != 'skipped' || needs.sweep-multi-node-evals.result != 'skipped' || needs.sweep-multi-node-agentic-evals.result != 'skipped') }} + if: ${{ always() && needs.setup.outputs.native-qualification != 'true' && needs.setup.result != 'skipped' && (needs.sweep-evals.result != 'skipped' || needs.sweep-agentic-evals.result != 'skipped' || needs.sweep-multi-node-evals.result != 'skipped' || needs.sweep-multi-node-agentic-evals.result != 'skipped') }} uses: $/.github/workflows/collect-evals.yml with: tooling-ref: ${{ needs.setup.outputs.tooling-ref }} @@ -847,7 +902,7 @@ jobs: upload-changelog-metadata: name: upload-changelog-metadata needs: [setup, collect-results] - if: ${{ always() && needs.setup.result == 'success' }} + if: ${{ always() && needs.setup.result == 'success' && needs.setup.outputs.native-qualification != 'true' }} runs-on: ubuntu-latest steps: - name: Extract and save changelog metadata @@ -984,6 +1039,7 @@ jobs: github.event_name == 'push' && github.ref == 'refs/heads/main' && needs.setup.result == 'success' && + needs.setup.outputs.native-qualification != 'true' && ( ( toJson(fromJson(needs.setup.outputs.search-space-config).single_node['agentic']) == 'null' || @@ -1056,6 +1112,7 @@ jobs: github.event_name == 'push' && github.ref == 'refs/heads/main' && needs.setup.result == 'success' && + needs.setup.outputs.native-qualification != 'true' && needs.upload-changelog-metadata.result == 'success' && ( ( diff --git a/benchmarks/multi_node/srt-slurm-recipes/configs/prepared-runtime-lock.json b/benchmarks/multi_node/srt-slurm-recipes/configs/prepared-runtime-lock.json index 9536c7efd0..afcf7a6d98 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/configs/prepared-runtime-lock.json +++ b/benchmarks/multi_node/srt-slurm-recipes/configs/prepared-runtime-lock.json @@ -1,13 +1,14 @@ { "schema_version": 1, "repository": "https://github.com/SemiAnalysisAI/srt-slurm.git", - "revision": "8e459d10d217cec7636a4d257e1e0f69d2a45a81", + "revision": "50c3dacc37def01606ee9e4e0ed873646d4f7cc5", "uv_lock_sha256": "f7c3ef25605ebe27bac9c6a54aa2acef7332210321fff918c3fa72b7049d4dea", "capabilities": [ "prepared-v1", "durable-intent-v1", "custom-argv-v1", "controller-observation-v1", - "bounded-cleanup-v1" + "bounded-cleanup-v1", + "prepared-direct-listener-ownership-v1" ] } diff --git a/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/vllm/h100-fp4/agentx/agg-tp8-dspark5.yaml b/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/vllm/h100-fp4/agentx/agg-tp8-dspark5.yaml index bf431b2c71..95a221c167 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/vllm/h100-fp4/agentx/agg-tp8-dspark5.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/dsv41flash/vllm/h100-fp4/agentx/agg-tp8-dspark5.yaml @@ -50,5 +50,8 @@ roles: health_check: max_attempts: 360 interval_seconds: 10 +observability: + tachometer: + enabled: false benchmark: type: custom diff --git a/docs/srt-slurm-phase1.md b/docs/srt-slurm-phase1.md index 59409ce157..12982258d7 100644 --- a/docs/srt-slurm-phase1.md +++ b/docs/srt-slurm-phase1.md @@ -2,7 +2,7 @@ **English** | [中文](./srt-slurm-phase1_zh.md) -Phase 1 implements the first native srt-slurm lane. Hardware qualification, reader deployment and publication remain open. Passing local tests does not close this phase. The approved migration plan’s Phase 1 acceptance contract is restated below. The full plan and its research archive remain in the separate planning worktree. +Phase 1 implements the first native srt-slurm lane. Hardware qualification, reader readiness, trusted collector deployment and publication remain open. The app reader was deployed and then rolled back at the user’s request; the additive database schema remains. See the ledger below. Passing local tests does not close this phase. The approved migration plan’s Phase 1 acceptance contract is restated below. The full plan and its research archive remain in the separate planning worktree. ## Scope and ownership @@ -21,7 +21,8 @@ flowchart LR S --> V[Direct vLLM TP8] V --> C[Python AgentX or real eval] C --> A[Raw and normalized artifacts] - A --> R[Trusted source receipt] + A --> D[PR qualification: nine-point diagnostic summary] + A --> R[Publication path: trusted source receipt] R --> I[App validated import] R --> U[Later publication record] U --> I @@ -31,8 +32,20 @@ The recipe follows the existing YAML hierarchy at [`agg-tp8-dspark5.yaml`](../be ## Call and file map +The recipe explicitly disables native tachometer telemetry. Native defaults otherwise launch DCGM, node/process exporters and a host scraper even when general observability is disabled. These unprovisioned services are outside the temporary Phase 1 power exception; AgentX still collects its required vLLM server metrics. + ```mermaid flowchart TD + E[e2e-tests.yml / site operation] --> SP[infx.srt_slurm.provision.main] + SP --> SI[inspect_assets: observed shared files] + SP --> SR[infx.srt_slurm.provision_runtime.provision] + E --> CQ[infx.srt_slurm.qualify_cancellation.qualify] + CQ --> NI + SR --> SD[Private runtimes, offline cache and site draft] + SD --> PQ[PreparedSite: same-repository PR qualification] + PQ --> F + SD --> DP[PilotSite: verified reader and collector deployment pins] + DP --> F W[benchmark-tmpl.yml / native step] --> F[infx.srt_slurm.workflow.main] F --> J[infx.srt_slurm.job.parse_job] F --> P[infx.srt_slurm.launch.prepare] @@ -50,6 +63,7 @@ flowchart TD AX --> AIP[Pinned Python 3.11 AIPerf child] EV --> LM[Pinned lm-eval child] X --> O[Closed output staging and failure diagnostics] + O --> QV[infx.srt_slurm.qualification: complete nine-point validation] ``` `ExecutionReference` binds recipe, profile, runtime lock, client policy and the policy's golden YAML bytes. Changed inputs, duplicate YAML keys, unsupported scope or missing explicit queue demand fail before allocation. `priority` and `queue-token` are scheduling metadata; they do not change the requested semantic point. @@ -58,7 +72,9 @@ Preparation records the actual installed native/wrapper/client files, interprete ## Provisioning before the first GPU run -The existing E2E dispatch accepts `phase1-site-operation: inspect`. It checks the explicit paths/revisions in `runners/srt-slurm/h100-phase1-provision.json` on an H100 login runner and preserves `inventory.json` as a run/attempt artifact. This operation enters the existing priority queue with `nodes:1`, submits no Slurm allocation, and does not modify the shared model or trace caches. The configuration comes from the retained H100 baseline; the report establishes which paths actually exist before runtime installation. Missing images, snapshots or weight shards fail inspection. An inventory is not hardware qualification or a deployed-reader declaration. +The existing E2E dispatch accepts `phase1-site-operation: inspect`. It checks the explicit paths/revisions in `runners/srt-slurm/h100-phase1-provision.json` on an H100 login runner and preserves `inventory.json` as a run/attempt artifact. This operation enters the existing priority queue with `nodes:1`, submits no Slurm allocation, and does not modify the shared model or trace caches. The configuration comes from the retained H100 baseline; the report establishes which paths actually exist before runtime installation. Missing images, snapshots or weight shards fail inspection. The current preflight also requires `sbatch`, `squeue`, `sacct`, `scancel`, `srun` and `scontrol`. An inventory is not hardware qualification or a deployed-reader declaration. + +`phase1-site-operation: provision` uses the same entry point to create a separate generation under the configured shared root. Supply the sweep’s exact PR merge SHA through the existing `ref` input so the installed wrapper matches the measured tree, including its current base revision. It installs the native runtime and exact-checkout wrapper noneditable, retains build/dependency identities, and prepares pinned AgentX/eval environments with private offline cache references. It preserves the existing model, image and trace payloads. Its output is a **site draft** using the strict `PreparedSite` schema. After provisioning succeeds, configure `INFX_H100_PHASE1_PREPARED_SITE_JSON` with that exact JSON to run PR qualification. Publication separately requires a `PilotSite` containing verified deployed reader/collector revisions. Candidate PR commits cannot serve as deployment pins. Provisioning also tests the installed eval backend and pinned AgentX tokenizer against the actual offline model snapshot before hashing the large assets. The renderer also binds the actual engine TP/PP/context/data-parallel arguments to the requested topology before native preparation. Conflicting underscore/hyphen aliases, noninteger sizes and enabled expert parallelism fail closed. @@ -69,13 +85,15 @@ Provision on shared Linux storage visible to the H100 login host and compute con 3. Materialize separate client environments and retain their resolved package artifacts/locks. AgentX must come from `754356e9a39acc6cc6afb242d123bb57c3fb6f75`; lm-eval must come from `b315ef3b05176acc9732bb7fdec116abe1ecc476`. Editable and wrong-source installations are rejected. Preparation captures every installed distribution, not just the named entry point. 4. Materialize the complete model/tokenizer snapshot, exact `semianalysisai/cc-traces-weka-062126` snapshot and GSM8K cache. Capture their real revisions; do not invent or substitute a revision. The client’s offline model `refs/main` and snapshot files must be bound assets, and its resolved model snapshot must be the exact canonical serving snapshot. This preserves nominal tokenizer names without permitting a different cached revision. Prepare the unchanged serving image as a verified squash file and record its provenance/hash. 5. Write one `ClientSite` JSON for AgentX and one for eval. These explicitly provide the interpreter, distributions, offline cache environment, environment removals, asset roots/files, model snapshot, timeout and termination grace. `RuntimeSpec` rejects credentials; execution strips ambient credentials and unqualified AIPerf overrides. The packaged task and 1,319 independent document hashes are included in installed wheels. -6. Write the `PilotSite` JSON with these two client-site paths, source/interpreter/model/image paths, mounts and actual deployed reader/collector revisions. The Pydantic models in [`render.py`](../infx/srt_slurm/render.py) and [`prepare.py`](../infx/benchmarks/prepare.py) are the exact schemas. +6. Preserve the generated `PreparedSite` JSON with these two client-site paths, source/interpreter/model/image paths and mounts. For publication, extend it to `PilotSite` with actual deployed reader/collector revisions. The Pydantic models in [`render.py`](../infx/srt_slurm/render.py) and [`prepare.py`](../infx/benchmarks/prepare.py) are the exact schemas. Preparation validates existing assets; it does not install packages, download models or repair incomplete snapshots on compute nodes. The derived mmap cache uses an owned namespace, file-integrity receipts, independent verified copies and corruption quarantine. Cold preparation on lock contention is explicit and bounded. -Before enabling sweeps, deploy the app reader and migration `016_measurement_snapshots.sql`, then land/deploy the trusted collector. Configure `INFX_H100_PHASE1_SITE_JSON`, `INFX_PHASE1_READER_REVISION` and `INFX_PHASE1_COLLECTOR_REVISION` in InferenceX. Configure `INFX_RECEIPT_ISSUER_SHAS` and `INFX_RECEIPT_ISSUER_WORKFLOW` in both repositories; the workflow is `.github/workflows/phase1-receipt.yml`. These values are absent in the inspected repository configuration. A source branch containing the code alone is not a deployed reader. +Before enabling publication, require a verified active app reader deployment, migration `016_measurement_snapshots.sql` and a deployed trusted collector. Configure `INFX_H100_PHASE1_SITE_JSON`, `INFX_PHASE1_READER_REVISION` and `INFX_PHASE1_COLLECTOR_REVISION` in InferenceX. Configure `INFX_RECEIPT_ISSUER_SHAS` and `INFX_RECEIPT_ISSUER_WORKFLOW` in both repositories; the workflow is `.github/workflows/phase1-receipt.yml`. `INFX_PHASE1_READER_REVISION` was removed after the app rollback; the site, collector and issuer settings remain pending. The reader is currently unavailable for native receipt ingestion, and publication remains gated. Retained schema reports and code on a source branch do not establish current reader readiness. -The GitHub native launch uses uv-managed Python 3.12 and does not require an ambient `python` command. The workflow checks the three site/deployment variables before preparation or Slurm allocation. Its error names missing variables, identifies invalid fields in the site JSON without echoing their values, and names reader/collector revision variables that disagree with the site configuration. This check requires explicit configuration and does not provision assets or deploy services. +The PR sweep has an explicit qualification route for an isolated native matrix from a same-repository `pull_request` event. It uses the prepared-site variable and records `purpose: pr-qualification` in the source and executable bundle. It runs the unchanged eight throughput points and full real c28 eval. It emits `native-qualification-run`, nine `native-qualification-` artifacts and a `native-qualification-summary`, with no normal benchmark/eval artifact names or `RESULT_FILENAME`. The summary revalidates bundle and member digests, execution and Slurm identities, native resources, AgentX raw/normalized results, and the complete scored GSM8K corpus. `complete: true` means this diagnostic sweep passed; `publication_eligible` remains false. Reuse, staging and receipt/publication validators reject qualification markers, including inventories mixed with normal artifacts. These results cannot later be promoted by approving the source run. + +The GitHub native launch uses uv-managed Python 3.12 and an isolated checkout for each run, attempt and queue token; it does not require an ambient `python` command or repair shared Git state. The default publication route still checks the three site/deployment variables before preparation or Slurm allocation. Its error names missing variables, identifies invalid fields in the site JSON without echoing their values, and names reader/collector revision variables that disagree with the site configuration. Neither route installs assets or deploys services during a benchmark. ## Preparation, execution and recovery @@ -90,12 +108,16 @@ python -m infx.srt_slurm.launch --job job.json --site site.json --root CHECKOUT Native preparation must resolve exactly `{nodes:1,gpus_per_node:8,serving_gpus:8,workers:1,cardinality:1}`. Throughput renders synthetic rejection from the committed golden curve with adaptive verification off. Eval renders real block rejection with adaptive verification on. Direct port 8000 is an explicit exclusive-node policy: a bind collision is a failure, not permission to contact another server. -Slurm allocation, claims, accepted IDs, scheduler observation and cancellation belong to the native runtime. The adapter obtains the journal path before the interruptible submit. It never repeats an ambiguous submission or cancels by runner name. Active controller state takes precedence over stale accounting; a failed-but-active requeue is not closed. Known owned allocations are cancelled and observed to terminal closure with bounded waits. An unresolved intent stays fenced for inspection. +Slurm allocation, claims, accepted IDs, scheduler observation and cancellation belong to the native runtime. The adapter obtains the journal path before the interruptible submit. It never repeats an ambiguous submission or cancels by runner name. Active controller state takes precedence over stale accounting; a failed-but-active requeue is not closed. Known owned allocations are cancelled and observed to terminal closure with bounded waits. Repeated catchable signals are ignored during that bounded cleanup and the original handlers are restored afterward. A forced process kill or host loss can still interrupt cleanup; the durable journal remains available for native reconciliation. An unresolved intent stays fenced for inspection. Successful publication requires both native terminal success and a closed client audit with no error, timeout, signal or orphaned writer. Failure diagnostics retain raw outputs, client audit, frozen inputs and native logs without producing an accepted execution manifest. The broad legacy pre/post runner cleanup is skipped for this lane. The legacy H100 launcher/script remains available for rollback until qualification and a reviewed retirement diff. ## Measurement receipt and publication +Before accepting a sweep, qualify real cancellation through the E2E `cancel-startup` and `cancel-client` site operations. Supply the actual provisioned `phase1-site-draft` path. They use separate diagnostic output/journals, the pinned image and TP8 worker, and native ownership-aware cancellation; they do not fabricate deployment pins or produce accepted benchmark manifests. Startup uses a 300-second allocation with a 240-second observation budget. Client interruption uses a 3,600-second allocation with a 3,300-second observation budget so model readiness can complete. Each has a separate 180-second cleanup budget. Preserve the resulting `qualification.json` and native terminal/writer evidence; a cancellation RPC or `COMPLETING` alone is insufficient. + +The native pin now requires `prepared-direct-listener-ownership-v1`. Before client traffic, it verifies the listening socket belongs to the recorded worker process tree, using PID start times and PID/network namespace identities. It repeats ownership checks during the client and before accepting exit 0. A foreign/replaced listener, reused PID or inaccessible ownership evidence fails the job and closes the client. Actual Pyxis namespace/proc visibility is part of cluster qualification. + The complete source contract is eight throughput points and one real c28 eval. GSM8K requires all 1,319 documents and both filters (2,638 scored rows), the preserved 16,384 context / 12,288 generation budgets, finite scores and complete sample identities. Aggregate eval metadata has `disagg:false`, `is_multinode:false`, eight serving GPUs and zero prefill/decode worker counts. 1. Create a reviewed `qualification/phase1/*.json` expectation using the `Approval` schema in [`phase1_publication.py`](../infx/workflows/phase1_publication.py). Copy point/execution/bundle/native-manifest identities from the independently prepared control records, not worker archives. Require the complete nine-point set and actual corpus revision. @@ -108,10 +130,16 @@ The merge helper preserves the latest explicit authorized `/use RUN_ID` (or `/re ## Qualification ledger +Historical evidence: app [PR1179](https://github.com/SemiAnalysisAI/InferenceX-app/pull/1179) was normally squash-merged to `481a8622cc9bc27feae775850e241ec967bac1e3` after 6,892 unit, 486 component and 1,033 integration tests per browser passed, with a clean Bugbot review. Migration 016 and its schema verification passed on [staging](https://github.com/SemiAnalysisAI/InferenceX-app/actions/runs/35478033492) and [production](https://github.com/SemiAnalysisAI/InferenceX-app/actions/runs/35478066182). Vercel Production deployment `6547210249` succeeded for that exact SHA. These reports remain valid evidence of the schema operations performed at that revision. + +Current readiness: at the user’s request, Vercel Instant Rollback restored production to `9bb7b13eb4985217a6282f340459fd5948613276` ([deployment](https://inferencemax-7ecuqzeqm-semianalysisai.vercel.app) `dpl_8H2dnpuKFDZhwe6pU7tb657pu5z3`), and `INFX_PHASE1_READER_REVISION` was removed. The exact code revert, [PR1180](https://github.com/SemiAnalysisAI/InferenceX-app/pull/1180), was merged with explicit user approval at `2026-09-20T00:29:48Z`; `master` is now `92fef485edd5ae61fe49d01f0e41b67492263bee`, whose tree exactly matches the pre-PR1179 revision `9bb7b13eb4985217a6282f340459fd5948613276`. The user subsequently reported that automatic production promotion was re-enabled. This does not restore the reverted reader code or its removed readiness variable. All other PR merges remain prohibited by the current user instruction. The additive `measurement_snapshots` table and migration ledger are retained. The Phase 1 reader is unavailable, so native receipt ingestion and publication remain gated. No Phase 1 native receipt or measurement has been imported. Collector [PR3298](https://github.com/SemiAnalysisAI/InferenceX/pull/3298) also remains open and subject to the repository’s Core/CODEOWNER approval requirements. + +[H100 inventory run 35477700047](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35477700047) passed on an actual login runner at `14d56f1bbf8f3c867ea79ae97a2f716f304aaaa2`. Both pinned snapshots and all indexed model shards exist at the configured canonical shared paths; the serving squash file is 21,390,860,288 bytes, and the four required Slurm commands are present. [CI 35477693002](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35477693002) passed the updated topology and inventory behavior plus the native Linux contract. Neither run submitted a GPU benchmark. + | Gate | Status / required evidence | | --- | --- | | Native and client behavior | CPU tests and installed-wheel checks; no GPU claim | -| Receipt, app and recovery | Local unit, database and browser smoke checks; deployment pending | +| Receipt, app and recovery | Reader rolled back and revision variable removed; additive schema retained; native receipt import and publication gated | | H100 throughput | c1,2,4,8,16,20,24,28 pending on the unchanged image | | Real evaluation | New c28 run pending; historical full raw eval passes validator | | Cancellation/cleanup | Local ownership/race/closure tests; real Slurm signal qualification pending | diff --git a/docs/srt-slurm-phase1_zh.md b/docs/srt-slurm-phase1_zh.md index 16af665b4b..6847d5856b 100644 --- a/docs/srt-slurm-phase1_zh.md +++ b/docs/srt-slurm-phase1_zh.md @@ -2,7 +2,7 @@ [English](./srt-slurm-phase1.md) | **中文** -阶段 1 实现首条原生 srt-slurm 路径。硬件验收、reader 部署与发布仍待完成;本地测试通过不代表阶段结束。下文重述已批准迁移计划中阶段 1 的验收要求。完整计划及研究资料仍保留在独立的规划 worktree。 +阶段 1 实现首条原生 srt-slurm 路径。硬件验收、reader 就绪、可信 collector 部署与发布仍待完成。app reader 曾完成部署,随后按用户要求回滚;新增数据库 schema 仍保留,详见下方账本。本地测试通过不代表阶段结束。下文重述已批准迁移计划中阶段 1 的验收要求。完整计划及研究资料仍保留在独立的规划 worktree。 ## 范围与职责 @@ -21,7 +21,8 @@ flowchart LR S --> V[直连 vLLM TP8] V --> C[Python AgentX 或真实 eval] C --> A[原始及规范化产物] - A --> R[受信任源测量回执] + A --> D[PR 验收:完整九点诊断汇总] + A --> R[发布路径:受信任源测量回执] R --> I[App 校验后导入] R --> U[后续发布记录] U --> I @@ -31,8 +32,20 @@ flowchart LR ## 调用与文件关系 +配方显式关闭原生 tachometer 遥测。即使通用 observability 关闭,原生默认设置仍会启动 DCGM、node/process exporter 和主机 scraper。这些未部署的服务不属于阶段 1 的临时功耗例外;AgentX 仍会采集必需的 vLLM 服务指标。 + ```mermaid flowchart TD + E[e2e-tests.yml / 站点操作] --> SP[infx.srt_slurm.provision.main] + SP --> SI[inspect_assets:实际共享文件] + SP --> SR[infx.srt_slurm.provision_runtime.provision] + E --> CQ[infx.srt_slurm.qualify_cancellation.qualify] + CQ --> NI + SR --> SD[专用运行环境、离线缓存与站点草稿] + SD --> PQ[PreparedSite:同仓库 PR 验收] + PQ --> F + SD --> DP[PilotSite:已验证的 reader 与 collector 部署 revision] + DP --> F W[benchmark-tmpl.yml / native step] --> F[infx.srt_slurm.workflow.main] F --> J[infx.srt_slurm.job.parse_job] F --> P[infx.srt_slurm.launch.prepare] @@ -50,6 +63,7 @@ flowchart TD AX --> AIP[固定 Python 3.11 AIPerf 子进程] EV --> LM[固定 lm-eval 子进程] X --> O[写入者关闭后的产物整理与失败诊断] + O --> QV[infx.srt_slurm.qualification:完整九点校验] ``` `ExecutionReference` 绑定配方、profile、runtime lock、client policy 以及 policy 指向的 golden YAML 字节。输入发生变化、YAML 重复键、超出范围或缺少明确排队节点需求,均在分配前失败。`priority` 与 `queue-token` 仅属于调度信息,不改变请求测量点。 @@ -58,7 +72,9 @@ flowchart TD ## 首次 GPU 运行前的部署 -现有 E2E 手动调度支持 `phase1-site-operation: inspect`,在 H100 登录 runner 上检查 `runners/srt-slurm/h100-phase1-provision.json` 显式提供的路径和 revision,并将 `inventory.json` 保存为绑定运行及 attempt 的 artifact。此操作以 `nodes:1` 进入现有优先级队列,不提交 Slurm allocation,也不修改共享模型或 trace 缓存。配置来自保留的 H100 基线;检查报告用于在安装运行环境之前确认实际存在的路径。镜像、快照或权重分片缺失会使检查失败。资源清单不代表硬件验收完成,也不代表 reader 已部署。 +现有 E2E 手动调度支持 `phase1-site-operation: inspect`,在 H100 登录 runner 上检查 `runners/srt-slurm/h100-phase1-provision.json` 显式提供的路径和 revision,并将 `inventory.json` 保存为绑定运行及 attempt 的 artifact。此操作以 `nodes:1` 进入现有优先级队列,不提交 Slurm allocation,也不修改共享模型或 trace 缓存。配置来自保留的 H100 基线;检查报告用于在安装运行环境之前确认实际存在的路径。镜像、快照或权重分片缺失会使检查失败。当前预检还要求 `sbatch`、`squeue`、`sacct`、`scancel`、`srun` 和 `scontrol` 全部可用。资源清单不代表硬件验收完成,也不代表 reader 已部署。 + +`phase1-site-operation: provision` 通过同一入口,在配置的共享根目录下建立独立 generation。通过已有 `ref` 输入提供 sweep 实际使用的完整 PR merge SHA,使已安装 wrapper 与包含当前 base revision 的测量代码树一致。它以非 editable 方式安装原生运行时和当前 checkout 的 wrapper,保留构建及依赖身份,并为固定版本的 AgentX/eval 建立运行环境与专用离线缓存引用。现有模型、镜像和 trace 内容保持不变。输出是使用严格 `PreparedSite` schema 的**站点草稿**。准备成功后,将完整的实际 JSON 配置为 `INFX_H100_PHASE1_PREPARED_SITE_JSON`,即可执行 PR 验收。发布另需 `PilotSite`,其中的 reader/collector revision 必须来自已验证的实际部署,不能使用候选 PR 提交替代。准备过程还会在大型资源哈希计算前,使用实际离线模型快照检查已安装的 eval backend 及固定 AgentX tokenizer。 渲染器还会在原生准备步骤之前,将引擎实际 TP/PP/上下文/数据并行参数与请求的拓扑绑定。下划线与连字符别名冲突、非整数并行度或启用专家并行都会被拒绝。 @@ -69,13 +85,15 @@ flowchart TD 3. 准备独立客户端环境并保留实际解析的包产物与锁。AgentX 必须来自 `754356e9a39acc6cc6afb242d123bb57c3fb6f75`;lm-eval 必须来自 `b315ef3b05176acc9732bb7fdec116abe1ecc476`。拒绝 editable 或错误来源。准备阶段记录所有已安装 distribution,而非仅入口包。 4. 完整准备模型/tokenizer snapshot、准确的 `semianalysisai/cc-traces-weka-062126` snapshot 与 GSM8K 缓存。记录真实 revision,不编造或替换。准备原样服务镜像的已校验 squash 文件,记录来源和哈希。客户端离线模型的 `refs/main` 与 snapshot 文件必须纳入资源绑定,解析后的模型 snapshot 必须与服务端的规范路径完全一致,保留 tokenizer 的模型名称同时禁止解析到其他缓存版本。 5. 分别编写 AgentX、eval 的 `ClientSite` JSON:解释器、distribution、离线缓存环境、移除变量、资源根目录/文件、模型 snapshot、超时与终止宽限。`RuntimeSpec` 拒绝凭证;执行时移除继承凭证及未验收的 AIPerf 覆盖项。wheel 包含 eval task 与 1,319 个独立文档哈希。 -6. 编写 `PilotSite` JSON,包含两个客户端配置路径、源码/解释器/模型/镜像路径、挂载以及实际部署的 reader/collector revision。准确 schema 见 [`render.py`](../infx/srt_slurm/render.py) 与 [`prepare.py`](../infx/benchmarks/prepare.py)。 +6. 保留生成的 `PreparedSite` JSON,包含两个客户端配置路径、源码/解释器/模型/镜像路径及挂载。发布时再添加实际部署的 reader/collector revision,形成 `PilotSite`。准确 schema 见 [`render.py`](../infx/srt_slurm/render.py) 与 [`prepare.py`](../infx/benchmarks/prepare.py)。 准备阶段校验已有资源,不在计算节点安装包、下载模型或修复不完整 snapshot。派生 mmap 缓存使用独立所属 namespace、文件完整性回执、独立校验副本及损坏隔离;锁竞争时的冷准备有明确界限。 -先部署 app reader 与 `016_measurement_snapshots.sql` migration,再合入/部署受信任 collector。InferenceX 需配置 `INFX_H100_PHASE1_SITE_JSON`、`INFX_PHASE1_READER_REVISION`、`INFX_PHASE1_COLLECTOR_REVISION`。两个仓库均需配置 `INFX_RECEIPT_ISSUER_SHAS`、`INFX_RECEIPT_ISSUER_WORKFLOW`,workflow 路径为 `.github/workflows/phase1-receipt.yml`。检查时这些变量尚不存在。分支中有代码不等于 reader 已部署。 +启用发布前,必须确认 app reader 当前部署已验证、`016_measurement_snapshots.sql` migration 已完成且受信任 collector 已部署。InferenceX 需配置 `INFX_H100_PHASE1_SITE_JSON`、`INFX_PHASE1_READER_REVISION`、`INFX_PHASE1_COLLECTOR_REVISION`。两个仓库均需配置 `INFX_RECEIPT_ISSUER_SHAS`、`INFX_RECEIPT_ISSUER_WORKFLOW`,workflow 路径为 `.github/workflows/phase1-receipt.yml`。app 回滚后已删除 `INFX_PHASE1_READER_REVISION`;站点、collector 与 issuer 设置仍待完成。当前 reader 无法用于原生回执导入,发布仍受门禁限制。保留的 schema 报告和源分支中的代码都不能证明当前 reader 已就绪。 -GitHub 原生启动步骤使用 uv 管理的 Python 3.12,不依赖环境中已有的 `python` 命令。workflow 在准备或申请 Slurm 资源前校验这三个站点/部署变量。错误会列出缺失变量,指出站点 JSON 的无效字段而不回显字段值,并列出与站点配置不一致的 reader/collector revision 变量。该校验要求显式配置,不会准备资源或部署服务。 +PR sweep 为同仓库 `pull_request` 事件的独立原生矩阵提供明确的验收路径。该路径读取 prepared-site 变量,在 source 与可执行 bundle 中记录 `purpose: pr-qualification`,保持原有八个吞吐点及完整真实 c28 eval 不变。它只产生 `native-qualification-run`、九个 `native-qualification-` artifact 和 `native-qualification-summary`,不产生普通 benchmark/eval artifact 名称或 `RESULT_FILENAME`。汇总重新校验 bundle 与成员摘要、执行及 Slurm 身份、原生资源、AgentX 原始及规范化结果,以及完整评分的 GSM8K 语料。`complete: true` 表示本次诊断 sweep 通过,`publication_eligible` 仍为 false。Reuse、staging、回执和 publication validator 会拒绝含验收标记的来源,包括混合了普通 artifact 的清单。之后批准该源 run 也不能将这些结果提升为可发布测量。 + +GitHub 原生启动步骤使用 uv 管理的 Python 3.12,并为每个 run、attempt 和 queue token 建立独立 checkout;不依赖环境中已有的 `python` 命令,也不修复共享 Git 状态。默认发布路径仍在准备或申请 Slurm 资源前校验三个站点/部署变量。错误会列出缺失变量,指出站点 JSON 的无效字段而不回显字段值,并列出与站点配置不一致的 reader/collector revision 变量。两条路径都不会在 benchmark 中安装资源或部署服务。 ## 准备、执行与恢复 @@ -90,12 +108,16 @@ python -m infx.srt_slurm.launch --job job.json --site site.json --root CHECKOUT 原生准备必须解析为 `{nodes:1,gpus_per_node:8,serving_gpus:8,workers:1,cardinality:1}`。吞吐从已提交 golden 曲线渲染 synthetic rejection,关闭 adaptive verification;eval 使用真实 block rejection,开启 adaptive verification。直连端口 8000 采用明确的独占节点策略:端口冲突即失败,不能连接其他服务器。 -Slurm 分配、claim、已接受 ID、调度器观察与取消均由原生运行时负责。适配器在可中断 submit 前取得日志路径,不重试不明确的提交,也不按 runner 名批量取消。活跃 controller 状态优先于陈旧 accounting;失败但仍活跃的 requeue 不算关闭。取消所有已确认属于该 intent 的资源,并在有界等待内观察终止;未解决的 intent 保持 fenced,等待检查。 +Slurm 分配、claim、已接受 ID、调度器观察与取消均由原生运行时负责。适配器在可中断 submit 前取得日志路径,不重试不明确的提交,也不按 runner 名批量取消。活跃 controller 状态优先于陈旧 accounting;失败但仍活跃的 requeue 不算关闭。取消所有已确认属于该 intent 的资源,并在有界等待内观察终止。在这段有界清理期间忽略重复的可捕获信号,结束后恢复原处理器。强制终止进程或主机故障仍可能中断清理;持久化 journal 保留,供原生 reconcile 恢复。未解决的 intent 保持 fenced,等待检查。 成功发布要求原生终止成功及客户端写入者关闭:无错误、超时、信号或孤儿 writer。失败诊断保留原始输出、client audit、冻结输入及原生日志,但不生成已接受 execution manifest。该路径跳过旧 runner 的宽泛前后清理。H100 旧 launcher/script 保留至硬件验收及退休差异审阅完成,以便回退。 ## 测量回执与发布 +接受 sweep 前,先通过 E2E 的 `cancel-startup` 与 `cancel-client` 站点操作验证真实取消行为,并显式提供已准备的 `phase1-site-draft` 路径。它们使用独立诊断输出和 journal、固定镜像与 TP8 worker,以及原生归属校验和取消接口,不编造部署 pin,也不生成可接受的 benchmark manifest。启动期探针申请 300 秒 allocation,观察预算为 240 秒;客户端中断探针申请 3,600 秒 allocation,观察预算为 3,300 秒,以便模型完成就绪。两种模式的清理预算均为额外 180 秒。保留 `qualification.json` 及原生终态/writer 证据;仅有取消 RPC 或 `COMPLETING` 不足以通过验收。 + +当前原生 pin 要求 `prepared-direct-listener-ownership-v1`。发送客户端流量前,它根据 PID 启动时间及 PID/network namespace 身份,验证监听套接字属于记录的 worker 进程树;客户端运行期间和接受退出码 0 之前也会重新验证。外部或替换监听器、PID 复用或无法读取的归属证据都会使任务失败并关闭客户端。实际 Pyxis 的 namespace/proc 可见性仍需集群验收。 + 完整源契约为八个吞吐点加一个真实 c28 eval。GSM8K 必须包含全部 1,319 文档及两种 filter(2,638 个评分行),保留 16,384 上下文 / 12,288 生成预算,验证有限分数和完整样本身份。聚合 eval 元数据为 `disagg:false`、`is_multinode:false`、八个服务 GPU、prefill/decode worker 数均为零。 1. 按 [`phase1_publication.py`](../infx/workflows/phase1_publication.py) 的 `Approval` schema 创建经审阅的 `qualification/phase1/*.json`。点、执行、bundle、原生 manifest 身份必须来自独立准备期控制记录,不得从 worker archive 推导。要求完整九点集合及实际数据集 revision。 @@ -108,10 +130,16 @@ Merge helper 保留最近明确授权的 `/use RUN_ID` 或 `/reuse-sweep-run RUN ## 验收账本 +历史证据:app [PR1179](https://github.com/SemiAnalysisAI/InferenceX-app/pull/1179) 在 6,892 项单元测试、486 项组件测试、每个浏览器 1,033 项集成测试和 Bugbot 审查通过后,通过正常 squash 流程合入 `481a8622cc9bc27feae775850e241ec967bac1e3`。[Staging](https://github.com/SemiAnalysisAI/InferenceX-app/actions/runs/35478033492) 与 [production](https://github.com/SemiAnalysisAI/InferenceX-app/actions/runs/35478066182) 均完成 migration 016 及 schema 验证。该精确 SHA 的 Vercel Production 部署 `6547210249` 曾成功。这些报告仍可证明当时在该 revision 执行的 schema 操作。 + +当前就绪状态:按用户要求,Vercel Instant Rollback 已将 production 恢复至 `9bb7b13eb4985217a6282f340459fd5948613276`([部署](https://inferencemax-7ecuqzeqm-semianalysisai.vercel.app) `dpl_8H2dnpuKFDZhwe6pU7tb657pu5z3`),并已删除 `INFX_PHASE1_READER_REVISION`。精确代码回退 [PR1180](https://github.com/SemiAnalysisAI/InferenceX-app/pull/1180) 已获用户明确批准,并于 `2026-09-20T00:29:48Z` 合并;`master` 现为 `92fef485edd5ae61fe49d01f0e41b67492263bee`,其代码树与 PR1179 之前的 revision `9bb7b13eb4985217a6282f340459fd5948613276` 完全一致。用户随后报告已重新启用 production 自动提升;这不会恢复已回退的 reader 代码或已删除的就绪变量。当前用户指令仍禁止合并所有其他 PR。新增的 `measurement_snapshots` 表和 migration ledger 均保留。阶段 1 reader 当前不可用,因此原生回执导入和发布仍受门禁限制。尚未导入任何阶段 1 原生回执或测量。Collector [PR3298](https://github.com/SemiAnalysisAI/InferenceX/pull/3298) 也仍未合并,并须满足仓库的 Core/CODEOWNER 审批要求。 + +[H100 资源检查运行 35477700047](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35477700047) 在实际登录 runner 上通过,提交为 `14d56f1bbf8f3c867ea79ae97a2f716f304aaaa2`。固定的两个快照及全部索引内模型分片均位于配置的规范共享路径,服务 squash 文件大小为 21,390,860,288 字节,四个必需的 Slurm 命令均可用。[CI 35477693002](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35477693002) 通过了更新后的拓扑、资源检查行为和原生 Linux 契约验证。这两次运行均未提交 GPU benchmark。 + | Gate | 状态 / 所需证据 | | --- | --- | | 原生及客户端行为 | CPU 测试、已安装 wheel 检查;不声称 GPU 验收 | -| 回执、app、恢复 | 本地单元/数据库/浏览器 smoke 检查;待部署 | +| 回执、app、恢复 | reader 已回滚,revision 变量已删除;新增 schema 保留;原生回执导入与发布仍受门禁限制 | | H100 吞吐 | 原镜像 c1、2、4、8、16、20、24、28 待运行 | | 真实 eval | 新 c28 待运行;历史完整原始 eval 通过新 validator | | 取消与清理 | 本地所属关系/race/closure 测试;实际 Slurm 信号验收待做 | diff --git a/infx/benchmarks/agentx.py b/infx/benchmarks/agentx.py index 124993a702..d31bf98768 100644 --- a/infx/benchmarks/agentx.py +++ b/infx/benchmarks/agentx.py @@ -234,13 +234,13 @@ def normalize(spec: AgentXSpec, artifact_root: Path) -> Path: return output -def _has_metric(value: Any, prefix: str) -> bool: +def has_metric(value: Any, prefix: str) -> bool: if isinstance(value, dict): return any( - key.startswith(prefix) or _has_metric(item, prefix) for key, item in value.items() + key.startswith(prefix) or has_metric(item, prefix) for key, item in value.items() ) if isinstance(value, list): - return any(_has_metric(item, prefix) for item in value) + return any(has_metric(item, prefix) for item in value) return False @@ -261,7 +261,7 @@ def finalize(spec: AgentXSpec, endpoint: str, artifact_root: Path) -> list[str]: if ( not csv.is_file() or csv.stat().st_size == 0 - or not _has_metric(metrics, spec.required_server_metric_prefix) + or not has_metric(metrics, spec.required_server_metric_prefix) ): errors.append("required vLLM JSON/CSV server metrics are absent") except (OSError, ValueError, TypeError, KeyError) as exc: diff --git a/infx/results/publication_receipt.py b/infx/results/publication_receipt.py index 91551fb797..7ce5b626c8 100644 --- a/infx/results/publication_receipt.py +++ b/infx/results/publication_receipt.py @@ -379,6 +379,10 @@ def validate_point_content(point: Point, archives: Path) -> None: def seal_receipt( expected: ExpectedContract, issuer: Issuer, inventory: list[dict[str, Any]], archives: Path ) -> SourceReceipt: + from infx.srt_slurm.qualification import qualification_artifacts + + if qualification_artifacts(row["name"] for row in inventory): + raise ValueError("nonpublishing qualification artifacts cannot be sealed") required = {artifact_id for point in expected.points for artifact_id in point.artifact_ids} rows = {int(row["id"]): row for row in inventory} if len(rows) != len(inventory) or set(rows) != required: @@ -417,6 +421,8 @@ def seal_receipt( raise ValueError("Execution evidence must be an object") native = execution.get("native_receipt", {}) source = execution.get("source", {}) + if isinstance(source, dict) and source.get("purpose", "publication") != "publication": + raise ValueError("nonpublishing execution cannot be sealed") if ( not isinstance(native, dict) or not isinstance(source, dict) diff --git a/infx/srt_slurm/launch.py b/infx/srt_slurm/launch.py index 5744fcc778..ce196fe64d 100644 --- a/infx/srt_slurm/launch.py +++ b/infx/srt_slurm/launch.py @@ -26,7 +26,7 @@ from infx.benchmarks.spec import RuntimeSpec from infx.srt_slurm.contracts import digest, load_mapping from infx.srt_slurm.job import JobSpec, file_digest, intent_id, parse_job, read_json -from infx.srt_slurm.render import ClientPolicy, PilotSite, client_spec, render_recipe +from infx.srt_slurm.render import ClientPolicy, PilotSite, PreparedSite, client_spec, render_recipe class RuntimeLock(BaseModel): @@ -45,6 +45,10 @@ def __init__(self, output: dict[str, Any], detail: str) -> None: super().__init__(detail) +class WorkflowCancelledError(RuntimeError): + """Unlike InterruptedError, this is not swallowed by selectors during communicate.""" + + def checked_json(argv: list[str], *, timeout: int = 600) -> dict[str, Any]: with subprocess.Popen( argv, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, start_new_session=True @@ -76,13 +80,13 @@ def checked_json(argv: list[str], *, timeout: int = 600) -> dict[str, Any]: return value -def native(site: PilotSite, *args: str, timeout: int = 600) -> dict[str, Any]: +def native(site: PreparedSite, *args: str, timeout: int = 600) -> dict[str, Any]: return checked_json( [site.native_python, "-I", "-m", "srtctl.cli.submit", *args, "--json"], timeout=timeout ) -def verify_site(job: JobSpec, site: PilotSite, root: Path) -> dict[str, Any]: +def verify_site(job: JobSpec, site: PreparedSite, root: Path) -> dict[str, Any]: reference = job.row.execution if reference is None: raise ValueError("missing native execution reference") @@ -149,16 +153,20 @@ def verify_wrapper_source(identity: dict[str, Any], root: Path) -> None: def effective_identity( job: JobSpec, - site: PilotSite, + site: PreparedSite, identity: dict[str, Any], runtime: RuntimeSpec, resources: dict[str, Any], + *, + client_identity: dict[str, Any] | None = None, ) -> tuple[str, str]: semantics = job.semantic_inputs() inputs = { "requested": semantics, "installed": identity, - "client_identity": read_json(Path(runtime.identity.path)), + "client_identity": ( + read_json(Path(runtime.identity.path)) if client_identity is None else client_identity + ), "client_assets": {asset.path: asset.sha256 for asset in runtime.assets}, "client_env": runtime.env, "client_env_unset": sorted(runtime.env_unset), @@ -175,13 +183,16 @@ def effective_identity( def prepare( job: JobSpec, - site: PilotSite, + site: PreparedSite, root: Path, source: dict[str, Any], ) -> dict[str, Any]: """A local preparation lock protects files; only native srtctl may claim/submit Slurm.""" from infx.benchmarks.prepare import ClientSite, prepare as prepare_client + if source.get("purpose") != "pr-qualification" and not isinstance(site, PilotSite): + raise ValueError("publication preparation requires deployed reader/collector identities") + execution = intent_id( source["repository"], str(source["run_id"]), str(source["attempt"]), job.point_id ) @@ -319,7 +330,7 @@ def verify_bundle(bundle: dict[str, Any]) -> None: raise ValueError(f"prepared input changed: {name}") -def verify_execution_clients(bundle: dict[str, Any], site: PilotSite) -> None: +def verify_execution_clients(bundle: dict[str, Any], site: PreparedSite) -> None: verify_file(site.image) actual = capture_identity(site.wrapper_python, ["infx"], dataset_loader=None) if actual != bundle["identity"]["wrapper_identity"]: @@ -400,7 +411,10 @@ def publish_outputs(bundle: dict[str, Any], receipt: dict[str, Any], workspace: def execute(bundle: dict[str, Any], workspace: Path, *, reconcile_timeout: int = 120) -> None: verify_bundle(bundle) - site = PilotSite.model_validate(bundle["site"]) + site_type = ( + PreparedSite if bundle.get("source", {}).get("purpose") == "pr-qualification" else PilotSite + ) + site = site_type.model_validate(bundle["site"]) verify_execution_clients(bundle, site) journal = Path(site.shared_root) / "journal" receipt_path = Path( @@ -420,7 +434,7 @@ def execute(bundle: dict[str, Any], workspace: Path, *, reconcile_timeout: int = previous = {} def interrupted(signum: int, _frame: Any) -> None: - raise InterruptedError(f"workflow interrupted by signal {signum}") + raise WorkflowCancelledError(f"workflow interrupted by signal {signum}") for signum in (signal.SIGINT, signal.SIGTERM): previous[signum] = signal.signal(signum, interrupted) @@ -463,8 +477,9 @@ def interrupted(signum: int, _frame: Any) -> None: publish_outputs(bundle, receipt, workspace) completed = True finally: - for signum, handler in previous.items(): - signal.signal(signum, handler) + # Repeated catchable cancellation must not interrupt owned allocation cleanup. + for signum in previous: + signal.signal(signum, signal.SIG_IGN) try: if not completed and receipt_path is not None and receipt_path.exists(): deadline = time.monotonic() + reconcile_timeout @@ -503,7 +518,11 @@ def interrupted(signum: int, _frame: Any) -> None: f"submission ownership remains unresolved; intent stays fenced: {receipt_path}" ) finally: - copy_diagnostics(bundle, workspace, complete=completed) + try: + copy_diagnostics(bundle, workspace, complete=completed) + finally: + for signum, handler in previous.items(): + signal.signal(signum, handler) def main() -> int: diff --git a/infx/srt_slurm/provision.py b/infx/srt_slurm/provision.py index eab87f0c47..48c2ac26a2 100644 --- a/infx/srt_slurm/provision.py +++ b/infx/srt_slurm/provision.py @@ -3,6 +3,7 @@ from __future__ import annotations import argparse +import os import platform import shutil import subprocess @@ -85,6 +86,10 @@ def inspect_assets(config: ProvisionConfig) -> dict[str, Any]: ready = ( all(value["exists"] for value in entries.values()) and bool(indexes) and not missing_shards ) + slurm_tools = { + name: shutil.which(name) + for name in ("sbatch", "squeue", "sacct", "scancel", "srun", "scontrol") + } return { "schema_version": 1, "platform": platform.platform(), @@ -92,9 +97,8 @@ def inspect_assets(config: ProvisionConfig) -> dict[str, Any]: "paths": entries, "model_indexes": [path.name for path in indexes], "missing_model_shards": sorted(missing_shards), - "slurm_tools": { - name: shutil.which(name) for name in ("sbatch", "squeue", "sacct", "scancel") - }, + "slurm_tools": slurm_tools, + "missing_slurm_tools": [name for name, path in slurm_tools.items() if path is None], "qualification_complete": False, } @@ -103,6 +107,7 @@ def main() -> int: parser = argparse.ArgumentParser() parser.add_argument("--config", type=Path, required=True) parser.add_argument("--output", type=Path, required=True) + parser.add_argument("--operation", choices=("inspect", "provision"), required=True) args = parser.parse_args() config = ProvisionConfig.model_validate(read_json(args.config)) report = inspect_assets(config) @@ -111,7 +116,14 @@ def main() -> int: ).stdout.strip() write_json(args.output / "inventory.json", report) print(f"Prepared asset inventory written; assets_present={report['assets_present']}") - return 0 if report["assets_present"] else 1 + if not report["assets_present"] or report["missing_slurm_tools"]: + return 1 + if args.operation == "provision": + from infx.srt_slurm.provision_runtime import provision + + namespace = f"{os.environ['GITHUB_RUN_ID']}-{os.environ['GITHUB_RUN_ATTEMPT']}" + provision(config, Path.cwd(), args.output.resolve(), namespace) + return 0 if __name__ == "__main__": diff --git a/infx/srt_slurm/provision_runtime.py b/infx/srt_slurm/provision_runtime.py new file mode 100644 index 0000000000..8dfbcf9b10 --- /dev/null +++ b/infx/srt_slurm/provision_runtime.py @@ -0,0 +1,766 @@ +"""Build a new shared, offline client generation without allocating Slurm resources.""" + +from __future__ import annotations + +import fcntl +import json +import os +import platform +import re +import shutil +import tomllib +from contextlib import contextmanager, suppress +from dataclasses import dataclass +from importlib.resources import files +from pathlib import Path +from typing import TYPE_CHECKING, Any + +from infx.benchmarks.common import child_failed, read_json, run_child, write_json +from infx.benchmarks.identity import AGENTX_REVISION, LM_EVAL_REVISION +from infx.benchmarks.prepare import ClientSite, bind_file +from infx.srt_slurm.launch import RuntimeLock +from infx.srt_slurm.provision import inspect_assets, snapshot + +if TYPE_CHECKING: + from collections.abc import Iterator, Mapping, Sequence + + from infx.srt_slurm.provision import ProvisionConfig + +NATIVE_LOCK = "benchmarks/multi_node/srt-slurm-recipes/configs/prepared-runtime-lock.json" +NATIVE_REPOSITORY = "https://github.com/SemiAnalysisAI/srt-slurm.git" +NVIDIA_REPOSITORY = "https://github.com/NVIDIA/srt-slurm.git" +NVIDIA_BASE = "984180e5b8755aef85e9995048b5a16cb5336bce" +CLIENT_REPOSITORIES = { + "agentx": ("aiperf", "https://github.com/SemiAnalysisAI/aiperf.git", AGENTX_REVISION, "3.11"), + "eval": ( + "lm-eval[api]", + "https://github.com/EleutherAI/lm-evaluation-harness.git", + LM_EVAL_REVISION, + "3.12", + ), +} + + +class ProvisionStepError(RuntimeError): + """Only a stage identifier leaves the retained, credential-free child log.""" + + +def installer_environment(generation: Path, ambient: Mapping[str, str]) -> dict[str, str]: + """Public downloads receive no ambient credentials, Python injection, or user config.""" + allowed = ("PATH", "LANG", "LC_ALL", "LC_CTYPE", "TZ", "SSL_CERT_FILE", "SSL_CERT_DIR") + environment = {name: ambient[name] for name in allowed if name in ambient} + environment.update( + { + "GIT_CONFIG_NOSYSTEM": "1", + "GIT_CONFIG_GLOBAL": os.devnull, + "GIT_TERMINAL_PROMPT": "0", + "NETRC": str(generation / "empty-config/netrc"), + "UV_KEYRING_PROVIDER": "disabled", + "UV_PYTHON_INSTALL_DIR": str(generation / "python"), + "UV_CACHE_DIR": str(generation / "uv-cache"), + "UV_LINK_MODE": "copy", + "XDG_CONFIG_HOME": str(generation / "empty-config"), + "XDG_CACHE_HOME": str(generation / "cache"), + "TMPDIR": str(generation / "temporary"), + "HF_HOME": str(generation / "hf"), + "HF_HUB_DISABLE_IMPLICIT_TOKEN": "1", + "HF_HUB_DISABLE_TELEMETRY": "1", + "DO_NOT_TRACK": "1", + } + ) + return environment + + +@dataclass +class Commands: + generation: Path + environment: dict[str, str] + current_stage: str = "starting" + number: int = 0 + + def run( + self, + stage: str, + argv: Sequence[str], + *, + cwd: Path, + timeout: int, + environment: Mapping[str, str] | None = None, + ) -> str: + self.current_stage = stage + self.number += 1 + log = self.generation / "logs" / f"{self.number:02d}-{stage}.log" + print(f"Provisioning {stage}; bounded to {timeout}s; log={log}", flush=True) + write_json(self.generation / "progress.json", {"stage": stage, "log": str(log)}) + status = run_child( + argv, + env={**self.environment, **(environment or {})}, + cwd=cwd, + log=log, + timeout_seconds=timeout, + terminate_grace_seconds=10, + ) + write_json(log.with_suffix(".json"), {"stage": stage, **status}) + if child_failed(status): + raise ProvisionStepError( + f"Provisioning stage failed: {stage}; inspect its retained log" + ) + return log.read_text().strip() + + +@contextmanager +def owned_generation(root: Path, namespace: str) -> Iterator[Path]: + """One root lock, a never-reused generation, and retained incomplete work on failure.""" + if re.fullmatch(r"[A-Za-z0-9][A-Za-z0-9_.-]{0,127}", namespace) is None: + raise ValueError("invalid provisioning namespace") + if not root.is_absolute() or root.resolve() != root or root.is_relative_to("/workspace"): + raise ValueError("generation root must be a canonical shared path outside /workspace") + root.mkdir(parents=True, exist_ok=True) + descriptor = os.open(root / ".provision.lock", os.O_CREAT | os.O_RDWR | os.O_NOFOLLOW, 0o600) + with os.fdopen(descriptor, "w") as lock: + try: + fcntl.flock(lock, fcntl.LOCK_EX | fcntl.LOCK_NB) + except BlockingIOError as error: + raise ValueError("another provisioning operation owns this shared root") from error + generations = root / "generations" + generations.mkdir(exist_ok=True) + if generations.resolve() != generations: + raise ValueError("generation directory may not be a symlink") + generation = generations / namespace + generation.mkdir(exist_ok=False) + for name in ("logs", "evidence", "envs", "projects", "temporary", "empty-config"): + (generation / name).mkdir() + (generation / "empty-config/netrc").touch(mode=0o600) + write_json( + generation / "state.json", {"state": "building", "qualification_complete": False} + ) + try: + yield generation + except BaseException as error: + write_json( + generation / "state.json", + { + "state": "failed", + "error_type": type(error).__name__, + "qualification_complete": False, + }, + ) + raise + + +def private_snapshot( + hub: Path, repository: str, revision: str, source: Path, *, dataset: bool +) -> Path: + """Create only a private reference and a symlink to the unchanged canonical snapshot.""" + if re.fullmatch(r"[0-9a-f]{40}", revision) is None or source.name != revision: + raise ValueError("snapshot revision does not match the explicit source directory") + if not source.is_dir() or source.resolve() != source: + raise ValueError("source snapshot must be an existing canonical directory") + prefix = "datasets--" if dataset else "models--" + base = hub / (prefix + repository.replace("/", "--")) + (base / "snapshots").mkdir(parents=True, exist_ok=False) + target = base / "snapshots" / revision + target.symlink_to(source, target_is_directory=True) + (base / "refs").mkdir() + reference = base / "refs/main" + with reference.open("x") as stream: + stream.write(revision + "\n") + reference.chmod(0o444) + return target + + +def _project(path: Path, dependencies: list[str], python: str, *, cpu_torch: bool = False) -> None: + path.mkdir() + content = ( + '[project]\nname = "infx-prepared-environment"\nversion = "0.0.0"\n' + f'requires-python = "=={python}.*"\ndependencies = {json.dumps(dependencies)}\n' + "[tool.uv]\npackage = false\n" + ) + if cpu_torch: + content += ( + '[tool.uv.sources]\ntorch = { index = "pytorch-cpu" }\n' + '[[tool.uv.index]]\nname = "pytorch-cpu"\n' + 'url = "https://download.pytorch.org/whl/cpu"\nexplicit = true\n' + ) + (path / "pyproject.toml").write_text(content) + + +def _sync(commands: Commands, uv: str, project: Path, name: str, python: str) -> Path: + environment = commands.generation / "envs" / name + commands.run( + f"install-{name}", + [ + uv, + "sync", + "--project", + str(project), + "--frozen", + "--no-default-groups", + "--no-install-project", + "--no-editable", + "--managed-python", + "--python", + python, + ], + cwd=project, + timeout=1800, + environment={"UV_PROJECT_ENVIRONMENT": str(environment)}, + ) + return environment / "bin/python" + + +def _native_source(commands: Commands, lock: RuntimeLock) -> Path: + source = commands.generation / "native-source" + source.mkdir() + prefix = ["git", "-c", "credential.helper=", "-c", "core.askPass="] + operations = [ + ("native-init", ["init"]), + ("native-origin", ["remote", "add", "origin", lock.repository]), + ("native-fetch", ["fetch", "--no-tags", "origin", lock.revision]), + ("native-checkout", ["checkout", "--detach", lock.revision]), + ( + "native-tag", + ["fetch", "--no-tags", NVIDIA_REPOSITORY, "refs/tags/v2.2.1:refs/tags/v2.2.1"], + ), + ] + for stage, arguments in operations: + commands.run(stage, [*prefix, *arguments], cwd=source, timeout=600) + for stage, reference, expected in ( + ("native-head", "HEAD", lock.revision), + ("native-version-tag", "v2.2.1^{commit}", NVIDIA_BASE), + ): + if ( + commands.run(stage, [*prefix, "rev-parse", reference], cwd=source, timeout=30) + != expected + ): + raise ValueError("native source or version tag differs from the reviewed pin") + commands.run( + "native-lineage", + [*prefix, "merge-base", "--is-ancestor", NVIDIA_BASE, "HEAD"], + cwd=source, + timeout=30, + ) + from infx.benchmarks.common import sha256_file + + if sha256_file(source / "uv.lock") != lock.uv_lock_sha256: + raise ValueError("native dependency lock differs from the reviewed pin") + return source + + +def _wheels(commands: Commands, uv: str, source: Path, checkout: Path) -> tuple[Path, Path]: + requirements = sorted( + { + requirement + for directory in (source, checkout) + for requirement in tomllib.loads((directory / "pyproject.toml").read_text())[ + "build-system" + ]["requires"] + } + ) + project = commands.generation / "projects/build" + _project(project, requirements, "3.12") + commands.run( + "lock-build-tools", + [uv, "lock", "--project", str(project), "--python", "3.12", "--managed-python"], + cwd=project, + timeout=600, + ) + shutil.copyfile(project / "uv.lock", commands.generation / "evidence/build-uv.lock") + python = _sync(commands, uv, project, "build", "3.12") + constraints = commands.generation / "evidence/build-constraints.txt" + commands.run( + "export-build-constraints", + [ + uv, + "export", + "--project", + str(project), + "--frozen", + "--no-emit-project", + "--output-file", + str(constraints), + ], + cwd=project, + timeout=60, + ) + result = [] + for name, directory in (("native", source), ("wrapper", checkout)): + output = commands.generation / "wheels" / name + commands.run( + f"build-{name}-wheel", + [ + uv, + "build", + str(directory), + "--wheel", + "--no-build-isolation", + "--force-pep517", + "--python", + str(python), + "--out-dir", + str(output), + ], + cwd=directory, + timeout=600, + ) + wheels = list(output.glob("*.whl")) + if len(wheels) != 1: + raise ValueError("expected exactly one retained wheel per package") + result.append(wheels[0]) + return result[0], result[1] + + +_GSM_SCRIPT = """import hashlib, json, os, pathlib, re, sys +from datasets import load_dataset +from huggingface_hub import snapshot_download +mode, expected_path, output = sys.argv[1:] +hub = pathlib.Path(os.environ["HF_HUB_CACHE"]) +base = hub / "datasets--openai--gsm8k" +if mode == "online": + snapshot = pathlib.Path(snapshot_download("openai/gsm8k", repo_type="dataset", revision="main", cache_dir=str(hub))) + if not re.fullmatch("[0-9a-f]{40}", snapshot.name): raise ValueError("GSM8K revision is not immutable") + (base / "refs").mkdir(exist_ok=True) + (base / "refs/main").write_text(snapshot.name + "\\n") +revision = (base / "refs/main").read_text().strip() +dataset = load_dataset("openai/gsm8k", "main", cache_dir=os.environ["HF_DATASETS_CACHE"], **({"revision": revision} if mode == "online" else {})) +expected = json.loads(pathlib.Path(expected_path).read_text()) +if len(dataset["test"]) != 1319 or set(expected) != {str(i) for i in range(1319)}: + raise ValueError("GSM8K requires exactly the canonical 1319 test documents") +for index, document in enumerate(dataset["test"]): + digest = hashlib.sha256(json.dumps(document, indent=2, ensure_ascii=False).encode()).hexdigest() + if digest != expected[str(index)]: raise ValueError("GSM8K document differs from packaged identity") +if len(dataset["train"]) < 5: raise ValueError("GSM8K five-shot training split is incomplete") +pathlib.Path(output).write_text(json.dumps({"revision": revision, "test_documents": len(dataset["test"]), "train_documents": len(dataset["train"]), "offline": mode == "offline"}) + "\\n") +""" + + +def _clients(commands: Commands, uv: str, build_constraints: Path) -> dict[str, Path]: + result = {} + for kind, (name, repository, revision, minor) in CLIENT_REPOSITORIES.items(): + project = commands.generation / "projects" / kind + dependencies = [f"{name} @ git+{repository}@{revision}"] + if kind == "eval": + # The pinned API backend imports models.utils, which imports both packages; + # lm-eval[api] itself does not declare either dependency. + dependencies.extend(["torch>=2,<3", "transformers>=4.56,<6"]) + _project(project, dependencies, minor, cpu_torch=kind == "eval") + commands.run( + f"lock-{kind}", + [uv, "lock", "--project", str(project), "--python", minor, "--managed-python"], + cwd=project, + timeout=1800, + environment={"UV_BUILD_CONSTRAINT": str(build_constraints)}, + ) + shutil.copyfile(project / "uv.lock", commands.generation / "evidence" / f"{kind}-uv.lock") + result[kind] = _sync(commands, uv, project, kind, minor) + return result + + +_CLIENT_PROBE_SCRIPT = """import importlib.metadata, json, pathlib, sys +kind, repository, snapshot, output = sys.argv[1:] +if kind == "agentx": + from huggingface_hub import snapshot_download + from aiperf.common.tokenizer import Tokenizer + resolved = pathlib.Path(snapshot_download(repository, revision="main", local_files_only=True)).resolve() + if resolved != pathlib.Path(snapshot).resolve(): + raise ValueError("nominal tokenizer cache resolves a different serving snapshot") + tokenizer = Tokenizer.from_pretrained(repository, trust_remote_code=True) + samples = ["InferenceX tokenizer preparation.", "你好,世界。"] + tokens = [tokenizer.encode(sample) for sample in samples] + if any(not row or any(type(token) is not int for token in row) for row in tokens): + raise ValueError("prepared tokenizer returned invalid token IDs") + decoded = [tokenizer.decode(row) for row in tokens] + if any(not isinstance(text, str) or not text for text in decoded): + raise ValueError("prepared tokenizer failed to decode") + lengths = tokenizer.encode_lengths_batch(samples) + if lengths != [len(row) for row in tokens]: + raise ValueError("prepared tokenizer batch lengths differ from individual encoding") + configuration = {} + for filename in ("config.json", "tokenizer_config.json"): + path = resolved / filename + if path.is_file(): + raw = json.loads(path.read_text()) + configuration[filename] = {key: raw[key] for key in ("model_type", "tokenizer_class", "auto_map", "transformers_version") if key in raw} + result = {"repository": repository, "snapshot": str(resolved), "tokens": tokens, "decoded": decoded, "batch_lengths": lengths, "configuration": configuration} +elif kind == "eval": + from lm_eval.models.openai_completions import LocalChatCompletion + model = LocalChatCompletion(model=repository, base_url="http://127.0.0.1:1/v1/chat/completions", tokenized_requests=False, max_length=16384, eos_string="") + messages = [{"role": "user", "content": "InferenceX client preparation."}] + formatted = model.create_message([model.apply_chat_template(messages)]) + if formatted != messages: + raise ValueError("prepared eval backend changed chat messages") + result = {"backend": type(model).__name__, "messages": formatted, "tokenizer_backend": model.tokenizer_backend} +else: + raise ValueError("unknown prepared client") +result["transformers_version"] = importlib.metadata.version("transformers") +pathlib.Path(output).write_text(json.dumps(result, ensure_ascii=False) + "\\n") +""" + + +def verify_clients( + commands: Commands, config: ProvisionConfig, clients: dict[str, Path], env: dict[str, str] +) -> None: + """Exercise the installed tokenizer and API chat backend offline before hashing assets.""" + script = commands.generation / "verify-client-behavior.py" + script.write_text(_CLIENT_PROBE_SCRIPT) + for kind, python in clients.items(): + commands.run( + f"offline-{kind}-behavior", + [ + str(python), + "-I", + str(script), + kind, + config.model_repository, + str(snapshot(config, dataset=False)), + str(commands.generation / "evidence" / f"{kind}-behavior.json"), + ], + cwd=commands.generation, + timeout=600, + environment=env, + ) + + +def _offline_env(generation: Path) -> dict[str, str]: + paths = { + "HF_HOME": generation / "hf", + "HF_HUB_CACHE": generation / "hf/hub", + "HF_DATASETS_CACHE": generation / "hf/datasets", + "HF_MODULES_CACHE": generation / "hf/modules", + "AIPERF_DATASET_MMAP_CACHE_DIR": generation / "mmap", + } + for path in paths.values(): + path.mkdir(parents=True, exist_ok=True) + return { + **{key: str(path) for key, path in paths.items()}, + "HF_HUB_OFFLINE": "1", + "HF_DATASETS_OFFLINE": "1", + } + + +def _sites( + config: ProvisionConfig, clients: dict[str, Path], env: dict[str, str] +) -> dict[str, ClientSite]: + model = snapshot(config, dataset=False) + trace = snapshot(config, dataset=True) + hub = Path(env["HF_HUB_CACHE"]) + model_view = private_snapshot( + hub, config.model_repository, config.model_revision, model, dataset=False + ) + trace_view = private_snapshot( + hub, config.dataset_repository, config.dataset_revision, trace, dataset=True + ) + roots = [ + str(model_view), + str(trace_view), + str(hub / "datasets--openai--gsm8k"), + env["HF_DATASETS_CACHE"], + ] + refs = [ + str(model_view.parent.parent / "refs/main"), + str(trace_view.parent.parent / "refs/main"), + ] + return { + kind: ClientSite( + python=str(python), + distributions=["aiperf" if kind == "agentx" else "lm-eval"], + env=env, + env_unset=["PYTHONPATH", "PYTHONHOME", "BASH_ENV", "ENV", "VIRTUAL_ENV"], + asset_roots=roots, + asset_files=refs, + model_path=str(model), + timeout_seconds=14400, + terminate_grace_seconds=60, + ) + for kind, python in clients.items() + } + + +_VERIFY_SCRIPT = """import json, pathlib, sys +from infx.benchmarks.common import write_json, verify_model_snapshot_assets, verify_snapshot_assets +from infx.benchmarks.identity import capture_identity, require_source_revision, AGENTX_REVISION, LM_EVAL_REVISION +from infx.benchmarks.prepare import ClientSite, bind_file, collect_assets +from infx.benchmarks.spec import RuntimeSpec +from infx.srt_slurm.launch import verify_wrapper_source +config = json.loads(pathlib.Path(sys.argv[1]).read_text()) +root = pathlib.Path(config["generation"]) +def require_shared_python(identity, minor): + if not identity["python_version"].startswith(minor + "."): + raise ValueError("prepared interpreter has the wrong Python minor") + for value in identity["python_paths"].values(): + if not pathlib.Path(value).resolve().is_relative_to(root): + raise ValueError("prepared interpreter or standard library escapes the shared generation") +wrapper = capture_identity(sys.executable, ["infx"], dataset_loader=None) +require_shared_python(wrapper, "3.12") +native = capture_identity(config["native_python"], ["srtctl"], dataset_loader=None) +require_shared_python(native, "3.12") +write_json(root / "evidence/native-identity.json", native) +import subprocess +capabilities = json.loads(subprocess.run([config["native_python"], "-I", "-m", "srtctl.cli.prepared", "capabilities", "--json"], capture_output=True, text=True, check=True, timeout=60).stdout) +if not set(config["capabilities"]) <= set(capabilities["capabilities"]): + raise ValueError("installed native runtime lacks required capabilities") +write_json(root / "evidence/native-capabilities.json", capabilities) +verify_wrapper_source(wrapper, pathlib.Path(config["checkout"])) +write_json(root / "evidence/wrapper-identity.json", wrapper) +assets = collect_assets(ClientSite.model_validate(config["sites"]["agentx"])) +write_json(root / "evidence/assets.json", [asset.model_dump() for asset in assets]) +write_json(root / "evidence/image.json", bind_file(pathlib.Path(config["image"])).model_dump()) +for kind, raw_site in config["sites"].items(): + site = ClientSite.model_validate(raw_site) + identity = capture_identity(site.python, site.distributions, dataset_loader="semianalysis_cc_traces_weka_062126" if kind == "agentx" else None, env={**__import__("os").environ, **site.env}) + require_shared_python(identity, "3.11" if kind == "agentx" else "3.12") + require_source_revision(identity, "aiperf" if kind == "agentx" else "lm-eval", AGENTX_REVISION if kind == "agentx" else LM_EVAL_REVISION) + if kind == "agentx" and identity.get("dataset_resolution", {}).get("metadata", {}).get("hf_dataset_name") != config["dataset_repository"]: + raise ValueError("AgentX plugin resolves a different dataset") + output = root / "evidence" / (kind + "-identity.json") + write_json(output, identity) + runtime = RuntimeSpec(python=site.python, identity=bind_file(output), distributions=site.distributions, env=site.env, env_unset=site.env_unset, assets=assets, timeout_seconds=site.timeout_seconds, terminate_grace_seconds=site.terminate_grace_seconds) + verify_model_snapshot_assets(runtime, config["model_repository"], expected_revision=config["model_revision"], expected_snapshot=pathlib.Path(site.model_path)) + verify_snapshot_assets(runtime, config["dataset_repository"], expected_revision=config["dataset_revision"], only_snapshot=True) + verify_snapshot_assets(runtime, "openai/gsm8k", expected_revision=None, only_snapshot=False) + write_json(root / "evidence" / (kind + "-runtime.json"), runtime.model_dump()) +""" + + +def require_clean_checkout(commands: Commands, checkout: Path, output: Path) -> str: + """Allow only the caller's untracked artifact directory, never tracked/source changes.""" + exclusions = [] + if output.is_relative_to(checkout): + relative = output.relative_to(checkout) + if ( + relative == Path() + or output.is_relative_to(checkout / "infx") + or output.is_relative_to(checkout / ".git") + ): + raise ValueError("artifact output must be outside source and Git metadata paths") + tracked = commands.run( + "artifact-path-check", + ["git", "ls-files", "--", str(relative)], + cwd=checkout, + timeout=30, + ) + if tracked: + raise ValueError("artifact output contains tracked repository files") + exclusions = [f":(exclude,literal){relative}"] + dirty = commands.run( + "candidate-clean", + ["git", "status", "--porcelain", "--untracked-files=all", "--", ".", *exclusions], + cwd=checkout, + timeout=30, + ) + if dirty: + raise ValueError("wrapper checkout must be clean before provisioning") + head = commands.run("candidate-head", ["git", "rev-parse", "HEAD"], cwd=checkout, timeout=30) + if re.fullmatch(r"[0-9a-f]{40}", head) is None: + raise ValueError("candidate revision must be an immutable full commit") + return head + + +def publish_evidence(generation: Path, output: Path) -> None: + """Retain small review artifacts on success and interrupted/failed installations.""" + output.mkdir(parents=True, exist_ok=True) + for name in ("evidence", "logs"): + shutil.copytree(generation / name, output / name, dirs_exist_ok=True) + + +def provision( + config: ProvisionConfig, checkout: Path, output: Path, namespace: str +) -> dict[str, Any]: + """Provision a draft; deployed reader/collector identities are deliberately absent.""" + checkout, output = checkout.resolve(), output.resolve() + if platform.system() != "Linux" or platform.machine() != "x86_64": + raise ValueError("H100 provisioning must run on the shared Linux x86_64 login runner") + inventory = inspect_assets(config) + if not inventory["assets_present"]: + raise ValueError("required existing image, model, or trace assets are incomplete") + root = Path(config.shared_root) + if root.is_relative_to(Path(config.hub_cache)) or output.is_relative_to(Path(config.hub_cache)): + raise ValueError("provisioning may not write into the existing Hugging Face cache") + lock = RuntimeLock.model_validate(read_json(checkout / NATIVE_LOCK)) + if lock.repository != NATIVE_REPOSITORY: + raise ValueError("native runtime repository is not allowlisted") + uv = shutil.which("uv") + if uv is None: + raise ValueError("provisioning requires an explicitly installed uv executable") + with owned_generation(root, namespace) as generation: + commands = Commands(generation, installer_environment(generation, os.environ)) + try: + head = require_clean_checkout(commands, checkout, output) + with Path(config.image_path).open("rb") as image: + if image.read(4) != b"hsqs": + raise ValueError("serving image is not a SquashFS image") + unsquashfs = shutil.which("unsquashfs") + if unsquashfs is not None: + commands.run( + "image-superblock", + [unsquashfs, "-s", config.image_path], + cwd=generation, + timeout=60, + ) + write_json(generation / "evidence/inventory.json", inventory) + commands.run( + "managed-python", + [uv, "python", "install", "3.12", "3.11", "--no-bin"], + cwd=generation, + timeout=600, + ) + source = _native_source(commands, lock) + shutil.copyfile(source / "uv.lock", generation / "evidence/native-uv.lock") + shutil.copyfile(checkout / "uv.lock", generation / "evidence/wrapper-uv.lock") + native_wheel, wrapper_wheel = _wheels(commands, uv, source, checkout) + wheels = [bind_file(path).model_dump() for path in (native_wheel, wrapper_wheel)] + write_json(generation / "evidence/wheels.json", wheels) + interpreters = {} + for name, directory, wheel in ( + ("native", source, native_wheel), + ("wrapper", checkout, wrapper_wheel), + ): + python = _sync(commands, uv, directory, name, "3.12") + commands.run( + f"install-{name}-wheel", + [uv, "pip", "install", "--python", str(python), "--no-deps", str(wheel)], + cwd=generation, + timeout=300, + ) + interpreters[name] = python + commands.run( + "verify-native-source", + [ + str(interpreters["native"]), + "-I", + "-c", + "import pathlib,sys; from srtctl.core.prepared import verify_loader_source; verify_loader_source(pathlib.Path(sys.argv[1]))", + str(source), + ], + cwd=generation, + timeout=600, + ) + if commands.run( + "native-clean", + ["git", "status", "--porcelain", "--untracked-files=all"], + cwd=source, + timeout=30, + ): + raise ValueError("native source changed during installation") + commands.environment["UV_BUILD_CONSTRAINT"] = str( + generation / "evidence/build-constraints.txt" + ) + clients = _clients(commands, uv, generation / "evidence/build-constraints.txt") + env = _offline_env(generation) + sites = _sites(config, clients, env) + verify_clients(commands, config, clients, env) + script = generation / "materialize-gsm8k.py" + script.write_text(_GSM_SCRIPT) + expected = generation / "evidence/gsm8k-test-doc-hashes.json" + expected.write_bytes( + files("infx.benchmarks") + .joinpath("resources/gsm8k-test-doc-hashes.json") + .read_bytes() + ) + for mode in ("online", "offline"): + commands.run( + f"gsm8k-{mode}", + [ + str(clients["eval"]), + "-I", + str(script), + mode, + str(expected), + str(generation / "evidence" / f"gsm8k-{mode}.json"), + ], + cwd=generation, + timeout=600, + environment={ + **env, + "HF_HUB_OFFLINE": "0" if mode == "online" else "1", + "HF_DATASETS_OFFLINE": "0" if mode == "online" else "1", + }, + ) + modules = Path(env["HF_MODULES_CACHE"]) + if any(path.is_file() for path in modules.rglob("*")): + sites = { + name: ClientSite.model_validate( + {**site.model_dump(), "asset_roots": [*site.asset_roots, str(modules)]} + ) + for name, site in sites.items() + } + verification = generation / "verify-input.json" + write_json( + verification, + { + "generation": str(generation), + "checkout": str(checkout), + "image": config.image_path, + "sites": {name: site.model_dump() for name, site in sites.items()}, + "native_python": str(interpreters["native"]), + "capabilities": lock.capabilities, + "model_repository": config.model_repository, + "model_revision": config.model_revision, + "dataset_repository": config.dataset_repository, + "dataset_revision": config.dataset_revision, + }, + ) + probe = generation / "verify-assets.py" + probe.write_text(_VERIFY_SCRIPT) + commands.run( + "hash-and-verify-assets", + [str(interpreters["wrapper"]), "-I", str(probe), str(verification)], + cwd=generation, + timeout=4800, + environment=env, + ) + client_sites = {} + for name, site in sites.items(): + path = generation / f"{name}-site.json" + write_json(path, site.model_dump()) + path.chmod(0o444) + client_sites[name] = str(path) + if require_clean_checkout(commands, checkout, output) != head: + raise ValueError("candidate checkout changed while provisioning") + draft = { + "schema_version": 1, + "cluster": "h100-dgxc", + "native_python": str(interpreters["native"]), + "native_source": str(source), + "wrapper_python": str(interpreters["wrapper"]), + "shared_root": str(generation), + "model_snapshot": str(snapshot(config, dataset=False)), + "model_revision": config.model_revision, + "image": read_json(generation / "evidence/image.json"), + "image_reference": config.image_reference, + "client_sites": client_sites, + "mounts": {str(path): str(path) for path in (generation, Path(config.hub_cache))}, + } + write_json(generation / "site-draft.json", draft) + report = { + "schema_version": 1, + "state": "prepared-draft", + "generation": str(generation), + "source_revision": head, + "native_revision": lock.revision, + "site_draft": str(generation / "site-draft.json"), + "client_sites": client_sites, + "qualification_complete": False, + "deployment_pins_required": ["reader_revision", "collector_revision"], + } + write_json(generation / "state.json", report) + output.mkdir(parents=True, exist_ok=True) + write_json(output / "site-draft.json", draft) + write_json(output / "provisioning.json", report) + publish_evidence(generation, output) + except BaseException as error: + with suppress(OSError): + publish_evidence(generation, output) + write_json( + output / "provisioning.json", + { + "schema_version": 1, + "state": "failed", + "generation": str(generation), + "stage": commands.current_stage, + "error_type": type(error).__name__, + "qualification_complete": False, + }, + ) + raise + return report diff --git a/infx/srt_slurm/qualification.py b/infx/srt_slurm/qualification.py new file mode 100644 index 0000000000..f8b5f8d4d0 --- /dev/null +++ b/infx/srt_slurm/qualification.py @@ -0,0 +1,372 @@ +"""Nonpublishing PR qualification: full workload evidence, never a measurement receipt.""" + +from __future__ import annotations + +import argparse +import json +import os +import re +import tempfile +import zipfile +from collections.abc import Mapping +from pathlib import Path +from typing import Any + +PURPOSE = "pr-qualification" +PREFIX = "native-qualification-" +CONCURRENCIES = (1, 2, 4, 8, 16, 20, 24, 28) + + +def require_pr_event(environment: Mapping[str, str]) -> dict[str, Any]: + """A reusable-workflow input cannot turn a dispatch/push/fork into qualification.""" + if environment.get("GITHUB_EVENT_NAME") != "pull_request": + raise ValueError("nonpublishing qualification requires a pull_request event") + event = json.loads(Path(environment["GITHUB_EVENT_PATH"]).read_text()) + pr = event.get("pull_request", {}) + repository = environment.get("GITHUB_REPOSITORY") + number = event.get("number") + if ( + not repository + or type(number) is not int + or number <= 0 + or environment.get("GITHUB_REF") != f"refs/pull/{number}/merge" + or pr.get("head", {}).get("repo", {}).get("full_name") != repository + or pr.get("base", {}).get("repo", {}).get("full_name") != repository + ): + raise ValueError("nonpublishing qualification requires an actual same-repository PR") + return event + + +def qualification_artifacts(names: Any) -> bool: + return any(name.startswith(PREFIX) for name in names) + + +def matrix_rows(plan: dict[str, Any]) -> list[dict[str, Any]]: + rows = [] + for group in ("single_node", "multi_node"): + for values in (plan.get(group) or {}).values(): + rows.extend( + dict(row, **{"run-eval": False, "eval-only": False}) for row in (values or []) + ) + for group in ("evals", "agentic_evals", "multinode_evals", "multinode_agentic_evals"): + rows.extend( + dict(row, **{"run-eval": True, "eval-only": True, "eval-framework": "lm-eval"}) + for row in (plan.get(group) or []) + ) + return rows + + +def qualification_jobs(plan: dict[str, Any], root: Path) -> dict[str, Any]: + from infx.srt_slurm.job import parse_job + + rows = matrix_rows(plan) + if not rows or any(row.get("execution", {}).get("runtime") != "srt-slurm" for row in rows): + raise ValueError("qualification requires an isolated native H100 matrix") + jobs = [parse_job(row, root, {"node-count": 1}) for row in rows] + expected = {("throughput", c) for c in CONCURRENCIES} | {("eval", 28)} + if len(jobs) != 9 or {(job.mode, job.row.conc) for job in jobs} != expected: + raise ValueError("qualification requires all eight throughput points and real c28 eval") + return {job.point_id: job for job in jobs} + + +def select_qualification(plan: dict[str, Any], root: Path, environment: Mapping[str, str]) -> bool: + rows = matrix_rows(plan) + native = any(row.get("execution", {}).get("runtime") == "srt-slurm" for row in rows) + if not native or environment.get("GITHUB_EVENT_NAME") != "pull_request": + return False + require_pr_event(environment) + qualification_jobs(plan, root) + return True + + +def validate_point( + directory: Path, jobs: dict[str, Any], source: dict[str, Any], root: Path +) -> dict: + """Recheck downloaded evidence against the caller matrix and committed client policy.""" + from infx.benchmarks import agentx, eval as real_eval + from infx.benchmarks.common import child_failed, read_json, require_finite + from infx.benchmarks.spec import AgentXSpec, EvalSpec, PreparedFile, RuntimeSpec + from infx.srt_slurm.contracts import digest, load_mapping + from infx.srt_slurm.job import JobSpec, file_digest + from infx.srt_slurm.launch import RuntimeLock, effective_identity, verify_wrapper_source + from infx.srt_slurm.render import ClientPolicy, PreparedSite, client_spec + + diagnostics = directory / "native-execution" + bundle = read_json(diagnostics / "bundle.json") + if bundle.get("bundle_digest") != digest( + {k: v for k, v in bundle.items() if k != "bundle_digest"} + ): + raise ValueError("qualification bundle digest differs") + if bundle.get("source") != source: + raise ValueError("qualification source identity or purpose differs") + requested = bundle.get("requested_point_id") + if requested not in jobs: + raise ValueError("qualification point is not in the expected matrix") + job = jobs[requested] + if JobSpec.model_validate(bundle["job"]).semantic_inputs() != job.semantic_inputs(): + raise ValueError("qualification matrix semantics differ") + lock = RuntimeLock.model_validate(read_json(root / job.row.execution.runtime_lock)) + if bundle["identity"]["runtime_lock"] != lock.model_dump(): + raise ValueError("qualification native runtime differs from the committed pin") + verify_wrapper_source(bundle["identity"]["wrapper_identity"], root) + point_id = bundle["point_id"] + if not re.fullmatch(r"[a-f0-9]{64}", point_id) or directory.name != PREFIX + point_id: + raise ValueError("qualification artifact identity differs") + original = Path(bundle["directory"]) + + def retained(name: str) -> Path: + relative = Path(name).relative_to(original) + if ".." in relative.parts: + raise ValueError("qualification prepared path escapes bundle") + path = diagnostics / "prepared" / relative + if path.is_symlink() or not path.resolve().is_relative_to(directory.resolve()): + raise ValueError("qualification prepared path escapes artifact") + if name not in bundle["files"] or file_digest(path) != bundle["files"][name]: + raise ValueError("qualification prepared file digest differs") + return path + + for name in bundle["files"]: + retained(name) + spec_value = read_json(retained(str(original / "client.json"))) + runtime = RuntimeSpec.model_validate(spec_value["runtime"]) + resources = read_json(retained(str(original / "client/prepared-resources.json"))) + policy = ClientPolicy.model_validate(load_mapping(root / job.row.execution.client_policy)) + site = PreparedSite.model_validate(bundle["site"]) + if site.image_reference != job.row.image: + raise ValueError("qualification serving image differs") + computed, curve = effective_identity( + job, + site, + bundle["identity"], + runtime, + resources, + client_identity=read_json(retained(runtime.identity.path)), + ) + if (point_id, bundle["effective_curve_id"]) != (computed, curve): + raise ValueError("qualification effective identity differs") + spec = client_spec(job, policy, runtime, resources, point_id) + if spec.model_dump(mode="json") != spec_value: + raise ValueError("qualification client policy differs") + prepared = bundle["prepared"] + if not set(lock.capabilities) <= set(prepared.get("capabilities", [])): + raise ValueError("qualification native capabilities differ") + if ( + prepared.get("resources") + != { + "nodes": 1, + "gpus_per_node": 8, + "serving_gpus": 8, + "workers": 1, + "cardinality": 1, + } + or file_digest(retained(str(original / "native/manifest.json"))) + != prepared["manifest_sha256"] + ): + raise ValueError("qualification native allocation or manifest differs") + execution = read_json(diagnostics / "execution.json") + if ( + execution.get("source") != source + or any( + execution.get(key) != bundle[key] + for key in ("point_id", "execution_id", "bundle_digest") + ) + or execution.get("mode") != job.mode + or execution.get("client_exit_code") != 0 + or execution.get("native_receipt", {}).get("state") != "COMPLETED" + or execution["native_receipt"].get("manifest_sha256") != prepared["manifest_sha256"] + or not re.fullmatch(r"[1-9][0-9]*", str(execution["native_receipt"].get("job_id", ""))) + or read_json(diagnostics / "output-state.json").get("complete") is not True + ): + raise ValueError("qualification lacks successful native/client closure") + audit = read_json(directory / "results/diagnostics/client-audit.json") + if audit.get("errors") != [] or child_failed(audit["status"]): + raise ValueError("qualification client audit failed") + endpoint = audit["endpoint"] + if job.mode == "eval": + if not isinstance(spec, EvalSpec): + raise ValueError("qualification requires the real eval client") + # Revalidate the full raw corpus with locally retained, digest-checked inputs. + relocated = spec.model_copy( + update={ + key: PreparedFile( + path=str(retained(getattr(spec, key).path)), sha256=getattr(spec, key).sha256 + ) + for key in ("task", "document_identities") + } + ) + errors = real_eval.validate_outputs( + relocated, + endpoint, + sorted(directory.glob("results*.json")), + sorted(directory.glob("samples*.jsonl")), + ) + if errors: + raise ValueError("qualification real eval failed: " + "; ".join(errors)) + else: + if not isinstance(spec, AgentXSpec): + raise ValueError("qualification requires the AgentX client") + artifact_root = directory / "results" + artifact_dir = agentx.resolve_artifact_dir(artifact_root) + aggregate = read_json(artifact_dir / "profile_export_aiperf.json") + require_finite(aggregate) + errors = agentx.validate_scenario(aggregate, spec, endpoint) + errors.extend(agentx.validate_result(artifact_dir, spec.failed_request_threshold)) + metrics = read_json(artifact_dir / "server_metrics_export.json") + metrics_csv = artifact_dir / "server_metrics_export.csv" + if ( + not metrics_csv.is_file() + or not metrics_csv.stat().st_size + or not agentx.has_metric(metrics, "vllm:") + ): + errors.append("vLLM server metrics are missing") + if errors: + raise ValueError("qualification AgentX raw validation failed: " + "; ".join(errors)) + validate_normalized(directory, job, spec, bundle) + return { + "requested_point_id": requested, + "point_id": point_id, + "mode": job.mode, + "concurrency": job.row.conc, + "execution_id": bundle["execution_id"], + "bundle_digest": bundle["bundle_digest"], + "slurm_job_id": execution["native_receipt"]["job_id"], + } + + +def validate_normalized(directory: Path, job: Any, spec: Any, bundle: dict) -> None: + """Use the existing independent content validator without creating/sealing a receipt.""" + from infx.results.publication_receipt import Point, Topology, validate_point_content + + evaluation = job.mode == "eval" + results = ( + sorted(directory.glob("results*.json")) + if evaluation + else [directory / f"{bundle['point_id']}.json"] + ) + samples = sorted(directory.glob("samples*.jsonl")) if evaluation else [] + if len(results) != 1 or (evaluation and len(samples) != 1): + raise ValueError("qualification normalized output is missing or ambiguous") + point = Point( + point_id=bundle["point_id"], + execution_id=bundle["execution_id"], + bundle_digest=bundle["bundle_digest"], + native_manifest_sha256=bundle["prepared"]["manifest_sha256"], + kind=job.mode, + concurrency=job.row.conc, + source_run_id=str(bundle["source"]["run_id"]), + source_attempt=bundle["source"]["attempt"], + topology=Topology(kind="aggregate", nodes=1, serving_gpus=8, tp=8, ep=1), + artifact_ids=[1], + execution_artifact_id=1, + execution_path="unused", + normalized_artifact_id=1, + normalized_path=results[0].name, + normalized_format="lm-eval" if evaluation else "normalized", + metadata_path="meta_env.json" if evaluation else None, + required_metrics=["em_strict", "em_flexible", "n_eff"] + if evaluation + else ["output_tput_tps", "total_tput_tps", "duration_seconds"], + config={ + "model": "dsv41flash", + "hardware": "h100", + "framework": "vllm", + "precision": "fp4", + "specMethod": "mtp", + }, + task="gsm8k" if evaluation else None, + filters=["strict-match", "flexible-extract"] if evaluation else [], + sample_count=1319 if evaluation else 0, + samples_artifact_id=1 if evaluation else None, + samples_path=samples[0].name if evaluation else None, + dataset={} + if evaluation + else { + "source_type": "public_dataset", + "loader": spec.dataset_loader, + "hf_dataset_name": spec.dataset_repository, + "hf_split": "train", + "num_dataset_entries": 393, + "hf_revision": spec.dataset_revision, + }, + ) + with tempfile.TemporaryDirectory(prefix="infx-qualification-content-") as temporary: + files = [*results, *samples] + ([directory / "meta_env.json"] if evaluation else []) + with zipfile.ZipFile(Path(temporary) / "1.zip", "w") as archive: + for path in files: + archive.write(path, path.name) + validate_point_content(point, Path(temporary)) + + +def summarize(artifacts: Path, plan: dict, root: Path, environment: Mapping[str, str]) -> dict: + event = require_pr_event(environment) + jobs = qualification_jobs(plan, root) + source = { + "repository": environment["GITHUB_REPOSITORY"], + "run_id": int(environment["GITHUB_RUN_ID"]), + "attempt": int(environment["GITHUB_RUN_ATTEMPT"]), + "head_sha": environment["GITHUB_SHA"], + "purpose": PURPOSE, + "event": "pull_request", + "pull_request": event["number"], + } + intent = json.loads((artifacts / (PREFIX + "run") / "qualification-intent.json").read_text()) + if intent != { + "purpose": PURPOSE, + "publication_eligible": False, + "run_id": environment["GITHUB_RUN_ID"], + "attempt": environment["GITHUB_RUN_ATTEMPT"], + }: + raise ValueError("qualification run intent differs") + directories = sorted(path for path in artifacts.iterdir() if path.name != PREFIX + "run") + if len(directories) != 9 or any(not path.is_dir() or path.is_symlink() for path in directories): + raise ValueError("qualification requires exactly nine distinct point artifacts") + points = [validate_point(path, jobs, source, root) for path in directories] + if {point["requested_point_id"] for point in points} != set(jobs): + raise ValueError("qualification has duplicate or missing matrix points") + if len({point["execution_id"] for point in points}) != 9: + raise ValueError("qualification execution identities must be distinct") + return { + "schema_version": 1, + "purpose": PURPOSE, + "publication_eligible": False, + "source": source, + "complete": True, + "points": points, + } + + +def main() -> None: + parser = argparse.ArgumentParser() + parser.add_argument("mode", choices=("plan", "summary")) + parser.add_argument("--artifacts", type=Path) + args = parser.parse_args() + root = Path(os.environ["GITHUB_WORKSPACE"]) + plan = json.loads(os.environ["SWEEP_MATRIX"]) + if args.mode == "plan": + selected = select_qualification(plan, root, os.environ) + with Path(os.environ["GITHUB_OUTPUT"]).open("a") as stream: + stream.write(f"native-qualification={str(selected).lower()}\n") + if selected: + (root / "qualification-intent.json").write_text( + json.dumps( + { + "purpose": PURPOSE, + "publication_eligible": False, + "run_id": os.environ["GITHUB_RUN_ID"], + "attempt": os.environ["GITHUB_RUN_ATTEMPT"], + } + ) + + "\n" + ) + else: + result = summarize(args.artifacts, plan, root, os.environ) + (root / "qualification-summary.json").write_text(json.dumps(result, indent=2) + "\n") + with Path(os.environ["GITHUB_STEP_SUMMARY"]).open("a") as stream: + stream.write( + "## Nonpublishing H100 qualification\n\nEight full-duration throughput points and the complete real c28 GSM8K eval passed. " + "Diagnostic evidence only; ineligible for receipt issuance, reuse or app import.\n" + ) + + +if __name__ == "__main__": + main() diff --git a/infx/srt_slurm/qualify_cancellation.py b/infx/srt_slurm/qualify_cancellation.py new file mode 100644 index 0000000000..187dd55efa --- /dev/null +++ b/infx/srt_slurm/qualify_cancellation.py @@ -0,0 +1,640 @@ +"""Disposable, owned native lifecycle probes; never an accepted benchmark execution.""" + +from __future__ import annotations + +import argparse +import copy +import json +import re +import signal +import subprocess +import time +from pathlib import Path +from typing import Any, Literal, Self + +import yaml +from pydantic import model_validator + +from infx.benchmarks.common import verify_file, write_json +from infx.srt_slurm.contracts import load_mapping +from infx.srt_slurm.job import file_digest, read_json +from infx.srt_slurm.launch import NativeCommandError, RuntimeLock, checked_json +from infx.srt_slurm.provision_runtime import NATIVE_LOCK +from infx.srt_slurm.render import PreparedSite + +RECIPE = ( + "benchmarks/multi_node/srt-slurm-recipes/dsv41flash/vllm/h100-fp4/agentx/agg-tp8-dspark5.yaml" +) +PROFILE = "runners/srt-slurm/h100-phase1.yaml" +OWNERSHIP_CAPABILITY = "prepared-direct-listener-ownership-v1" +RESOURCES = {"nodes": 1, "gpus_per_node": 8, "serving_gpus": 8, "workers": 1, "cardinality": 1} + +# This literal stdlib client sends no requests. A closed marker is written only +# after its heartbeat file has been flushed, fsynced and closed. +CLIENT_WRITER = r""" +import argparse, json, os, signal, sys, time +from pathlib import Path +parser = argparse.ArgumentParser() +parser.add_argument("--output", required=True) +parser.add_argument("--token", required=True) +parser.add_argument("--endpoint") +args = parser.parse_args() +root = Path(args.output) +root.mkdir(exist_ok=True) +endpoint = args.endpoint or os.environ["SRT_ENDPOINT"] +identity = {"token": args.token, "job_id": os.environ["SRT_JOB_ID"], + "pid": os.getpid(), "endpoint": endpoint, "cwd": str(Path.cwd())} +stopped = None +def stop(number, frame): + global stopped + stopped = number +for number in (signal.SIGTERM, signal.SIGINT, signal.SIGHUP): + signal.signal(number, stop) +def publish(name, value): + temporary = root / (name + ".tmp") + with temporary.open("x") as stream: + json.dump(value, stream, sort_keys=True) + stream.flush() + os.fsync(stream.fileno()) + temporary.replace(root / name) +count = 0 +with (root / "heartbeat.jsonl").open("x") as stream: + while stopped is None: + stream.write(json.dumps({"sequence": count, "token": args.token}) + "\n") + stream.flush() + os.fsync(stream.fileno()) + count += 1 + if count == 1: + publish("started.json", identity) + time.sleep(0.1) +publish("closed.json", {**identity, "signal": stopped, "records": count, + "bytes": (root / "heartbeat.jsonl").stat().st_size, "writer_closed": True}) +sys.exit(128 + stopped) +""" + +# Scheduler parsing and allocation ownership stay in the pinned native runtime. +# A worker PID is evidence of container entry, not server readiness. +WORKER_PROBE = r""" +import json, sys +from pathlib import Path +from srtctl.core.prepared import validate_receipt +from srtctl.core.observation import observe_job +from srtctl.core.processes import list_step_ids +receipt = validate_receipt(Path(sys.argv[1])) +observed = observe_job(receipt["job_id"], command_timeout=5, + expected_comment=receipt["scheduler_comment"]) +steps = list_step_ids(receipt["job_id"], timeout=5) if observed["state"] == "active" else None +print(json.dumps({"observation": observed, "steps": steps})) +""" + + +class QualificationInterruptedError(RuntimeError): + """Escape subprocess selectors, which deliberately swallow InterruptedError.""" + + +class DraftSite(PreparedSite): + """A provisioned runtime without any deployment declaration.""" + + @model_validator(mode="after") + def qualification_paths(self) -> Self: + shared = Path(self.shared_root) + if shared.resolve() != shared: + raise ValueError("qualification requires canonical shared storage") + for value in ( + self.native_python, + self.native_source, + self.wrapper_python, + self.shared_root, + self.model_snapshot, + ): + self.require_visible(value) + if set(self.client_sites) != {"agentx", "eval"}: + raise ValueError("draft must retain both provisioned client-site references") + return self + + +class Native: + def __init__(self, site: DraftSite, directory: Path) -> None: + self.site = site + self.directory = directory + self.number = 0 + + def run(self, command: str, *args: str, timeout: int = 60) -> dict[str, Any]: + return self.invoke( + command, + [self.site.native_python, "-I", "-m", "srtctl.cli.submit", command, *args, "--json"], + timeout=timeout, + ) + + def probe(self, receipt: Path, *, timeout: int) -> dict[str, Any]: + return self.invoke( + "worker-observation", + [self.site.native_python, "-I", "-c", WORKER_PROBE, str(receipt)], + timeout=timeout, + ) + + def invoke(self, name: str, argv: list[str], *, timeout: int) -> dict[str, Any]: + self.number += 1 + path = self.directory / f"{self.number:04d}-{name}.json" + try: + result = checked_json(argv, timeout=timeout) + except NativeCommandError as error: + result = error.output + except BaseException as error: + write_json(path, {"state": "interrupted", "error_type": type(error).__name__}) + raise + write_json(path, result) + return result + + +def verify_inputs(root: Path, site: DraftSite) -> RuntimeLock: + lock = RuntimeLock.model_validate(read_json(root / NATIVE_LOCK)) + source = Path(site.native_source) + revision = subprocess.run( + ["git", "-C", str(source), "rev-parse", "HEAD"], + capture_output=True, + text=True, + check=True, + timeout=30, + ).stdout.strip() + dirty = subprocess.run( + ["git", "-C", str(source), "status", "--porcelain", "--untracked-files=all"], + capture_output=True, + text=True, + check=True, + timeout=30, + ).stdout + if revision != lock.revision or dirty or file_digest(source / "uv.lock") != lock.uv_lock_sha256: + raise ValueError("native runtime checkout or dependency lock differs from selected pin") + if OWNERSHIP_CAPABILITY not in lock.capabilities: + raise ValueError("runtime lock does not qualify direct listener ownership") + verify_file(site.image) + snapshot = Path(site.model_snapshot) + if ( + snapshot.name != site.model_revision + or not snapshot.is_dir() + or snapshot.resolve() != snapshot + ): + raise ValueError("draft model must be the canonical immutable serving snapshot") + return lock + + +def render_probe( + root: Path, site: DraftSite, directory: Path, namespace: str, walltime_seconds: int +) -> tuple[dict[str, Any], dict[str, Any]]: + recipe = copy.deepcopy(load_mapping(root / RECIPE)) + profile = copy.deepcopy(load_mapping(root / PROFILE)) + role = recipe.get("roles", {}).get("agg", {}) + if ( + set(recipe.get("roles", {})) != {"agg"} + or recipe.get("engine") != "vllm" + or recipe.get("frontend", {}).get("type") != "vllm" + or (role.get("nodes"), role.get("workers"), role.get("gpus")) != (1, 1, 8) + or role.get("args", {}).get("tensor-parallel-size") != 8 + or profile.get("use_exclusive_sbatch_directive") is not True + or recipe["model"]["container"] != site.image_reference + ): + raise ValueError("qualification requires the selected exclusive H100 aggregate TP8 recipe") + hours, remainder = divmod(walltime_seconds, 3600) + minutes, seconds = divmod(remainder, 60) + walltime = f"{hours:02d}:{minutes:02d}:{seconds:02d}" + recipe["slurm"]["time_limit"] = walltime + recipe["name"] = "cancellation-" + namespace + recipe["identity"] = { + "model": {"repo": recipe["model"]["path"], "revision": site.model_revision}, + "container": {"image": site.image_reference}, + } + recipe["model"].update(path=site.model_snapshot, container=site.image.path) + recipe["benchmark"] = { + "type": "custom", + "argv": [ + site.native_python, + "-I", + "-c", + CLIENT_WRITER, + "--output", + str(directory / "writer"), + "--token", + namespace, + ], + "cwd": str(directory), + "env": {"HF_HUB_OFFLINE": "1", "HF_DATASETS_OFFLINE": "1"}, + "env_unset": [ + "PYTHONPATH", + "PYTHONHOME", + "BASH_ENV", + "ENV", + "HF_TOKEN", + "HUGGING_FACE_HUB_TOKEN", + "MODAL_TOKEN_ID", + "MODAL_TOKEN_SECRET", + ], + "container_image": site.image.path, + } + profile.update( + default_time_limit=walltime, + srtctl_root=site.native_source, + output_dir=str(directory / "native-output"), + default_mounts=site.mounts, + ) + return recipe, profile + + +def worker_evidence(receipt: dict[str, Any], observation: dict[str, Any]) -> dict[str, Any] | None: + observed = observation.get("observation", {}) + if observed.get("terminal") or observed.get("state") == "failed": + raise RuntimeError("allocation ended or changed generation before the cancellation trigger") + if observed.get("state") != "active" or observed.get("identity_mismatch"): + return None + job_id = receipt["job_id"] + steps = observation.get("steps") or {} + aggregate = { + name: step + for name, step in steps.items() + if re.fullmatch(r"agg_0_.+", name) and re.fullmatch(re.escape(job_id) + r"\.\d+", step) + } + if len(aggregate) != 1: + return None + path = Path(receipt["output_dir"]) / "logs/direct-vllm-worker.json" + try: + identity = read_json(path) + except (OSError, ValueError): + return None + if ( + not isinstance(identity, dict) + or any( + type(identity.get(name)) is not int or identity[name] <= 0 + for name in ("pid", "start_ticks") + ) + or any( + not isinstance(identity.get(name), list) + or len(identity[name]) != 2 + or any(type(value) is not int or value < 0 for value in identity[name]) + for name in ("pid_namespace", "net_namespace") + ) + ): + return None + return {"allocation": observed, "aggregate_steps": aggregate, "worker": identity} + + +def writer_record( + directory: Path, name: str, receipt: dict[str, Any], namespace: str +) -> dict[str, Any]: + record = read_json(directory / "writer" / name) + if ( + record.get("token") != namespace + or record.get("job_id") != receipt["job_id"] + or type(record.get("pid")) is not int + or record["pid"] <= 0 + or record.get("cwd") != str(directory) + or not isinstance(record.get("endpoint"), str) + or not record["endpoint"].startswith("http://") + ): + raise ValueError("diagnostic writer identity differs from the owned allocation") + return record + + +def await_trigger( + native: Native, + receipt_path: Path, + receipt: dict[str, Any], + directory: Path, + namespace: str, + mode: str, + timeout: int, +) -> dict[str, Any]: + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + state = native.run( + "wait", + "--receipt", + str(receipt_path), + "--timeout", + "0.2", + "--poll", + "0.2", + timeout=min(15, max(1, int(deadline - time.monotonic()))), + ) + if ( + state.get("terminal") + or state.get("identity_mismatch") + or state.get("state") == "failed" + ): + raise RuntimeError("native allocation failed before the cancellation trigger") + observation = native.probe( + receipt_path, timeout=min(20, max(1, int(deadline - time.monotonic()))) + ) + evidence = worker_evidence(receipt, observation) + if evidence is not None: + started = directory / "writer/started.json" + if mode == "startup": + if started.exists(): + raise RuntimeError("startup probe missed the pre-client cancellation window") + return evidence + client_step = (observation.get("steps") or {}).get("benchmark-client", "") + if started.exists() and re.fullmatch( + re.escape(receipt["job_id"]) + r"\.\d+", client_step + ): + evidence["writer"] = writer_record(directory, "started.json", receipt, namespace) + return evidence + time.sleep(min(1, max(0, deadline - time.monotonic()))) + raise TimeoutError("owned allocation did not reach the requested cancellation trigger") + + +def close_owned(native: Native, receipt_path: Path, timeout: int) -> dict[str, Any]: + deadline = time.monotonic() + timeout + recovery: dict[str, Any] = {} + while receipt_path.exists() and time.monotonic() < deadline: + recovery = native.run( + "reconcile", + "--receipt", + str(receipt_path), + timeout=min(65, max(1, int(deadline - time.monotonic()))), + ) + if recovery.get("accepted_ids"): + break + time.sleep(min(1, max(0, deadline - time.monotonic()))) + if not recovery.get("accepted_ids"): + raise RuntimeError(f"submission remains unresolved and fenced: {receipt_path}") + cancelled = native.run( + "cancel-known", + "--receipt", + str(receipt_path), + timeout=max(1, int(deadline - time.monotonic())), + ) + remaining = max(0.1, deadline - time.monotonic()) + closure = native.run( + "wait-known", + "--receipt", + str(receipt_path), + "--timeout", + str(remaining), + "--poll", + "1", + timeout=max(1, int(remaining) + 5), + ) + if closure.get("terminal") is not True or closure.get("state") != "closed": + raise RuntimeError( + f"owned allocations have not reached physical terminal closure: {receipt_path}" + ) + return {"reconciled": recovery, "cancellation": cancelled, "closure": closure} + + +def verify_closed_writer( + directory: Path, receipt: dict[str, Any], namespace: str +) -> dict[str, Any]: + started = writer_record(directory, "started.json", receipt, namespace) + closed = writer_record(directory, "closed.json", receipt, namespace) + if ( + any(started[name] != closed[name] for name in ("pid", "token", "job_id", "endpoint")) + or closed.get("writer_closed") is not True + or closed.get("signal") not in (signal.SIGTERM, signal.SIGINT, signal.SIGHUP) + ): + raise ValueError("diagnostic writer did not close after interruption") + heartbeat = directory / "writer/heartbeat.jsonl" + before = file_digest(heartbeat) + count = 0 + with heartbeat.open() as stream: + for count, line in enumerate(stream, 1): + if json.loads(line) != {"sequence": count - 1, "token": namespace}: + raise ValueError("diagnostic writer output is incomplete or foreign") + if ( + count < 1 + or closed.get("records") != count + or closed.get("bytes") != heartbeat.stat().st_size + ): + raise ValueError("diagnostic writer closure does not match its final output") + time.sleep(1) + if file_digest(heartbeat) != before: + raise ValueError("diagnostic writer output changed after terminal allocation closure") + return {**closed, "heartbeat_sha256": before} + + +def retain_evidence(directory: Path, output: Path, report: dict[str, Any]) -> None: + evidence = output / "evidence" + evidence.mkdir(exist_ok=False) + copied = [] + for source in sorted(directory.rglob("*")): + if ( + source.is_symlink() + or not source.is_file() + or not source.resolve().is_relative_to(directory) + ): + continue + relative = source.relative_to(directory) + target = evidence / relative + target.parent.mkdir(parents=True, exist_ok=True) + size = source.stat().st_size + with source.open("rb") as stream: + stream.seek(max(0, size - 16 * 1024 * 1024)) + target.write_bytes(stream.read(16 * 1024 * 1024)) + copied.append( + {"path": str(relative), "source_bytes": size, "retained_bytes": target.stat().st_size} + ) + report["retained_files"] = copied + write_json(output / "qualification.json", report) + + +def qualify( + root: Path, + site_draft: Path, + output: Path, + namespace: str, + *, + mode: Literal["startup", "client"], + walltime_seconds: int, + observation_timeout_seconds: int, + cleanup_timeout_seconds: int, +) -> dict[str, Any]: + if re.fullmatch(r"[A-Za-z0-9][A-Za-z0-9_.-]{0,95}", namespace) is None: + raise ValueError("qualification namespace must be one unique safe path component") + if ( + (mode == "startup" and walltime_seconds != 300) + or (mode == "client" and not 1800 <= walltime_seconds <= 7200) + or mode not in {"startup", "client"} + ): + raise ValueError( + "startup walltime must be 300s; client walltime must be explicit 1800-7200s" + ) + if ( + not 0 < observation_timeout_seconds < walltime_seconds + or not 0 < cleanup_timeout_seconds <= 600 + ): + raise ValueError("observation/cleanup deadlines must be positive and explicitly bounded") + root = root.resolve(strict=True) + site = DraftSite.model_validate(read_json(site_draft)) + lock = verify_inputs(root, site) + base = Path(site.shared_root) / "cancellation-qualification" + base.mkdir(exist_ok=True) + if base.resolve() != base: + raise ValueError("qualification generation parent cannot be a symlink") + directory = base / namespace + output = output.resolve() + if output.is_relative_to("/workspace"): + raise ValueError("qualification artifacts must stay outside /workspace") + if output.is_relative_to(directory) or directory.is_relative_to(output): + raise ValueError("artifact output and owned runtime generation must be disjoint") + output.mkdir(parents=True, exist_ok=True) + if (output / "qualification.json").exists() or (output / "evidence").exists(): + raise FileExistsError("qualification artifact output is already owned") + directory.mkdir(exist_ok=False) + (directory / "commands").mkdir() + native = Native(site, directory / "commands") + report: dict[str, Any] = { + "schema_version": 1, + "state": "failed", + "mode": mode, + "namespace": namespace, + "qualification_complete": False, + "accepted_benchmark": False, + "lifecycle_qualified": False, + "directory": str(directory), + "native_revision": lock.revision, + "walltime_seconds": walltime_seconds, + "observation_timeout_seconds": observation_timeout_seconds, + "cleanup_timeout_seconds": cleanup_timeout_seconds, + } + receipt_path: Path | None = None + receipt: dict[str, Any] | None = None + attempted = False + error: BaseException | None = None + handlers = {} + + def interrupted(signum: int, _frame: Any) -> None: + raise QualificationInterruptedError(f"qualification interrupted by signal {signum}") + + for number in (signal.SIGINT, signal.SIGTERM): + handlers[number] = signal.signal(number, interrupted) + try: + capabilities = native.run("capabilities") + if not set(lock.capabilities).issubset(capabilities.get("capabilities", [])): + raise ValueError("installed native runtime lacks pinned qualification capabilities") + recipe, profile = render_probe(root, site, directory, namespace, walltime_seconds) + for name, value in (("recipe.yaml", recipe), ("profile.yaml", profile)): + (directory / name).write_text(yaml.safe_dump(value, sort_keys=False)) + write_json(directory / "site-draft.json", site.model_dump()) + write_json(directory / "runtime-lock.json", lock.model_dump()) + prepared = native.run( + "prepare", + "--recipe", + str(directory / "recipe.yaml"), + "--profile", + str(directory / "profile.yaml"), + "--output", + str(directory / "prepared"), + "--expected-nodes", + "1", + "--runtime-python", + site.native_python, + timeout=600, + ) + if prepared.get("state") != "prepared" or prepared.get("resources") != RESOURCES: + raise ValueError("native preflight did not resolve exactly one TP8 aggregate worker") + if prepared.get("prepared_dir") != str(directory / "prepared") or prepared.get( + "output_root" + ) != str(directory / "native-output"): + raise ValueError("native prepared paths escaped the owned qualification generation") + report["prepared"] = prepared + intent = "cancellation-qualification:" + namespace + location = native.run( + "intent-path", + "--intent", + intent, + "--cluster", + site.cluster, + "--journal-dir", + str(directory / "journal"), + ) + receipt_path = Path(location["receipt_path"]) + if receipt_path.resolve() != receipt_path or not receipt_path.is_relative_to( + directory / "journal" + ): + raise ValueError("native intent journal escaped the owned qualification generation") + report["receipt_path"] = str(receipt_path) + attempted = True + receipt = native.run( + "submit-prepared", + "--prepared-dir", + prepared["prepared_dir"], + "--intent", + intent, + "--cluster", + site.cluster, + "--journal-dir", + str(directory / "journal"), + timeout=150, + ) + report["submission"] = receipt + if receipt.get("state") != "accepted" or receipt.get("accepted_ids") != [ + receipt.get("job_id") + ]: + raise RuntimeError( + "submission was not uniquely accepted; intent stays fenced without retry" + ) + if not re.fullmatch(r"[0-9]+", str(receipt.get("job_id", ""))) or receipt.get( + "output_dir" + ) != str(directory / "native-output" / receipt["job_id"]): + raise ValueError( + "accepted native output differs from the owned qualification generation" + ) + report["trigger"] = await_trigger( + native, receipt_path, receipt, directory, namespace, mode, observation_timeout_seconds + ) + except BaseException as caught: # noqa: BLE001 - signals must retain evidence and close owned jobs + error = caught + report["error_type"] = type(caught).__name__ + report["error"] = str(caught) + finally: + # A second workflow signal must not skip the bounded owned cleanup. + for number in handlers: + signal.signal(number, signal.SIG_IGN) + try: + if attempted and receipt_path is not None: + report["cleanup"] = close_owned(native, receipt_path, cleanup_timeout_seconds) + if error is None and receipt is not None: + cancellation = report["cleanup"]["cancellation"] + requests = cancellation.get("jobs", []) + if ( + cancellation.get("state") != "cancellation_requested" + or [job.get("job_id") for job in requests] != [receipt["job_id"]] + or any(job.get("already_terminal") for job in requests) + ): + raise RuntimeError("native runtime did not confirm a new cancellation request") + closure = report["cleanup"]["closure"] + jobs = closure.get("jobs", []) + if not jobs or any(job.get("slurm_state") != "CANCELLED" for job in jobs): + raise RuntimeError("allocation closed without observed explicit cancellation") + if mode == "client": + report["writer_closure"] = verify_closed_writer(directory, receipt, namespace) + report.update(state="passed", lifecycle_qualified=True) + except BaseException as cleanup_error: # noqa: BLE001 - preserve both execution and cleanup failures + report["cleanup_error_type"] = type(cleanup_error).__name__ + report["cleanup_error"] = str(cleanup_error) + error = error or cleanup_error + finally: + for number, handler in handlers.items(): + signal.signal(number, handler) + write_json(directory / "qualification.json", report) + retain_evidence(directory, output, report) + if error is not None: + raise RuntimeError( + f"cancellation qualification failed; inspect {output / 'qualification.json'}" + ) from error + return report + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + for name in ("root", "site-draft", "output"): + parser.add_argument("--" + name, type=Path, required=True) + parser.add_argument("--namespace", required=True) + parser.add_argument("--mode", choices=("startup", "client"), required=True) + for name in ("walltime-seconds", "observation-timeout-seconds", "cleanup-timeout-seconds"): + parser.add_argument("--" + name, type=int, required=True) + arguments = vars(parser.parse_args()) + print(json.dumps(qualify(**arguments), sort_keys=True)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/infx/srt_slurm/render.py b/infx/srt_slurm/render.py index c56155f65e..1801d2bda5 100644 --- a/infx/srt_slurm/render.py +++ b/infx/srt_slurm/render.py @@ -40,7 +40,7 @@ class ClientPolicy(BaseModel): telemetry: Literal["temporary-parity-exception-no-native-power"] -class PilotSite(BaseModel): +class PreparedSite(BaseModel): """Provisioned shared paths; no implicit host/environment fallback.""" model_config = ConfigDict(extra="forbid", strict=True) @@ -57,9 +57,6 @@ class PilotSite(BaseModel): image_reference: str client_sites: dict[Literal["agentx", "eval"], str] mounts: dict[str, str] - # Receipt reader deployment is a release prerequisite, not inferred from code presence. - reader_revision: str = Field(pattern=r"^[0-9a-f]{40}$") - collector_revision: str = Field(pattern=r"^[0-9a-f]{40}$") @field_validator( "native_python", "native_source", "wrapper_python", "shared_root", "model_snapshot" @@ -113,6 +110,13 @@ def require_interpreter( raise ValueError(f"pilot interpreter must use Python {python_minor}") +class PilotSite(PreparedSite): + """Publication additionally requires independently deployed reader/collector pins.""" + + reader_revision: str = Field(pattern=r"^[0-9a-f]{40}$") + collector_revision: str = Field(pattern=r"^[0-9a-f]{40}$") + + def golden_acceptance(root: Path, policy: ClientPolicy, draft_tokens: int) -> float: path = (root / policy.golden_curve).resolve(strict=True) if not path.is_relative_to(root.resolve()): @@ -193,7 +197,7 @@ def client_spec( def render_recipe( job: JobSpec, root: Path, - site: PilotSite, + site: PreparedSite, policy: ClientPolicy, spec_path: Path, output: Path, diff --git a/infx/srt_slurm/workflow.py b/infx/srt_slurm/workflow.py index 43f4d4b18c..054450a1e9 100644 --- a/infx/srt_slurm/workflow.py +++ b/infx/srt_slurm/workflow.py @@ -13,11 +13,29 @@ from infx.benchmarks.common import write_json from infx.srt_slurm.job import parse_job from infx.srt_slurm.launch import execute, prepare -from infx.srt_slurm.render import PilotSite +from infx.srt_slurm.qualification import PURPOSE, require_pr_event +from infx.srt_slurm.render import PilotSite, PreparedSite -def load_site(environment: Mapping[str, str]) -> PilotSite: +def load_site(environment: Mapping[str, str]) -> PreparedSite: """Explain missing deployment configuration before touching preparation or Slurm.""" + purpose = environment.get("NATIVE_PURPOSE", "publication") + if purpose == PURPOSE: + require_pr_event(environment) + raw = environment.get("NATIVE_PREPARED_SITE_JSON", "") + if not raw.strip(): + raise ValueError( + "Nonpublishing qualification requires INFX_H100_PHASE1_PREPARED_SITE_JSON" + ) + try: + return PreparedSite.model_validate_json(raw) + except ValidationError: + raise ValueError( + "INFX_H100_PHASE1_PREPARED_SITE_JSON must contain valid deployment-free " + "PreparedSite JSON" + ) from None + if purpose != "publication": + raise ValueError("unsupported native execution purpose") variables = { "NATIVE_SITE_JSON": "INFX_H100_PHASE1_SITE_JSON", "NATIVE_READER_REVISION": "INFX_PHASE1_READER_REVISION", @@ -60,7 +78,7 @@ def load_site(environment: Mapping[str, str]) -> PilotSite: def main() -> int: site = load_site(os.environ) - root = Path(os.environ["GITHUB_WORKSPACE"]).resolve() + root = Path(os.environ.get("NATIVE_CHECKOUT_ROOT", os.environ["GITHUB_WORKSPACE"])).resolve() raw = json.loads(os.environ["NATIVE_CONFIG_JSON"]) if os.environ["NATIVE_AGENTX_FAST"] != "false" or os.environ["NATIVE_EVAL_LIMIT"] not in ( "", @@ -94,12 +112,45 @@ def main() -> int: ["git", "rev-parse", "HEAD"], check=True, text=True, capture_output=True ).stdout.strip(), } - bundle = prepare(job, site, root, source) - with Path(os.environ["GITHUB_ENV"]).open("a") as stream: - stream.write( - f"RESULT_FILENAME={bundle['point_id']}\nNATIVE_POINT_ID={bundle['point_id']}\nGPU_COUNT=8\n" + qualification = os.environ.get("NATIVE_PURPOSE") == PURPOSE + if qualification: + event = require_pr_event(os.environ) + if source["head_sha"] != os.environ["GITHUB_SHA"]: + raise ValueError("qualification checkout must match the actual PR workflow commit") + source.update(purpose=PURPOSE, event="pull_request", pull_request=event["number"]) + output = root / "native-qualification" if qualification else root + output.mkdir(exist_ok=True) + if qualification: + # Preparation may fail before an effective identity exists. Retain that failure + # under the requested point; the summary still requires a complete bound bundle. + with Path(os.environ["GITHUB_ENV"]).open("a") as stream: + stream.write(f"NATIVE_POINT_ID={job.point_id}\n") + write_json( + output / "preparation-request.json", + { + "requested_point_id": job.point_id, + "source": source, + }, ) - diagnostics = root / "native-execution" + try: + bundle = prepare(job, site, root, source) + except Exception as error: + if qualification: + write_json( + output / "preparation-failure.json", + { + "requested_point_id": job.point_id, + "source": source, + "error_type": type(error).__name__, + "error": str(error)[:2000], + }, + ) + raise + with Path(os.environ["GITHUB_ENV"]).open("a") as stream: + stream.write(f"NATIVE_POINT_ID={bundle['point_id']}\nGPU_COUNT=8\n") + if not qualification: + stream.write(f"RESULT_FILENAME={bundle['point_id']}\n") + diagnostics = output / "native-execution" diagnostics.mkdir(exist_ok=True) write_json( diagnostics / "prepared.json", @@ -112,7 +163,7 @@ def main() -> int: "source": source, }, ) - execute(bundle, root) + execute(bundle, output) return 0 diff --git a/infx/workflows/phase1_publication.py b/infx/workflows/phase1_publication.py index 7984bf57b6..a32e0f410a 100644 --- a/infx/workflows/phase1_publication.py +++ b/infx/workflows/phase1_publication.py @@ -28,6 +28,7 @@ inspect_archive, ) from infx.srt_slurm.contracts import digest +from infx.srt_slurm.qualification import qualification_artifacts Sha = Annotated[str, Field(pattern=r"^[0-9a-f]{64}$")] GitSha = Annotated[str, Field(pattern=r"^[0-9a-f]{40}$")] @@ -93,6 +94,8 @@ def expected_contract(approval: Approval) -> ExpectedContract: paginate=True, ) inventory = [item for page in pages for item in page["artifacts"] if not item["expired"]] + if qualification_artifacts(item["name"] for page in pages for item in page["artifacts"]): + raise ValueError("nonpublishing qualification cannot be approved for receipt issuance") def artifact(name: str) -> int: matching = [item for item in inventory if item["name"] == name] diff --git a/infx/workflows/phase1_record.py b/infx/workflows/phase1_record.py index dfc2e238ee..241dd77ff8 100644 --- a/infx/workflows/phase1_record.py +++ b/infx/workflows/phase1_record.py @@ -15,6 +15,7 @@ from infx.benchmarks.common import decode_json, read_json from infx.results.publication_receipt import PublicationRecord, inspect_archive, verify_receipt +from infx.srt_slurm.qualification import qualification_artifacts from infx.workflows.phase1_publication import api @@ -99,6 +100,12 @@ def validate_record( or source.get("conclusion") != "success" ): raise ValueError("source receipt does not match its completed original attempt") + pages = api( + f"repos/{repository}/actions/runs/{record.source_run_id}/artifacts?per_page=100", + paginate=True, + ) + if qualification_artifacts(item["name"] for page in pages for item in page["artifacts"]): + raise ValueError("nonpublishing qualification cannot become a publication record") merge = api(f"repos/{repository}/actions/runs/{record.merge_run_id}") if ( not isinstance(merge, dict) diff --git a/infx/workflows/sweep_runs.py b/infx/workflows/sweep_runs.py index 2ed116ae0e..6ce327a17d 100644 --- a/infx/workflows/sweep_runs.py +++ b/infx/workflows/sweep_runs.py @@ -6,6 +6,7 @@ from typing import Any from infx import github +from infx.srt_slurm.qualification import qualification_artifacts REUSABLE_AGGREGATE_ARTIFACTS = { "results_bmk", @@ -51,6 +52,8 @@ def pr_commit_shas(repo: str, pr_number: int, token: str) -> set[str]: def has_reusable_result_artifacts(names: set[str]) -> bool: """Return whether a run produced ingest-relevant result artifacts.""" + if qualification_artifacts(names): + return False return bool(names & REUSABLE_AGGREGATE_ARTIFACTS) or any( name.startswith("bmk_agentic_") for name in names ) diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 5831057630..46c6eb1d37 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8464,3 +8464,12 @@ - "Migrate the H100 aggregate TP8 vLLM DSV4.1-Flash AgentX lane to an isolated pinned native srt-slurm runtime with prepared Python clients, immutable measurement receipts and a separate real c28 GSM8K evaluation; preserve the image, eight concurrency points and 480-minute exclusive allocation. Power telemetry remains an explicit temporary parity exception." - "将 H100 聚合 TP8 vLLM DSV4.1-Flash AgentX 路径迁移至独立固定的原生 srt-slurm 运行时,使用预备式 Python 客户端、不可变测量回执及独立真实 c28 GSM8K eval;保留镜像、八个并发点及 480 分钟独占分配。功耗遥测保留明确临时一致性例外。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3299 + +- config-keys: + - dsv41flash-fp4-h100-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Require owned direct-vLLM listeners and explicitly disable unprovisioned native telemetry for the H100 Phase 1 pilot; qualify the complete unchanged throughput/eval workload using isolated PR-only artifacts while app publication is unavailable." + - "H100 阶段 1 试点要求直连 vLLM 监听端口的进程归属可验证,并显式关闭尚未部署的原生遥测;在 app 发布能力不可用期间,使用隔离的 PR 验收产物运行完整且参数不变的吞吐与 eval 工作负载。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3299 diff --git a/utils/test_native_provision.py b/utils/test_native_provision.py index cacc54fbf6..17b7fdb879 100644 --- a/utils/test_native_provision.py +++ b/utils/test_native_provision.py @@ -5,7 +5,7 @@ import pytest -from infx.srt_slurm.provision import ProvisionConfig, inspect_assets +from infx.srt_slurm.provision import ProvisionConfig, inspect_assets, main def asset_config(tmp_path: Path) -> ProvisionConfig: @@ -54,3 +54,33 @@ def test_inspection_rejects_shard_path_escape(tmp_path): ) with pytest.raises(ValueError, match="unsafe shard path"): inspect_assets(config) + + +def test_inspection_cli_requires_worker_and_controller_slurm_tools( + tmp_path, monkeypatch +): + config = asset_config(tmp_path) + path = tmp_path / "site.json" + path.write_text(config.model_dump_json()) + output = tmp_path / "report" + monkeypatch.setattr( + "infx.srt_slurm.provision.shutil.which", + lambda name: None if name in {"srun", "scontrol"} else "/usr/bin/" + name, + ) + monkeypatch.setattr( + "sys.argv", + [ + "provision", + "--config", + str(path), + "--output", + str(output), + "--operation", + "inspect", + ], + ) + assert main() == 1 + report = json.loads((output / "inventory.json").read_text()) + assert report["assets_present"] is True + assert report["missing_slurm_tools"] == ["srun", "scontrol"] + assert not Path(config.shared_root).exists() diff --git a/utils/test_native_qualification.py b/utils/test_native_qualification.py new file mode 100644 index 0000000000..9c417f266a --- /dev/null +++ b/utils/test_native_qualification.py @@ -0,0 +1,537 @@ +"""Exercise diagnostic-only PR execution and revalidation with controlled workload evidence.""" + +import copy +import json +import shutil +import subprocess +from pathlib import Path +from types import SimpleNamespace + +import pytest +from test_native_pilot import inputs as _pilot_inputs +from test_python_benchmark_clients import ( + bound_file, + raw_agentx, +) +from test_python_benchmark_clients import ( + eval_case as _eval_case, +) +from test_python_benchmark_clients import ( + spec_inputs as _spec_inputs, +) + +from infx.benchmarks import agentx +from infx.benchmarks import eval as real_eval +from infx.benchmarks.common import read_json, write_json +from infx.benchmarks.spec import RuntimeSpec +from infx.srt_slurm import qualification as q +from infx.srt_slurm import workflow +from infx.srt_slurm.contracts import digest, load_mapping +from infx.srt_slurm.job import file_digest, intent_id +from infx.srt_slurm.launch import effective_identity +from infx.srt_slurm.render import ClientPolicy, PreparedSite, client_spec +from infx.workflows.sweep_runs import has_reusable_result_artifacts + + +@pytest.fixture +def pilot_inputs(tmp_path): + return _pilot_inputs.__wrapped__(tmp_path) + + +@pytest.fixture +def spec_inputs(tmp_path): + return _spec_inputs.__wrapped__(tmp_path) + + +@pytest.fixture +def eval_case(tmp_path, spec_inputs): + return _eval_case.__wrapped__(tmp_path, spec_inputs) + + +@pytest.fixture +def pr_environment(tmp_path): + event = tmp_path / "event.json" + event.write_text( + json.dumps( + { + "number": 3299, + "pull_request": { + "head": {"repo": {"full_name": "SemiAnalysisAI/InferenceX"}}, + "base": {"repo": {"full_name": "SemiAnalysisAI/InferenceX"}}, + }, + } + ) + ) + return { + "GITHUB_EVENT_NAME": "pull_request", + "GITHUB_EVENT_PATH": str(event), + "GITHUB_REPOSITORY": "SemiAnalysisAI/InferenceX", + "GITHUB_REF": "refs/pull/3299/merge", + "GITHUB_SHA": "a" * 40, + "GITHUB_RUN_ID": "1234", + "GITHUB_RUN_ATTEMPT": "2", + } + + +@pytest.fixture +def plan(pilot_inputs): + root, row, scheduling, _ = pilot_inputs + write_json( + root / "runtime.json", + { + "schema_version": 1, + "repository": "https://example.invalid/native.git", + "revision": "c" * 40, + "uv_lock_sha256": "d" * 64, + "capabilities": ["prepared-v1"], + }, + ) + row = row | scheduling + return { + "single_node": {"agentic": [row | {"conc": c} for c in q.CONCURRENCIES]}, + "agentic_evals": [row | {"conc": 28}], + } + + +def test_deployment_free_site_requires_actual_pr(pilot_inputs, pr_environment): + *_, site = pilot_inputs + environment = pr_environment | { + "NATIVE_PURPOSE": q.PURPOSE, + "NATIVE_PREPARED_SITE_JSON": json.dumps( + site.model_dump(exclude={"reader_revision", "collector_revision"}) + ), + } + actual = workflow.load_site(environment) + assert type(actual) is PreparedSite + assert actual.native_python == site.native_python + with pytest.raises(ValueError, match="repository variables"): + workflow.load_site(environment | {"NATIVE_PURPOSE": "publication"}) + with pytest.raises(ValueError, match="deployment-free"): + workflow.load_site( + environment | {"NATIVE_PREPARED_SITE_JSON": site.model_dump_json()} + ) + for name in ("push", "workflow_dispatch", "pull_request_target"): + with pytest.raises(ValueError, match="pull_request event"): + workflow.load_site(environment | {"GITHUB_EVENT_NAME": name}) + event = json.loads(Path(environment["GITHUB_EVENT_PATH"]).read_text()) + event["pull_request"]["head"]["repo"]["full_name"] = "fork/InferenceX" + Path(environment["GITHUB_EVENT_PATH"]).write_text(json.dumps(event)) + with pytest.raises(ValueError, match="same-repository"): + workflow.load_site(environment) + + +def test_matrix_requires_exact_grid_and_isolates_publication( + pilot_inputs, plan, pr_environment +): + root, *_ = pilot_inputs + assert q.select_qualification(plan, root, pr_environment) + assert not q.select_qualification( + plan, root, pr_environment | {"GITHUB_EVENT_NAME": "push"} + ) + for mutate in ( + lambda value: value["single_node"]["agentic"].pop(), + lambda value: value["agentic_evals"].clear(), + lambda value: value["single_node"]["agentic"].append( + value["single_node"]["agentic"][0] + ), + lambda value: value["single_node"]["agentic"][0].pop("execution"), + ): + bad = copy.deepcopy(plan) + mutate(bad) + with pytest.raises(ValueError): + q.select_qualification(bad, root, pr_environment) + normal = {"results_bmk", "eval_results_all", "bmk_agentic_point"} + assert has_reusable_result_artifacts(normal) + assert not has_reusable_result_artifacts(normal | {"native-qualification-run"}) + + +def test_workflow_exports_only_diagnostics( + pilot_inputs, pr_environment, monkeypatch, tmp_path +): + root, row, scheduling, site = pilot_inputs + env_path = tmp_path / "github-env" + environment = pr_environment | { + "GITHUB_WORKSPACE": str(tmp_path), + "NATIVE_CHECKOUT_ROOT": str(root), + "GITHUB_ENV": str(env_path), + "NATIVE_PURPOSE": q.PURPOSE, + "NATIVE_PREPARED_SITE_JSON": json.dumps( + site.model_dump(exclude={"reader_revision", "collector_revision"}) + ), + "NATIVE_CONFIG_JSON": json.dumps(row), + "NATIVE_AGENTX_FAST": "false", + "NATIVE_EVAL_LIMIT": "", + "NATIVE_REQUIRE_POWER": "false", + "NATIVE_RUN_EVAL": "false", + "NATIVE_EVAL_ONLY": "false", + "NATIVE_PRIORITY": scheduling["priority"], + "NATIVE_QUEUE_TOKEN": scheduling["queue-token"], + } + calls = [] + + def prepare(job, prepared_site, checkout, source): + assert type(prepared_site) is PreparedSite + assert source["purpose"] == q.PURPOSE and checkout == root + return { + "point_id": "e" * 64, + "execution_id": "f" * 64, + "bundle_digest": "d" * 64, + "prepared": {"manifest_sha256": "c" * 64}, + } + + monkeypatch.setattr(workflow.os, "environ", environment) + monkeypatch.setattr( + workflow.subprocess, "run", lambda *a, **kw: SimpleNamespace(stdout="a" * 40) + ) + monkeypatch.setattr(workflow, "prepare", prepare) + monkeypatch.setattr( + workflow, "execute", lambda bundle, output: calls.append(output) + ) + assert workflow.main() == 0 + assert calls == [root / "native-qualification"] + assert dict(line.split("=", 1) for line in env_path.read_text().splitlines()) == { + "NATIVE_POINT_ID": "e" * 64, + "GPU_COUNT": "8", + } + assert ( + read_json(calls[0] / "native-execution/prepared.json")["source"]["purpose"] + == q.PURPOSE + ) + + def failed_prepare(*args): + raise ValueError("prepared image digest differs") + + monkeypatch.setattr(workflow, "prepare", failed_prepare) + with pytest.raises(ValueError, match="image digest"): + workflow.main() + failure = read_json(root / "native-qualification/preparation-failure.json") + assert failure["error_type"] == "ValueError" + assert failure["source"]["purpose"] == q.PURPOSE + assert ( + failure["requested_point_id"] + == env_path.read_text().splitlines()[-1].split("=", 1)[1] + ) + assert "RESULT_FILENAME=" not in env_path.read_text() + + +@pytest.fixture +def evidence(pilot_inputs, plan, pr_environment, spec_inputs, eval_case, tmp_path): + root, _, _, original_site = pilot_inputs + site = PreparedSite.model_validate( + original_site.model_dump(exclude={"reader_revision", "collector_revision"}) + ) + jobs = q.qualification_jobs(plan, root) + artifacts = tmp_path / "downloaded" + artifacts.mkdir() + write_json( + artifacts / "native-qualification-run/qualification-intent.json", + { + "purpose": q.PURPOSE, + "publication_eligible": False, + "run_id": "1234", + "attempt": "2", + }, + ) + source = { + "repository": pr_environment["GITHUB_REPOSITORY"], + "run_id": 1234, + "attempt": 2, + "head_sha": "a" * 40, + "purpose": q.PURPOSE, + "event": "pull_request", + "pull_request": 3299, + } + eval_spec, _, sample_file, eval_result, _ = eval_case + policy = ClientPolicy.model_validate( + load_mapping(root / next(iter(jobs.values())).row.execution.client_policy) + ) + + def build(job, variant=""): + original = tmp_path / "shared" / (job.mode + str(job.row.conc) + variant) + original.mkdir(parents=True) + runtime_data = copy.deepcopy(spec_inputs["runtime"]) + runtime_data["env"]["HF_DATASETS_CACHE"] += variant + identity = original / "client/identity.json" + identity.parent.mkdir() + shutil.copyfile(spec_inputs["runtime"]["identity"]["path"], identity) + runtime_data["identity"] = bound_file(identity) + if job.mode == "eval": + runtime_data["distributions"] = ["lm-eval"] + task = original / "client/gsm8k.yaml" + documents = original / "client/documents.json" + shutil.copyfile(eval_spec.task.path, task) + shutil.copyfile(eval_spec.document_identities.path, documents) + resources = { + "task": bound_file(task), + "document_identities": bound_file(documents), + } + else: + resources = {"dataset_revision": "f" * 40} + runtime = RuntimeSpec.model_validate(runtime_data) + write_json(original / "client/prepared-resources.json", resources) + write_json(original / "native/manifest.json", {"nodes": 1}) + installed = { + "runtime_lock": read_json(root / "runtime.json"), + "wrapper_identity": {"distributions": {"infx": {"files": {}}}}, + } + point, curve = effective_identity(job, site, installed, runtime, resources) + spec = client_spec(job, policy, runtime, resources, point) + write_json(original / "client.json", spec.model_dump(mode="json")) + bundle = { + "schema_version": 1, + "directory": str(original), + "source": source, + "requested_point_id": job.point_id, + "point_id": point, + "effective_curve_id": curve, + "execution_id": intent_id(source["repository"], "1234", "2", job.point_id), + "job": job.model_dump(mode="json", by_alias=True), + "site": site.model_dump(), + "identity": installed, + "prepared": { + "manifest_sha256": file_digest(original / "native/manifest.json"), + "capabilities": ["prepared-v1"], + "resources": { + "nodes": 1, + "gpus_per_node": 8, + "serving_gpus": 8, + "workers": 1, + "cardinality": 1, + }, + }, + "files": { + str(path): file_digest(path) + for path in original.rglob("*") + if path.is_file() + }, + } + bundle["bundle_digest"] = digest(bundle) + output = artifacts / (q.PREFIX + point) + shutil.copytree(original, output / "native-execution/prepared") + write_json(output / "native-execution/bundle.json", bundle) + write_json( + output / "native-execution/execution.json", + { + key: bundle[key] + for key in ("point_id", "execution_id", "bundle_digest", "source") + } + | { + "mode": job.mode, + "client_exit_code": 0, + "native_receipt": { + "state": "COMPLETED", + "job_id": "123", + "manifest_sha256": bundle["prepared"]["manifest_sha256"], + }, + }, + ) + write_json(output / "native-execution/output-state.json", {"complete": True}) + endpoint = ( + "http://worker:9000" if job.mode == "eval" else "http://worker.example:9123" + ) + write_json( + output / "results/diagnostics/client-audit.json", + { + "errors": [], + "endpoint": endpoint, + "status": { + "returncode": 0, + "cancelled_by_signal": None, + "timed_out": False, + "orphaned_descendants": False, + }, + }, + ) + if job.mode == "eval": + result = copy.deepcopy(eval_result) + result["config"]["model_args"]["model"] = job.row.model + write_json(output / "results_case_conc28.json", result) + shutil.copyfile(sample_file, output / "samples_case_conc28.jsonl") + write_json( + output / "meta_env.json", real_eval.eval_metadata(spec, complete=True) + ) + else: + raw = raw_agentx(output / "results") + aggregate = read_json(raw / "profile_export_aiperf.json") + aggregate["input_config"]["models"]["items"][0]["name"] = job.row.model + aggregate["input_config"]["tokenizer"]["name"] = job.row.model + aggregate["input_config"]["phases"][0]["concurrency"] = job.row.conc + write_json(raw / "profile_export_aiperf.json", aggregate) + normalized = agentx.normalize(spec, output / "results") + shutil.copyfile(normalized, output / normalized.name) + return output + + return ( + root, + artifacts, + jobs, + source, + build, + Path(eval_spec.document_identities.path), + ) + + +def test_downloaded_point_revalidates_runtime_raw_settings_and_digests(evidence): + root, _, jobs, source, build, _ = evidence + job = next(job for job in jobs.values() if job.mode == "throughput") + output = build(job) + assert q.validate_point(output, jobs, source, root)["concurrency"] == 1 + raw = output / "results/aiperf_artifacts/profile_export_aiperf.json" + data = read_json(raw) + data["input_config"]["phases"][0]["duration"] = 1200 + write_json(raw, data) + with pytest.raises(ValueError, match="profiling.duration"): + q.validate_point(output, jobs, source, root) + (output / "native-execution/prepared/client.json").write_text("{}") + with pytest.raises(ValueError, match="file digest"): + q.validate_point(output, jobs, source, root) + + +def test_full_controlled_grid_and_eval_score_are_rechecked( + evidence, plan, pr_environment, monkeypatch, tmp_path +): + root, artifacts, jobs, _, build, identities = evidence + # Replace only the independent corpus oracle with a controlled 1,319-document split. + oracle = tmp_path / "oracle/resources" + oracle.mkdir(parents=True) + shutil.copyfile(identities, oracle / "gsm8k-test-doc-hashes.json") + monkeypatch.setattr( + "infx.results.publication_receipt.files", lambda _: oracle.parent + ) + outputs = [build(job) for job in jobs.values()] + summary = q.summarize(artifacts, plan, root, pr_environment) + assert summary["complete"] is True and summary["publication_eligible"] is False + assert len(summary["points"]) == 9 + eval_output = next(path for path in outputs if list(path.glob("samples*.jsonl"))) + result = next(eval_output.glob("results*.json")) + data = read_json(result) + data["results"]["gsm8k"]["exact_match,strict-match"] = 0.1 + write_json(result, data) + with pytest.raises(ValueError, match="threshold or disagrees"): + q.summarize(artifacts, plan, root, pr_environment) + shutil.rmtree(eval_output) + with pytest.raises(ValueError, match="exactly nine"): + q.summarize(artifacts, plan, root, pr_environment) + + +def test_isolated_checkout_ignores_stale_parent_git_state(tmp_path): + parent = tmp_path / "runner" + parent.mkdir() + subprocess.run(["git", "init", str(parent)], check=True, capture_output=True) + (parent / ".git/index.lock").write_text("stale") + corrupt = parent / ".git/modules/utils/aiperf" + corrupt.mkdir(parents=True) + (corrupt / "HEAD").write_text("invalid") + fresh = parent / "native-candidate-1234-2-token" + subprocess.run(["git", "init", str(fresh)], check=True, capture_output=True) + (fresh / "candidate.txt").write_text("candidate") + subprocess.run(["git", "-C", str(fresh), "add", "candidate.txt"], check=True) + assert ( + subprocess.check_output( + ["git", "-C", str(fresh), "diff", "--cached", "--name-only"], text=True + ).strip() + == "candidate.txt" + ) + assert (parent / ".git/index.lock").read_text() == "stale" + + +def test_foreign_source_and_duplicate_matrix_points_are_rejected( + evidence, plan, pr_environment +): + root, artifacts, jobs, source, build, _ = evidence + throughput = [job for job in jobs.values() if job.mode == "throughput"] + first = build(throughput[0]) + with pytest.raises(ValueError, match="source identity"): + q.validate_point(first, jobs, source | {"attempt": 3}, root) + for job in throughput[1:]: + build(job) + build(throughput[0], variant="duplicate") + with pytest.raises(ValueError, match="duplicate or missing matrix points"): + q.summarize(artifacts, plan, root, pr_environment) + + +def test_canonical_eval_corpus_cannot_be_replaced_by_self_consistent_documents( + evidence, +): + root, _, jobs, source, build, _ = evidence + evaluation = next(job for job in jobs.values() if job.mode == "eval") + output = build(evaluation) + with pytest.raises(ValueError, match="document/target differs"): + q.validate_point(output, jobs, source, root) + + +def test_repeated_sigterm_cannot_interrupt_owned_cleanup(pilot_inputs, tmp_path): + import os + import signal + import sys + import time + + _, _, _, site = pilot_inputs + native = tmp_path / "native-control.py" + native.write_text("""import json, pathlib, sys, time +root = pathlib.Path(sys.argv[1]) +command = sys.argv[2] +receipt = root / 'receipt.json' +if command == 'intent-path': + result = {'receipt_path': str(receipt)} +elif command == 'submit-prepared': + result = {'receipt_path': str(receipt), 'state': 'accepted', 'accepted_ids': ['123'], 'job_id': '123'} + receipt.write_text(json.dumps(result)) +elif command == 'wait': + (root / 'waiting').touch() + while True: time.sleep(0.01) +elif command == 'reconcile': + result = {'accepted_ids': ['123']} +elif command == 'cancel-known': + (root / 'cancelling').touch() + while not (root / 'release').exists(): time.sleep(0.01) + result = {'accepted_ids': ['123']} +elif command == 'wait-known': + (root / 'terminal').touch() + result = {'terminal': True} +else: + raise ValueError(command) +print(json.dumps(result), flush=True) +""") + code = tmp_path / "execute.py" + code.write_text(f"""import pathlib, sys +sys.path.insert(0, {str(Path(__file__).resolve().parents[1])!r}) +from infx.srt_slurm import launch +root = pathlib.Path({str(tmp_path)!r}) +site = {site.model_dump()!r} +bundle = {{'files': {{}}, 'site': site, 'source': {{}}, 'execution_id': 'owned', 'prepared': {{'prepared_dir': str(root)}}}} +bundle['bundle_digest'] = launch.digest(bundle) +launch.verify_execution_clients = lambda *a: None +launch.native = lambda site, command, *a, **kw: launch.checked_json([sys.executable, {str(native)!r}, str(root), command], timeout=10) +launch.copy_diagnostics = lambda *a, **kw: (root / 'diagnostics-kept').write_text(str(kw['complete'])) +try: + launch.execute(bundle, root) +except launch.WorkflowCancelledError: + sys.exit(0) +raise RuntimeError('cancellation did not interrupt native wait') +""") + process = subprocess.Popen( + [sys.executable, str(code)], stdout=subprocess.PIPE, stderr=subprocess.PIPE + ) + + def wait_for(name): + deadline = time.monotonic() + 8 + while not (tmp_path / name).exists(): + if process.poll() is not None or time.monotonic() >= deadline: + raise AssertionError(f"execution did not reach {name}") + time.sleep(0.01) + + try: + wait_for("waiting") + os.kill(process.pid, signal.SIGTERM) + wait_for("cancelling") + os.kill(process.pid, signal.SIGTERM) + (tmp_path / "release").touch() + _, stderr = process.communicate(timeout=8) + assert process.returncode == 0, stderr.decode() + assert (tmp_path / "terminal").exists() + assert (tmp_path / "diagnostics-kept").read_text() == "False" + finally: + if process.poll() is None: + process.kill() + process.wait(timeout=8) diff --git a/utils/test_phase1_receipt_control.py b/utils/test_phase1_receipt_control.py index 690cc6a4bf..16ef373f70 100644 --- a/utils/test_phase1_receipt_control.py +++ b/utils/test_phase1_receipt_control.py @@ -111,6 +111,13 @@ def api(self, endpoint, **kwargs): return [{"artifacts": self.inventory}] raise AssertionError(endpoint) + def test_qualification_is_rejected_even_with_all_approved_normal_artifacts(self): + self.inventory.append( + {"id": 9999, "name": "native-qualification-run", "expired": True} + ) + with self.assertRaisesRegex(ValueError, "nonpublishing qualification"): + expected_contract(Approval.model_validate(self.approval)) + def download(self, argv, *, stdout, check): self.assertEqual( argv[-1], "repos/SemiAnalysisAI/InferenceX/actions/artifacts/1080/zip" @@ -334,7 +341,9 @@ def setUp(self): patcher.start() self.addCleanup(patcher.stop) - def api(self, endpoint): + def api(self, endpoint, **kwargs): + if endpoint.endswith("/runs/100/artifacts?per_page=100"): + return [{"artifacts": getattr(self, "source_artifacts", [])}] prefix = "repos/org/repo/actions/" if endpoint.startswith(prefix + "artifacts/"): return self.metadata[int(endpoint.rsplit("/", 1)[-1])] @@ -342,6 +351,13 @@ def api(self, endpoint): return self.runs[endpoint.removeprefix(prefix + "runs/")] raise AssertionError(endpoint) + def test_later_publication_cannot_accept_a_nonpublishing_source(self): + self.source_artifacts = [ + {"id": 9999, "name": "native-qualification-run", "expired": True} + ] + with self.assertRaisesRegex(ValueError, "nonpublishing qualification"): + self.validate() + def download(self, argv, **kwargs): return subprocess.CompletedProcess( argv, 0, stdout=self.archives[int(argv[-1].split("/")[-2])] diff --git a/utils/test_provision_runtime.py b/utils/test_provision_runtime.py new file mode 100644 index 0000000000..a1d8ade12b --- /dev/null +++ b/utils/test_provision_runtime.py @@ -0,0 +1,411 @@ +"""Shared provisioning boundaries with tiny assets and public-download collaborators.""" + +from __future__ import annotations + +import json +import os +import subprocess +import sys +from pathlib import Path + +import pytest + +from infx.benchmarks.common import ( + read_json, + verify_model_snapshot_assets, + verify_snapshot_assets, +) +from infx.benchmarks.prepare import bind_file, collect_assets +from infx.benchmarks.spec import RuntimeSpec +from infx.srt_slurm.provision import ProvisionConfig, snapshot +from infx.srt_slurm.provision_runtime import ( + Commands, + ProvisionStepError, + _GSM_SCRIPT, + _offline_env, + _sites, + installer_environment, + owned_generation, + publish_evidence, + require_clean_checkout, + verify_clients, +) + + +@pytest.fixture +def assets(tmp_path): + root = tmp_path.resolve() + config = ProvisionConfig( + schema_version=1, + shared_root=str(root / "prepared"), + hub_cache=str(root / "legacy-hub"), + image_path=str(root / "image.sqsh"), + image_reference="fixture/image@sha256:abc", + model_repository="fixture/model", + model_revision="a" * 40, + dataset_repository="semianalysisai/cc-traces-weka-062126", + dataset_revision="b" * 40, + ) + Path(config.image_path).write_bytes(b"hsqs inert test payload") + model = snapshot(config, dataset=False) + model.mkdir(parents=True) + (model / "config.json").write_text('{"model_type":"fixture"}') + (model / "tokenizer.json").write_text('{"fixture":true}') + (model / "model.safetensors.index.json").write_text( + '{"weight_map":{"weight":"one.safetensors"}}' + ) + (model / "one.safetensors").write_bytes(b"weight") + trace = snapshot(config, dataset=True) + trace.mkdir(parents=True) + (trace / "train.parquet").write_bytes(b"trace") + for source in (model, trace): + (source.parent.parent / "refs").mkdir() + (source.parent.parent / "refs/main").write_text("c" * 40) + (source.parent / ("d" * 40)).mkdir() + return config + + +def test_private_views_bind_original_snapshot_without_mutating_legacy_refs(assets): + with owned_generation(Path(assets.shared_root), "run-1") as generation: + env = _offline_env(generation) + sites = _sites( + assets, {"agentx": Path(sys.executable), "eval": Path(sys.executable)}, env + ) + gsm = Path(env["HF_HUB_CACHE"]) / "datasets--openai--gsm8k" + (gsm / "snapshots" / ("e" * 40)).mkdir(parents=True) + (gsm / "snapshots" / ("e" * 40) / "test.parquet").write_bytes(b"gsm") + (gsm / "refs").mkdir() + (gsm / "refs/main").write_text("e" * 40) + (Path(env["HF_DATASETS_CACHE"]) / "test.arrow").write_bytes(b"arrow") + identity = generation / "identity.json" + identity.write_text("{}") + runtime = RuntimeSpec( + python=sites["agentx"].python, + identity=bind_file(identity), + distributions=["aiperf"], + env=sites["agentx"].env, + env_unset=sites["agentx"].env_unset, + assets=collect_assets(sites["agentx"]), + timeout_seconds=10, + terminate_grace_seconds=1, + ) + original_model = snapshot(assets, dataset=False) + assert ( + verify_model_snapshot_assets( + runtime, + "fixture/model", + expected_revision="a" * 40, + expected_snapshot=original_model, + ) + == original_model + ) + assert ( + verify_snapshot_assets( + runtime, + assets.dataset_repository, + expected_revision="b" * 40, + only_snapshot=True, + ) + == "b" * 40 + ) + assert ( + verify_snapshot_assets( + runtime, "openai/gsm8k", expected_revision="e" * 40, only_snapshot=True + ) + == "e" * 40 + ) + assert (original_model.parent.parent / "refs/main").read_text() == "c" * 40 + assert ( + snapshot(assets, dataset=True).parent.parent / "refs/main" + ).read_text() == "c" * 40 + assert (original_model.parent / ("d" * 40)).is_dir() + assert sites["eval"].env["HF_HUB_OFFLINE"] == "1" + assert "HF_HUB_DISABLE_IMPLICIT_TOKEN" not in sites["eval"].env + + +def test_owned_generation_rejects_overlap_and_reuse_and_retains_failure(tmp_path): + root = tmp_path.resolve() / "prepared" + with pytest.raises(RuntimeError, match="controlled failure"): + with owned_generation(root, "attempt-1") as generation: + (generation / "evidence/progress.txt").write_text("completed stage") + with pytest.raises(ValueError, match="another provisioning"): + with owned_generation(root, "attempt-2"): + pytest.fail("lock was bypassed") + raise RuntimeError("controlled failure") + assert read_json(generation / "state.json") == { + "state": "failed", + "error_type": "RuntimeError", + "qualification_complete": False, + } + assert (generation / "evidence/progress.txt").read_text() == "completed stage" + with pytest.raises(FileExistsError): + with owned_generation(root, "attempt-1"): + pytest.fail("failed generation was silently reused") + assert not (root / "generations/attempt-2").exists() + + +@pytest.mark.parametrize("namespace", ["../escape", "", "/absolute"]) +def test_namespace_cannot_escape_owned_root(tmp_path, namespace): + with pytest.raises(ValueError, match="namespace"): + with owned_generation(tmp_path.resolve() / "prepared", namespace): + pytest.fail("unsafe namespace accepted") + + +def test_child_gets_no_ambient_credentials_and_failure_logs_are_publishable(tmp_path): + with owned_generation(tmp_path.resolve() / "prepared", "run") as generation: + ambient = { + "PATH": os.environ["PATH"], + "HF_TOKEN": "sensitive-hf", + "GITHUB_TOKEN": "sensitive-git", + "PYTHONPATH": "injected", + "GIT_CONFIG_COUNT": "99", + "AWS_ACCESS_KEY_ID": "sensitive-aws", + } + commands = Commands(generation, installer_environment(generation, ambient)) + result = commands.run( + "environment", + [ + sys.executable, + "-I", + "-c", + "import json,os; print(json.dumps({k:os.environ.get(k) for k in ['HF_TOKEN','GITHUB_TOKEN','PYTHONPATH','AWS_ACCESS_KEY_ID','GIT_CONFIG_COUNT','HF_HUB_DISABLE_IMPLICIT_TOKEN']}))", + ], + cwd=generation, + timeout=10, + ) + assert json.loads(result) == { + "HF_TOKEN": None, + "GITHUB_TOKEN": None, + "PYTHONPATH": None, + "AWS_ACCESS_KEY_ID": None, + "GIT_CONFIG_COUNT": None, + "HF_HUB_DISABLE_IMPLICIT_TOKEN": "1", + } + with pytest.raises(ProvisionStepError, match="controlled-exit"): + commands.run( + "controlled-exit", + [ + sys.executable, + "-I", + "-c", + "print('download collaborator failed'); raise SystemExit(7)", + ], + cwd=generation, + timeout=10, + ) + output = tmp_path / "artifact" + publish_evidence(generation, output) + status = read_json(output / "logs/02-controlled-exit.json") + assert status["returncode"] == 7 + assert not status["timed_out"] + assert ( + "download collaborator failed" + in (output / "logs/02-controlled-exit.log").read_text() + ) + assert "sensitive" not in (output / "logs/01-environment.log").read_text() + + +def git(checkout, *arguments): + return subprocess.run( + ["git", *arguments], cwd=checkout, capture_output=True, text=True, check=True + ).stdout.strip() + + +def test_checkout_exclusion_allows_only_explicit_untracked_artifact_directory(tmp_path): + checkout = tmp_path.resolve() / "checkout" + checkout.mkdir() + git(checkout, "init") + git(checkout, "config", "user.email", "test@example.invalid") + git(checkout, "config", "user.name", "Fixture") + (checkout / "tracked.txt").write_text("source") + git(checkout, "add", "tracked.txt") + git(checkout, "commit", "-m", "fixture") + expected_revision = git(checkout, "rev-parse", "HEAD") + output = checkout / "report" + output.mkdir() + (output / "inventory.json").write_text("{}") + with owned_generation(tmp_path.resolve() / "shared", "run") as generation: + commands = Commands(generation, installer_environment(generation, os.environ)) + assert require_clean_checkout(commands, checkout, output) == expected_revision + (checkout / "unrelated.txt").write_text("not an artifact") + with pytest.raises(ValueError, match="must be clean"): + require_clean_checkout(commands, checkout, output) + (checkout / "unrelated.txt").unlink() + (checkout / "tracked.txt").write_text("changed source") + with pytest.raises(ValueError, match="must be clean"): + require_clean_checkout(commands, checkout, output) + with pytest.raises(ValueError, match="outside source"): + require_clean_checkout(commands, checkout, checkout / "infx/report") + with pytest.raises(ValueError, match="tracked repository"): + require_clean_checkout(commands, checkout, checkout / "tracked.txt") + + +@pytest.fixture +def isolated_python(tmp_path): + environment = tmp_path / "python" + subprocess.run( + [sys.executable, "-m", "venv", "--without-pip", str(environment)], check=True + ) + python = environment / "bin/python" + purelib = Path( + subprocess.run( + [ + str(python), + "-I", + "-c", + "import sysconfig; print(sysconfig.get_path('purelib'))", + ], + capture_output=True, + text=True, + check=True, + ).stdout.strip() + ) + return python, purelib + + +def test_installed_client_probes_retain_behavior_and_reject_invalid_tokenization( + assets, isolated_python +): + python, purelib = isolated_python + for package in ("aiperf", "aiperf/common", "lm_eval", "lm_eval/models"): + directory = purelib / package + directory.mkdir(exist_ok=True) + (directory / "__init__.py").touch() + metadata = purelib / "transformers-5.0.dist-info" + metadata.mkdir() + (metadata / "METADATA").write_text("Name: transformers\nVersion: 5.0\n") + (purelib / "huggingface_hub.py").write_text( + "import os,pathlib\n" + "def snapshot_download(repository, *, revision, local_files_only):\n" + " assert revision == 'main' and local_files_only\n" + " root=pathlib.Path(os.environ['HF_HUB_CACHE'])/('models--'+repository.replace('/','--'))\n" + " return str(root/'snapshots'/(root/'refs/main').read_text().strip())\n" + ) + (purelib / "aiperf/common/tokenizer.py").write_text( + "import os\n" + "class Tokenizer:\n" + " @classmethod\n" + " def from_pretrained(cls, repository, *, trust_remote_code):\n" + " assert repository == 'fixture/model' and trust_remote_code\n" + " assert os.environ['HF_HUB_OFFLINE'] == '1'\n" + " return cls()\n" + " def encode(self, text): return [17,19]\n" + " def decode(self, tokens): return 'decoded text'\n" + " def encode_lengths_batch(self, texts):\n" + " return [2,2] if 'FIXTURE_INVALID_LENGTHS' not in os.environ else [1,2]\n" + ) + (purelib / "lm_eval/models/openai_completions.py").write_text( + "class LocalChatCompletion:\n" + " def __init__(self, **kwargs):\n" + " assert kwargs['tokenized_requests'] is False\n" + " self.tokenizer_backend=None\n" + " def apply_chat_template(self, messages): return messages\n" + " def create_message(self, batch): return batch[0]\n" + ) + with owned_generation(Path(assets.shared_root), "client-probes") as generation: + env = _offline_env(generation) + clients = {"agentx": python, "eval": python} + _sites(assets, clients, env) + commands = Commands(generation, installer_environment(generation, os.environ)) + verify_clients(commands, assets, clients, env) + agent = read_json(generation / "evidence/agentx-behavior.json") + assert agent["tokens"] == [[17, 19], [17, 19]] + assert agent["batch_lengths"] == [2, 2] + assert agent["snapshot"] == str(snapshot(assets, dataset=False)) + assert agent["configuration"] == {"config.json": {"model_type": "fixture"}} + evaluation = read_json(generation / "evidence/eval-behavior.json") + assert evaluation["messages"] == [ + {"role": "user", "content": "InferenceX client preparation."} + ] + assert evaluation["tokenizer_backend"] is None + with pytest.raises(ProvisionStepError, match="offline-agentx-behavior"): + verify_clients( + commands, assets, clients, {**env, "FIXTURE_INVALID_LENGTHS": "1"} + ) + assert ( + "batch lengths differ" + in (generation / "logs/03-offline-agentx-behavior.log").read_text() + ) + + +def test_gsm_materialization_pins_online_revision_then_checks_nominal_offline_lookup( + tmp_path, isolated_python +): + python, purelib = isolated_python + payload = tmp_path / "dataset.json" + payload.write_text( + json.dumps( + { + "train": [{"question": "one?", "answer": "#### 1"}] * 5, + "test": [{"question": "one?", "answer": "#### 1"}] * 1319, + } + ) + ) + (purelib / "datasets.py").write_text( + "import json,os,pathlib\n" + "def load_dataset(repository, name, **kwargs):\n" + " assert repository == 'openai/gsm8k' and name == 'main'\n" + " if os.environ['HF_HUB_OFFLINE'] == '0':\n" + " assert kwargs['revision'] == 'eeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeee'\n" + " else:\n" + " assert 'revision' not in kwargs\n" + f" return json.loads(pathlib.Path({str(payload)!r}).read_text())\n" + ) + (purelib / "huggingface_hub.py").write_text( + "import pathlib\n" + "def snapshot_download(repository, *, repo_type, revision, cache_dir):\n" + " assert (repository,repo_type,revision) == ('openai/gsm8k','dataset','main')\n" + " path=pathlib.Path(cache_dir)/'datasets--openai--gsm8k'/'snapshots'/('e'*40)\n" + " path.mkdir(parents=True)\n" + " return str(path)\n" + ) + expected = tmp_path / "hashes.json" + expected.write_text( + json.dumps( + { + str( + i + ): "198133a7f6c2d658eaef3a9bbd3695f496257a5407021ef31471834fbb2c8fe4" + for i in range(1319) + } + ) + ) + script = tmp_path / "materialize.py" + script.write_text(_GSM_SCRIPT) + output = tmp_path / "receipt.json" + for mode in ("online", "offline"): + subprocess.run( + [str(python), "-I", str(script), mode, str(expected), str(output)], + env={ + **os.environ, + "HF_HUB_CACHE": str(tmp_path / "hub"), + "HF_DATASETS_CACHE": str(tmp_path / "datasets"), + "HF_HUB_OFFLINE": "0" if mode == "online" else "1", + }, + capture_output=True, + text=True, + check=True, + ) + assert read_json(output) == { + "revision": "e" * 40, + "test_documents": 1319, + "train_documents": 5, + "offline": True, + } + data = read_json(payload) + data["test"][0]["answer"] = "changed document" + payload.write_text(json.dumps(data)) + failed = subprocess.run( + [str(python), "-I", str(script), "offline", str(expected), str(output)], + env={ + **os.environ, + "HF_HUB_CACHE": str(tmp_path / "hub"), + "HF_DATASETS_CACHE": str(tmp_path / "datasets"), + "HF_HUB_OFFLINE": "1", + }, + capture_output=True, + text=True, + check=False, + ) + assert failed.returncode != 0 + assert "GSM8K document differs" in failed.stderr diff --git a/utils/test_publication_receipt.py b/utils/test_publication_receipt.py index 0b3b16fc32..1dacdc6cd1 100644 --- a/utils/test_publication_receipt.py +++ b/utils/test_publication_receipt.py @@ -111,6 +111,26 @@ def setUp(self): } ] + def test_nonpublication_marker_and_source_purpose_cannot_be_sealed(self): + self.inventory[0]["name"] = "native-qualification-point" + with self.assertRaisesRegex(ValueError, "cannot be sealed"): + seal_receipt(self.expected, self.issuer, self.inventory, self.root) + self.inventory[0]["name"] = "bmk_result" + archive_path = self.root / "101.zip" + with zipfile.ZipFile(archive_path) as archive: + values = {name: archive.read(name) for name in archive.namelist()} + execution = json.loads(values["execution.json"]) + execution["source"]["purpose"] = "pr-qualification" + values["execution.json"] = json.dumps(execution).encode() + with zipfile.ZipFile(archive_path, "w") as archive: + for name, value in values.items(): + archive.writestr(name, value) + self.inventory[0]["digest"] = ( + "sha256:" + hashlib.sha256(archive_path.read_bytes()).hexdigest() + ) + with self.assertRaisesRegex(ValueError, "nonpublishing execution"): + seal_receipt(self.expected, self.issuer, self.inventory, self.root) + def test_preserves_original_execution_and_exact_uploaded_member(self): receipt = seal_receipt(self.expected, self.issuer, self.inventory, self.root) self.assertEqual(receipt.points[0].source_attempt, 1) diff --git a/utils/test_qualify_cancellation.py b/utils/test_qualify_cancellation.py new file mode 100644 index 0000000000..164ed80bd2 --- /dev/null +++ b/utils/test_qualify_cancellation.py @@ -0,0 +1,496 @@ +"""Real diagnostic writes/signals with an external native scheduler collaborator.""" + +import hashlib +import json +import os +import signal +import subprocess +import sys +import time +from pathlib import Path + +import pytest +import yaml + +from infx.srt_slurm.qualify_cancellation import ( + DraftSite, + qualify, + render_probe, + verify_closed_writer, +) + +# This executable replaces only the external native/Slurm boundary. Real +# preparation, path validation, trigger selection, cleanup orchestration and +# diagnostic client code run from the production module. +NATIVE_COLLABORATOR = r""" +import json, os, signal, subprocess, sys, time +from pathlib import Path +import yaml +control_path = Path(__file__).with_name("control.json") +control = json.loads(control_path.read_text()) +state_path = Path(__file__).with_name("state.json") +state = json.loads(state_path.read_text()) if state_path.exists() else {} +args = sys.argv[1:] +def value(name): return args[args.index(name) + 1] +def persist(): state_path.write_text(json.dumps(state)) +def output(payload, code=0): + persist() + print(json.dumps(payload)) + raise SystemExit(code) +def receipt_path(): return Path(state["receipt"]) +def load_receipt(): return json.loads(receipt_path().read_text()) +if args[:2] == ["-I", "-c"]: + if "--output" in args: + os.execv(sys.executable, [sys.executable, *args]) + state.setdefault("calls", []).append("worker-observation") + receipt = load_receipt() + logs = Path(receipt["output_dir"]) / "logs" + logs.mkdir(parents=True, exist_ok=True) + (logs / "direct-vllm-worker.json").write_text(json.dumps({ + "pid": 100, "start_ticks": 200, "pid_namespace": [1, 2], "net_namespace": [1, 3]})) + steps = {"agg_0_node": "71.0"} + if control["scenario"] == "foreign-step": steps = {"agg_0_node": "999.0"} + if control["scenario"] in ("client", "unclosed-writer"): + recipe = state["recipe"] + if not state.get("writer_pid"): + environment = {**os.environ, "SRT_JOB_ID": "71", "SRT_ENDPOINT": "http://node:8000"} + child = subprocess.Popen(recipe["benchmark"]["argv"], cwd=recipe["benchmark"]["cwd"], + env=environment, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, + start_new_session=True) + state["writer_pid"] = child.pid + persist() + started = Path(recipe["benchmark"]["cwd"]) / "writer/started.json" + deadline = time.monotonic() + 5 + while not started.exists() and time.monotonic() < deadline: time.sleep(0.01) + steps["benchmark-client"] = "71.1" + output({"observation": {"state": "active", "terminal": False, + "job_id": "71", "slurm_state": "RUNNING"}, "steps": steps}) +command = args[3] +state.setdefault("calls", []).append(command) +if command == "capabilities": + output({"state": "supported", "capabilities": control["capabilities"]}) +if command == "prepare": + state["recipe"] = yaml.safe_load(Path(value("--recipe")).read_text()) + profile = yaml.safe_load(Path(value("--profile")).read_text()) + state["profile"] = profile + state["prepared_dir"] = value("--output") + state["output_root"] = profile["output_dir"] + Path(state["prepared_dir"]).mkdir() + resources = {"nodes": 1, "gpus_per_node": 8, "serving_gpus": 8, "workers": 1, "cardinality": 1} + if control["scenario"] == "wrong-resources": resources["nodes"] = 2 + output({"state": "prepared", "prepared_dir": state["prepared_dir"], "resources": resources, + "output_root": state["output_root"], "manifest_sha256": "a" * 64}) +if command == "intent-path": + path = Path(value("--journal-dir")) / "owned-intent/receipt.json" + if control["scenario"] == "escaping-intent": path = control_path.parent / "foreign-receipt.json" + state["receipt"] = str(path) + output({"state": "intent", "receipt_path": str(path)}) +if command == "submit-prepared": + path = receipt_path() + path.parent.mkdir(parents=True) + receipt = {"state": "accepted", "job_id": "71", "accepted_ids": ["71"], + "receipt_path": str(path), "output_dir": str(Path(state["output_root"]) / "71")} + if control["scenario"] in ("ambiguous", "unresolved"): + receipt.update(state="unknown", accepted_ids=[]) + path.write_text(json.dumps(receipt)) + if control["scenario"] == "interrupt-submit": + persist() + control_path.with_name("submit-pending").touch() + time.sleep(30) + output(receipt, 0 if receipt["state"] == "accepted" else 2) +if command == "wait": + output({"state": "unknown", "terminal": False, "slurm_state": "RUNNING"}, 2) +if command == "reconcile": + receipt = load_receipt() + if control["scenario"] == "ambiguous": + receipt["accepted_ids"] = ["71", "72"] + receipt_path().write_text(json.dumps(receipt)) + output(receipt, 0 if receipt["state"] == "accepted" else 2) +if command == "cancel-known": + receipt = load_receipt() + state["cancelled_ids"] = receipt["accepted_ids"] + if state.get("writer_pid"): + os.kill(state["writer_pid"], signal.SIGTERM) + closed = Path(state["recipe"]["benchmark"]["cwd"]) / "writer/closed.json" + deadline = time.monotonic() + 5 + while not closed.exists() and time.monotonic() < deadline: time.sleep(0.01) + if control["scenario"] == "unclosed-writer" and closed.exists(): closed.unlink() + output({"state": "cancellation_requested", "jobs": [ + {"state": "cancellation_requested", "job_id": job, + "already_terminal": control["scenario"] == "already-terminal"} for job in receipt["accepted_ids"]]}) +if command == "wait-known": + closed = control["scenario"] != "cleanup-unresolved" + output({"state": "closed" if closed else "unknown", "terminal": closed, + "jobs": [{"job_id": job, "terminal": closed, "slurm_state": "CANCELLED" if closed else "COMPLETING"} + for job in load_receipt()["accepted_ids"]]}, 0 if closed else 2) +raise SystemExit("unexpected external native command: " + command) +""" + + +def store(path, value): + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(value)) + + +@pytest.fixture +def pilot(tmp_path): + # Resolve macOS /var aliases because production requires canonical mounts. + root = tmp_path.resolve() + checkout = root / "checkout" + checkout.mkdir() + source = root / "native-source" + source.mkdir() + (source / "uv.lock").write_text("controlled-runtime-lock\n") + for arguments in ( + ["init", "--quiet"], + ["add", "uv.lock"], + [ + "-c", + "user.name=Fixture", + "-c", + "user.email=fixture@example.test", + "commit", + "--quiet", + "-m", + "fixture", + ], + ): + subprocess.run( + ["git", "-C", str(source), *arguments], check=True, capture_output=True + ) + revision = subprocess.check_output( + ["git", "-C", str(source), "rev-parse", "HEAD"], text=True + ).strip() + native_python = root / "native-python" + native_python.write_text(f"#!{sys.executable}\n" + NATIVE_COLLABORATOR) + native_python.chmod(0o755) + image = root / "image.sqsh" + image.write_bytes(b"controlled squash contents") + shared = root / "shared" + shared.mkdir() + snapshot = root / "hub/models--controlled--model/snapshots" / ("b" * 40) + snapshot.mkdir(parents=True) + draft = { + "schema_version": 1, + "cluster": "h100-dgxc", + "native_python": str(native_python), + "native_source": str(source), + "wrapper_python": str(native_python), + "shared_root": str(shared), + "model_snapshot": str(snapshot), + "model_revision": "b" * 40, + "image": { + "path": str(image), + "sha256": hashlib.sha256(image.read_bytes()).hexdigest(), + }, + "image_reference": "controlled/image:retained", + "client_sites": { + "agentx": str(root / "agentx.json"), + "eval": str(root / "eval.json"), + }, + "mounts": {str(root): str(root)}, + } + capabilities = ["prepared-v1", "prepared-direct-listener-ownership-v1"] + store( + checkout + / "benchmarks/multi_node/srt-slurm-recipes/configs/prepared-runtime-lock.json", + { + "schema_version": 1, + "repository": "https://example.test/native.git", + "revision": revision, + "uv_lock_sha256": hashlib.sha256( + (source / "uv.lock").read_bytes() + ).hexdigest(), + "capabilities": capabilities, + }, + ) + recipe = { + "schema": 2, + "name": "controlled", + "engine": "vllm", + "slurm": {"time_limit": "08:00:00"}, + "model": { + "path": "controlled/model", + "container": "controlled/image:retained", + "precision": "fp4", + }, + "frontend": {"type": "vllm"}, + "roles": { + "agg": { + "nodes": 1, + "workers": 1, + "gpus": 8, + "env": {"UNCHANGED": "retained"}, + "args": { + "tensor-parallel-size": 8, + "max-model-len": 8192, + "max-num-batched-tokens": 256, + }, + } + }, + } + recipe_path = ( + checkout + / "benchmarks/multi_node/srt-slurm-recipes/dsv41flash/vllm/h100-fp4/agentx/agg-tp8-dspark5.yaml" + ) + recipe_path.parent.mkdir(parents=True) + recipe_path.write_text(yaml.safe_dump(recipe)) + profile_path = checkout / "runners/srt-slurm/h100-phase1.yaml" + profile_path.parent.mkdir(parents=True) + profile_path.write_text( + yaml.safe_dump( + { + "use_exclusive_sbatch_directive": True, + "default_account": "controlled-account", + "default_partition": "controlled-partition", + "gpus_per_node": 8, + } + ) + ) + store(root / "site-draft.json", draft) + store(root / "control.json", {"scenario": "startup", "capabilities": capabilities}) + yield root, checkout, draft + state = root / "state.json" + if state.exists() and (pid := json.loads(state.read_text()).get("writer_pid")): + try: + os.kill(pid, signal.SIGKILL) + except ProcessLookupError: + pass + + +def run_probe(pilot, *, scenario="startup", mode="startup", namespace="probe"): + root, checkout, _ = pilot + control = json.loads((root / "control.json").read_text()) + store(root / "control.json", {**control, "scenario": scenario}) + return qualify( + checkout, + root / "site-draft.json", + root / "artifacts", + namespace, + mode=mode, + walltime_seconds=300 if mode == "startup" else 1800, + observation_timeout_seconds=1 if scenario == "foreign-step" else 10, + cleanup_timeout_seconds=1 if scenario == "unresolved" else 10, + ) + + +def test_startup_cancels_the_entered_worker_and_preserves_diagnostic_evidence(pilot): + root, _, _ = pilot + report = run_probe(pilot) + state = json.loads((root / "state.json").read_text()) + assert report["state"] == "passed" + assert report["lifecycle_qualified"] is True + assert report["qualification_complete"] is False + assert report["accepted_benchmark"] is False + assert report["trigger"]["aggregate_steps"] == {"agg_0_node": "71.0"} + assert state["cancelled_ids"] == ["71"] + assert state["calls"].count("submit-prepared") == 1 + assert state["recipe"]["roles"]["agg"]["args"] == { + "tensor-parallel-size": 8, + "max-model-len": 8192, + "max-num-batched-tokens": 256, + } + assert state["profile"]["default_account"] == "controlled-account" + assert state["profile"]["default_partition"] == "controlled-partition" + assert state["profile"]["default_time_limit"] == "00:05:00" + assert state["recipe"]["slurm"]["time_limit"] == "00:05:00" + retained = root / "artifacts/evidence/native-output/71/logs/direct-vllm-worker.json" + assert json.loads(retained.read_text())["start_ticks"] == 200 + with pytest.raises(FileExistsError): + qualify( + pilot[1], + root / "site-draft.json", + root / "another-output", + "probe", + mode="startup", + walltime_seconds=300, + observation_timeout_seconds=10, + cleanup_timeout_seconds=10, + ) + + +def test_client_cancellation_requires_a_real_signal_closed_writer(pilot): + root, _, _ = pilot + report = run_probe(pilot, scenario="client", mode="client") + assert report["state"] == "passed" + assert report["writer_closure"]["signal"] == signal.SIGTERM + assert report["writer_closure"]["writer_closed"] is True + assert report["writer_closure"]["records"] >= 1 + assert report["trigger"]["writer"]["pid"] == report["writer_closure"]["pid"] + assert report["writer_closure"]["endpoint"] == "http://node:8000" + assert report["cleanup"]["closure"]["terminal"] is True + heartbeat = root / "artifacts/evidence/writer/heartbeat.jsonl" + assert ( + len(heartbeat.read_text().splitlines()) == report["writer_closure"]["records"] + ) + + +@pytest.mark.parametrize( + "scenario, expected_cancelled, message", + [ + ("ambiguous", ["71", "72"], "uniquely accepted"), + ("unresolved", None, "uniquely accepted"), + ("cleanup-unresolved", ["71"], "physical terminal closure"), + ("already-terminal", ["71"], "new cancellation request"), + ("foreign-step", ["71"], "cancellation trigger"), + ("unclosed-writer", ["71"], "closed.json"), + ], +) +def test_failures_remain_unqualified_and_cancel_only_known_native_ownership( + pilot, scenario, expected_cancelled, message +): + root, _, _ = pilot + with pytest.raises(RuntimeError, match="inspect"): + run_probe( + pilot, + scenario=scenario, + mode="client" if scenario == "unclosed-writer" else "startup", + ) + report = json.loads((root / "artifacts/qualification.json").read_text()) + state = json.loads((root / "state.json").read_text()) + assert report["state"] == "failed" + assert report["lifecycle_qualified"] is False + assert report["accepted_benchmark"] is False + assert state.get("cancelled_ids") == expected_cancelled + assert state["calls"].count("submit-prepared") == 1 + assert message in report.get("error", "") + report.get("cleanup_error", "") + if scenario == "ambiguous": + assert report["cleanup"]["closure"]["terminal"] is True + assert "worker-observation" not in state["calls"] + if scenario == "unresolved": + assert "fenced" in report["cleanup_error"] + + +@pytest.mark.parametrize("scenario", ["wrong-resources", "escaping-intent"]) +def test_native_preflight_rejects_wrong_demand_or_unowned_journal_before_submit( + pilot, scenario +): + root, _, _ = pilot + with pytest.raises(RuntimeError, match="inspect"): + run_probe(pilot, scenario=scenario) + state = json.loads((root / "state.json").read_text()) + assert "submit-prepared" not in state["calls"] + assert "cancel-known" not in state["calls"] + assert ( + json.loads((root / "artifacts/qualification.json").read_text())["state"] + == "failed" + ) + + +def test_unqualified_pin_and_deployment_pins_fail_before_native_submission(pilot): + root, _, draft = pilot + with pytest.raises(ValueError, match="Extra inputs"): + DraftSite.model_validate({**draft, "reader_revision": "a" * 40}) + source = Path(draft["native_source"]) + (source / "uv.lock").write_text("changed dependency resolution\n") + with pytest.raises(ValueError, match="differs from selected pin"): + run_probe(pilot) + assert not (root / "state.json").exists() + + +def test_owned_output_cannot_alias_the_shared_generation(pilot): + root, checkout, draft = pilot + owned = Path(draft["shared_root"]) / "cancellation-qualification/probe" + with pytest.raises(ValueError, match="disjoint"): + qualify( + checkout, + root / "site-draft.json", + owned / "artifacts", + "probe", + mode="startup", + walltime_seconds=300, + observation_timeout_seconds=10, + cleanup_timeout_seconds=10, + ) + assert not owned.exists() + assert not (root / "state.json").exists() + + +def test_literal_writer_accepts_endpoint_and_rejects_incomplete_closed_evidence(pilot): + _, checkout, draft = pilot + directory = Path(draft["shared_root"]) / "standalone-writer" + directory.mkdir() + recipe, _ = render_probe( + checkout, DraftSite.model_validate(draft), directory, "writer-test", 1800 + ) + argv = [*recipe["benchmark"]["argv"], "--endpoint", "http://explicit:8000"] + with subprocess.Popen( + argv, cwd=directory, env={**os.environ, "SRT_JOB_ID": "81"} + ) as child: + try: + deadline = time.monotonic() + 5 + while not (directory / "writer/started.json").exists(): + if child.poll() is not None or time.monotonic() >= deadline: + pytest.fail("diagnostic writer did not produce its start record") + time.sleep(0.01) + child.send_signal(signal.SIGTERM) + assert child.wait(timeout=5) == 128 + signal.SIGTERM + finally: + if child.poll() is None: + child.kill() + child.wait(timeout=5) + receipt = {"job_id": "81"} + closed = verify_closed_writer(directory, receipt, "writer-test") + assert closed["endpoint"] == "http://explicit:8000" + assert closed["pid"] == child.pid + with (directory / "writer/heartbeat.jsonl").open("a") as stream: + stream.write('{"sequence":99,"token":"writer-test"}\n') + with pytest.raises(ValueError, match="incomplete or foreign"): + verify_closed_writer(directory, receipt, "writer-test") + + +def test_signal_before_submission_stdout_recovers_receipt_and_closes_allocation(pilot): + root, checkout, _ = pilot + control = json.loads((root / "control.json").read_text()) + store(root / "control.json", {**control, "scenario": "interrupt-submit"}) + argv = [ + sys.executable, + "-m", + "infx.srt_slurm.qualify_cancellation", + "--root", + str(checkout), + "--site-draft", + str(root / "site-draft.json"), + "--output", + str(root / "artifacts"), + "--namespace", + "probe", + "--mode", + "startup", + "--walltime-seconds", + "300", + "--observation-timeout-seconds", + "10", + "--cleanup-timeout-seconds", + "10", + ] + with subprocess.Popen( + argv, + cwd=Path(__file__).resolve().parents[1], + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + text=True, + ) as process: + try: + deadline = time.monotonic() + 10 + while not (root / "submit-pending").exists(): + if process.poll() is not None or time.monotonic() >= deadline: + pytest.fail( + "external native submission did not reach the acceptance window" + ) + time.sleep(0.01) + process.send_signal(signal.SIGTERM) + _, stderr = process.communicate(timeout=15) + assert process.returncode != 0 + assert "cancellation qualification failed" in stderr + finally: + if process.poll() is None: + process.kill() + process.communicate(timeout=5) + report = json.loads((root / "artifacts/qualification.json").read_text()) + state = json.loads((root / "state.json").read_text()) + assert report["error_type"] == "QualificationInterruptedError" + assert report["cleanup"]["closure"]["terminal"] is True + assert report["lifecycle_qualified"] is False + assert state["cancelled_ids"] == ["71"] + assert state["calls"].count("submit-prepared") == 1 From 065cb56de373dc89343f17307e56851cd0a4df24 Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sat, 19 Sep 2026 21:18:39 -0400 Subject: [PATCH 11/16] fix: prepare real offline H100 client caches and probes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 修复私有 Hugging Face 引用格式,按固定版本准备 AgentX 数据集缓存并逐行验证离线加载,使用正式兼容补丁校验 lm-eval 客户端行为。 --- .github/workflows/run-sweep.yml | 2 +- docs/srt-slurm-phase1.md | 4 +- docs/srt-slurm-phase1_zh.md | 4 +- infx/srt_slurm/provision_runtime.py | 121 ++++++++++++++-- utils/test_provision_runtime.py | 217 +++++++++++++++++++++++++--- 5 files changed, 313 insertions(+), 35 deletions(-) diff --git a/.github/workflows/run-sweep.yml b/.github/workflows/run-sweep.yml index 696afe87d0..8c751ab09d 100644 --- a/.github/workflows/run-sweep.yml +++ b/.github/workflows/run-sweep.yml @@ -396,7 +396,7 @@ jobs: id: qualification env: SWEEP_MATRIX: ${{ steps.setup.outputs.search-space-config }} - run: uv run --no-project --python 3.12 --with pydantic --with pyyaml python -m infx.srt_slurm.qualification plan + run: uv run --no-project --exclude-newer PT12H --python 3.12 --with pydantic --with pyyaml python -m infx.srt_slurm.qualification plan - name: Retain nonpublication intent if: steps.qualification.outputs.native-qualification == 'true' diff --git a/docs/srt-slurm-phase1.md b/docs/srt-slurm-phase1.md index 12982258d7..ee196b8525 100644 --- a/docs/srt-slurm-phase1.md +++ b/docs/srt-slurm-phase1.md @@ -83,10 +83,12 @@ Provision on shared Linux storage visible to the H100 login host and compute con 1. Install the pinned native source using its committed `uv.lock` and a noneditable environment (`uv sync --frozen --no-editable --no-dev --python 3.12`). Keep that source checkout clean. Preserve the hashed Linux-built wheel and its build-tool constraints: `uv.lock` freezes runtime dependencies but does not pin the upstream Hatch build dependencies. If rebuilding, fetch and verify NVIDIA’s `v2.2.1` tag at `984180e5b8755aef85e9995048b5a16cb5336bce` to retain the same hatch-vcs version lineage. 2. Install a noneditable InferenceX wheel from the exact measured checkout into a shared Python 3.12 environment. Its installed package bytes are compared against the checkout before allocation. 3. Materialize separate client environments and retain their resolved package artifacts/locks. AgentX must come from `754356e9a39acc6cc6afb242d123bb57c3fb6f75`; lm-eval must come from `b315ef3b05176acc9732bb7fdec116abe1ecc476`. Editable and wrong-source installations are rejected. Preparation captures every installed distribution, not just the named entry point. -4. Materialize the complete model/tokenizer snapshot, exact `semianalysisai/cc-traces-weka-062126` snapshot and GSM8K cache. Capture their real revisions; do not invent or substitute a revision. The client’s offline model `refs/main` and snapshot files must be bound assets, and its resolved model snapshot must be the exact canonical serving snapshot. This preserves nominal tokenizer names without permitting a different cached revision. Prepare the unchanged serving image as a verified squash file and record its provenance/hash. +4. Materialize the complete model/tokenizer snapshot, exact `semianalysisai/cc-traces-weka-062126` snapshot and GSM8K cache. Capture their real revisions; do not invent or substitute a revision. The client’s offline model `refs/main` and snapshot files must be bound assets, and its resolved model snapshot must be the exact canonical serving snapshot. This preserves nominal tokenizer names without permitting a different cached revision. Private Hugging Face `refs/main` files contain exactly the revision bytes, with no trailing newline, as required by the actual cache reader. Prepare the unchanged serving image as a verified squash file and record its provenance/hash. 5. Write one `ClientSite` JSON for AgentX and one for eval. These explicitly provide the interpreter, distributions, offline cache environment, environment removals, asset roots/files, model snapshot, timeout and termination grace. `RuntimeSpec` rejects credentials; execution strips ambient credentials and unqualified AIPerf overrides. The packaged task and 1,319 independent document hashes are included in installed wheels. 6. Preserve the generated `PreparedSite` JSON with these two client-site paths, source/interpreter/model/image paths and mounts. For publication, extend it to `PilotSite` with actual deployed reader/collector revisions. The Pydantic models in [`render.py`](../infx/srt_slurm/render.py) and [`prepare.py`](../infx/benchmarks/prepare.py) are the exact schemas. +The downloaded trace snapshot alone does not satisfy AgentX's offline `datasets.load_dataset` call. Provision its nominal dataset repository at the explicit revision into the generation's private `HF_DATASETS_CACHE`, then verify a fresh offline nominal load against every row, in order, from the pinned snapshot. Include that prepared cache in the bound assets. A cache generated by loading a local directory has a different identity and cannot substitute for this check. The eval behavior probe also runs the same packaged lm-eval compatibility patch as the benchmark. + Preparation validates existing assets; it does not install packages, download models or repair incomplete snapshots on compute nodes. The derived mmap cache uses an owned namespace, file-integrity receipts, independent verified copies and corruption quarantine. Cold preparation on lock contention is explicit and bounded. Before enabling publication, require a verified active app reader deployment, migration `016_measurement_snapshots.sql` and a deployed trusted collector. Configure `INFX_H100_PHASE1_SITE_JSON`, `INFX_PHASE1_READER_REVISION` and `INFX_PHASE1_COLLECTOR_REVISION` in InferenceX. Configure `INFX_RECEIPT_ISSUER_SHAS` and `INFX_RECEIPT_ISSUER_WORKFLOW` in both repositories; the workflow is `.github/workflows/phase1-receipt.yml`. `INFX_PHASE1_READER_REVISION` was removed after the app rollback; the site, collector and issuer settings remain pending. The reader is currently unavailable for native receipt ingestion, and publication remains gated. Retained schema reports and code on a source branch do not establish current reader readiness. diff --git a/docs/srt-slurm-phase1_zh.md b/docs/srt-slurm-phase1_zh.md index 6847d5856b..ea472cd0d9 100644 --- a/docs/srt-slurm-phase1_zh.md +++ b/docs/srt-slurm-phase1_zh.md @@ -83,10 +83,12 @@ flowchart TD 1. 使用原生源代码提交中的 `uv.lock` 安装非 editable 环境:`uv sync --frozen --no-editable --no-dev --python 3.12`。保持该源码 checkout 干净。保留带哈希的 Linux wheel 及构建工具约束:`uv.lock` 固定运行依赖,但未固定上游 Hatch 构建依赖。重新构建时,获取并核实 NVIDIA 的 `v2.2.1` tag 指向 `984180e5b8755aef85e9995048b5a16cb5336bce`,保留相同 hatch-vcs 版本谱系。 2. 从实际测量 checkout 构建并安装非 editable InferenceX wheel,使用共享 Python 3.12 环境。分配前逐文件比较已安装包与 checkout。 3. 准备独立客户端环境并保留实际解析的包产物与锁。AgentX 必须来自 `754356e9a39acc6cc6afb242d123bb57c3fb6f75`;lm-eval 必须来自 `b315ef3b05176acc9732bb7fdec116abe1ecc476`。拒绝 editable 或错误来源。准备阶段记录所有已安装 distribution,而非仅入口包。 -4. 完整准备模型/tokenizer snapshot、准确的 `semianalysisai/cc-traces-weka-062126` snapshot 与 GSM8K 缓存。记录真实 revision,不编造或替换。准备原样服务镜像的已校验 squash 文件,记录来源和哈希。客户端离线模型的 `refs/main` 与 snapshot 文件必须纳入资源绑定,解析后的模型 snapshot 必须与服务端的规范路径完全一致,保留 tokenizer 的模型名称同时禁止解析到其他缓存版本。 +4. 完整准备模型/tokenizer snapshot、准确的 `semianalysisai/cc-traces-weka-062126` snapshot 与 GSM8K 缓存。记录真实 revision,不编造或替换。准备原样服务镜像的已校验 squash 文件,记录来源和哈希。客户端离线模型的 `refs/main` 与 snapshot 文件必须纳入资源绑定,解析后的模型 snapshot 必须与服务端的规范路径完全一致,保留 tokenizer 的模型名称同时禁止解析到其他缓存版本。私有 Hugging Face `refs/main` 文件只包含 revision 字节,不带尾部换行,符合实际缓存 reader 的要求。 5. 分别编写 AgentX、eval 的 `ClientSite` JSON:解释器、distribution、离线缓存环境、移除变量、资源根目录/文件、模型 snapshot、超时与终止宽限。`RuntimeSpec` 拒绝凭证;执行时移除继承凭证及未验收的 AIPerf 覆盖项。wheel 包含 eval task 与 1,319 个独立文档哈希。 6. 保留生成的 `PreparedSite` JSON,包含两个客户端配置路径、源码/解释器/模型/镜像路径及挂载。发布时再添加实际部署的 reader/collector revision,形成 `PilotSite`。准确 schema 见 [`render.py`](../infx/srt_slurm/render.py) 与 [`prepare.py`](../infx/benchmarks/prepare.py)。 +仅有下载好的 trace snapshot,无法满足 AgentX 离线调用 `datasets.load_dataset` 的要求。准备时须用数据集的标准仓库名称及明确 revision,在该 generation 的私有 `HF_DATASETS_CACHE` 中生成缓存;随后启动新的离线进程,通过标准仓库名称加载,并按顺序逐行与固定 snapshot 比较。生成的缓存也须纳入资源绑定。通过本地目录加载生成的缓存具有不同身份,不能替代上述检查。eval 行为探测也会执行与正式 benchmark 相同的、随包分发的 lm-eval 兼容补丁。 + 准备阶段校验已有资源,不在计算节点安装包、下载模型或修复不完整 snapshot。派生 mmap 缓存使用独立所属 namespace、文件完整性回执、独立校验副本及损坏隔离;锁竞争时的冷准备有明确界限。 启用发布前,必须确认 app reader 当前部署已验证、`016_measurement_snapshots.sql` migration 已完成且受信任 collector 已部署。InferenceX 需配置 `INFX_H100_PHASE1_SITE_JSON`、`INFX_PHASE1_READER_REVISION`、`INFX_PHASE1_COLLECTOR_REVISION`。两个仓库均需配置 `INFX_RECEIPT_ISSUER_SHAS`、`INFX_RECEIPT_ISSUER_WORKFLOW`,workflow 路径为 `.github/workflows/phase1-receipt.yml`。app 回滚后已删除 `INFX_PHASE1_READER_REVISION`;站点、collector 与 issuer 设置仍待完成。当前 reader 无法用于原生回执导入,发布仍受门禁限制。保留的 schema 报告和源分支中的代码都不能证明当前 reader 已就绪。 diff --git a/infx/srt_slurm/provision_runtime.py b/infx/srt_slurm/provision_runtime.py index 8dfbcf9b10..4e4277c478 100644 --- a/infx/srt_slurm/provision_runtime.py +++ b/infx/srt_slurm/provision_runtime.py @@ -164,7 +164,7 @@ def private_snapshot( (base / "refs").mkdir() reference = base / "refs/main" with reference.open("x") as stream: - stream.write(revision + "\n") + stream.write(revision) reference.chmod(0o444) return target @@ -174,7 +174,7 @@ def _project(path: Path, dependencies: list[str], python: str, *, cpu_torch: boo content = ( '[project]\nname = "infx-prepared-environment"\nversion = "0.0.0"\n' f'requires-python = "=={python}.*"\ndependencies = {json.dumps(dependencies)}\n' - "[tool.uv]\npackage = false\n" + '[tool.uv]\npackage = false\nexclude-newer = "PT12H"\n' ) if cpu_torch: content += ( @@ -320,7 +320,7 @@ def _wheels(commands: Commands, uv: str, source: Path, checkout: Path) -> tuple[ snapshot = pathlib.Path(snapshot_download("openai/gsm8k", repo_type="dataset", revision="main", cache_dir=str(hub))) if not re.fullmatch("[0-9a-f]{40}", snapshot.name): raise ValueError("GSM8K revision is not immutable") (base / "refs").mkdir(exist_ok=True) - (base / "refs/main").write_text(snapshot.name + "\\n") + (base / "refs/main").write_text(snapshot.name) revision = (base / "refs/main").read_text().strip() dataset = load_dataset("openai/gsm8k", "main", cache_dir=os.environ["HF_DATASETS_CACHE"], **({"revision": revision} if mode == "online" else {})) expected = json.loads(pathlib.Path(expected_path).read_text()) @@ -334,6 +334,86 @@ def _wheels(commands: Commands, uv: str, source: Path, checkout: Path) -> tuple[ """ +_TRACE_SCRIPT = """import hashlib, json, pathlib, sys +from datasets import load_dataset +mode, repository, revision, source, output = sys.argv[1:] +snapshot = pathlib.Path(source) +if snapshot.resolve() != snapshot or snapshot.name != revision: + raise ValueError("trace source is not the canonical pinned snapshot") +if mode not in ("online", "offline"): + raise ValueError("unknown trace preparation mode") +options = {"name": None, "split": "train", "trust_remote_code": False, "streaming": False} +dataset = load_dataset(repository, **options, **({"revision": revision} if mode == "online" else {})) +if len(dataset) == 0: + raise ValueError("trace dataset is empty") +result = {"repository": repository, "revision": revision, "offline": mode == "offline", "rows": len(dataset), "columns": dataset.column_names, "features": dataset.features.to_dict(), "cache_files": dataset.cache_files} +if mode == "offline": + expected = load_dataset(str(snapshot), **options) + if dataset.column_names != expected.column_names or not dataset.data.schema.equals(expected.data.schema, check_metadata=False): + raise ValueError("offline nominal trace schema differs from pinned snapshot") + if len(dataset) != len(expected): + raise ValueError("offline nominal trace row count differs from pinned snapshot") + digest = hashlib.sha256() + for index, (actual_row, expected_row) in enumerate(zip(dataset, expected, strict=True)): + actual = json.dumps(actual_row, sort_keys=True, ensure_ascii=False, separators=(",", ":")).encode() + original = json.dumps(expected_row, sort_keys=True, ensure_ascii=False, separators=(",", ":")).encode() + if actual != original: + raise ValueError(f"offline nominal trace row {index} differs from pinned snapshot") + digest.update(len(actual).to_bytes(8, "big")) + digest.update(actual) + result["ordered_rows_sha256"] = digest.hexdigest() +pathlib.Path(output).write_text(json.dumps(result, ensure_ascii=False) + "\\n") +""" + + +def materialize_trace( + commands: Commands, config: ProvisionConfig, python: Path, env: dict[str, str] +) -> None: + """Seed nominal Arrow identity without writing through the canonical snapshot view.""" + source = snapshot(config, dataset=True) + hub = commands.generation / "hf/trace-preparation-hub" + seeded = ( + hub + / ("datasets--" + config.dataset_repository.replace("/", "--")) + / "snapshots" + / config.dataset_revision + ) + seeded.mkdir(parents=True) + for original in source.rglob("*"): + target = seeded / original.relative_to(source) + if original.is_dir(): + target.mkdir(parents=True, exist_ok=True) + elif original.is_file(): + target.parent.mkdir(parents=True, exist_ok=True) + target.symlink_to(original.resolve(strict=True)) + else: + raise ValueError("trace snapshot contains an unsupported filesystem entry") + script = commands.generation / "materialize-trace.py" + script.write_text(_TRACE_SCRIPT) + for mode in ("online", "offline"): + commands.run( + f"trace-{mode}", + [ + str(python), + "-I", + str(script), + mode, + config.dataset_repository, + config.dataset_revision, + str(source), + str(commands.generation / "evidence" / f"trace-{mode}.json"), + ], + cwd=commands.generation, + timeout=1200 if mode == "online" else 600, + environment={ + **env, + "HF_HUB_CACHE": str(hub) if mode == "online" else env["HF_HUB_CACHE"], + "HF_HUB_OFFLINE": "0" if mode == "online" else "1", + "HF_DATASETS_OFFLINE": "0" if mode == "online" else "1", + }, + ) + + def _clients(commands: Commands, uv: str, build_constraints: Path) -> dict[str, Path]: result = {} for kind, (name, repository, revision, minor) in CLIENT_REPOSITORIES.items(): @@ -356,14 +436,21 @@ def _clients(commands: Commands, uv: str, build_constraints: Path) -> dict[str, return result -_CLIENT_PROBE_SCRIPT = """import importlib.metadata, json, pathlib, sys -kind, repository, snapshot, output = sys.argv[1:] +_CLIENT_PROBE_SCRIPT = """import importlib.metadata, json, pathlib, runpy, sys +kind, repository, snapshot, output, compatibility_patch = sys.argv[1:] if kind == "agentx": from huggingface_hub import snapshot_download - from aiperf.common.tokenizer import Tokenizer resolved = pathlib.Path(snapshot_download(repository, revision="main", local_files_only=True)).resolve() if resolved != pathlib.Path(snapshot).resolve(): raise ValueError("nominal tokenizer cache resolves a different serving snapshot") + configuration = {} + for filename in ("config.json", "tokenizer_config.json"): + path = resolved / filename + if path.is_file(): + raw = json.loads(path.read_text()) + configuration[filename] = {key: raw[key] for key in ("model_type", "tokenizer_class", "auto_map", "transformers_version") if key in raw} + pathlib.Path(output).with_suffix(".configuration.json").write_text(json.dumps({"repository": repository, "snapshot": str(resolved), "configuration": configuration}, ensure_ascii=False) + "\\n") + from aiperf.common.tokenizer import Tokenizer tokenizer = Tokenizer.from_pretrained(repository, trust_remote_code=True) samples = ["InferenceX tokenizer preparation.", "你好,世界。"] tokens = [tokenizer.encode(sample) for sample in samples] @@ -375,21 +462,23 @@ def _clients(commands: Commands, uv: str, build_constraints: Path) -> dict[str, lengths = tokenizer.encode_lengths_batch(samples) if lengths != [len(row) for row in tokens]: raise ValueError("prepared tokenizer batch lengths differ from individual encoding") - configuration = {} - for filename in ("config.json", "tokenizer_config.json"): - path = resolved / filename - if path.is_file(): - raw = json.loads(path.read_text()) - configuration[filename] = {key: raw[key] for key in ("model_type", "tokenizer_class", "auto_map", "transformers_version") if key in raw} result = {"repository": repository, "snapshot": str(resolved), "tokens": tokens, "decoded": decoded, "batch_lengths": lengths, "configuration": configuration} elif kind == "eval": + runpy.run_path(compatibility_patch) from lm_eval.models.openai_completions import LocalChatCompletion - model = LocalChatCompletion(model=repository, base_url="http://127.0.0.1:1/v1/chat/completions", tokenized_requests=False, max_length=16384, eos_string="") + model = LocalChatCompletion(model=repository, base_url="http://127.0.0.1:1/v1/chat/completions", api_key="EMPTY", eos_string="", max_retries=5, num_concurrent=28, timeout=1800, tokenized_requests=False, max_length=16384) messages = [{"role": "user", "content": "InferenceX client preparation."}] formatted = model.create_message([model.apply_chat_template(messages)]) if formatted != messages: raise ValueError("prepared eval backend changed chat messages") - result = {"backend": type(model).__name__, "messages": formatted, "tokenizer_backend": model.tokenizer_backend} + payload = model._create_payload(formatted, generate=True, gen_kwargs={"max_tokens": 12288, "temperature": 0, "top_p": 1, "until": ["", "<|im_end|>"], "do_sample": False}, eos="") + expected_payload = {"messages": messages, "model": repository, "max_tokens": 12288, "temperature": 0, "top_p": 1, "stop": ["", "<|im_end|>"], "seed": 1234} + if payload != expected_payload: + raise ValueError("prepared eval backend changed generation payload") + parsed = model.parse_generations({"choices": [{"index": 0, "message": {"content": "final answer"}}, {"index": 1, "message": {"content": "", "reasoning_content": "reasoning answer"}}]}) + if parsed != ["final answer", "reasoning answer"]: + raise ValueError("prepared eval backend changed response parsing") + result = {"backend": type(model).__name__, "messages": formatted, "tokenizer_backend": model.tokenizer_backend, "payload": payload, "parsed_generations": parsed} else: raise ValueError("unknown prepared client") result["transformers_version"] = importlib.metadata.version("transformers") @@ -403,6 +492,8 @@ def verify_clients( """Exercise the installed tokenizer and API chat backend offline before hashing assets.""" script = commands.generation / "verify-client-behavior.py" script.write_text(_CLIENT_PROBE_SCRIPT) + patch = commands.generation / "evidence/lm_eval_sitecustomize.py" + patch.write_bytes(files("infx.evals.patches").joinpath("lm_eval_sitecustomize.py").read_bytes()) for kind, python in clients.items(): commands.run( f"offline-{kind}-behavior", @@ -414,6 +505,7 @@ def verify_clients( config.model_repository, str(snapshot(config, dataset=False)), str(commands.generation / "evidence" / f"{kind}-behavior.json"), + str(patch), ], cwd=commands.generation, timeout=600, @@ -649,6 +741,7 @@ def provision( env = _offline_env(generation) sites = _sites(config, clients, env) verify_clients(commands, config, clients, env) + materialize_trace(commands, config, clients["agentx"], env) script = generation / "materialize-gsm8k.py" script.write_text(_GSM_SCRIPT) expected = generation / "evidence/gsm8k-test-doc-hashes.json" diff --git a/utils/test_provision_runtime.py b/utils/test_provision_runtime.py index a1d8ade12b..8d208964a5 100644 --- a/utils/test_provision_runtime.py +++ b/utils/test_provision_runtime.py @@ -9,6 +9,7 @@ from pathlib import Path import pytest +from huggingface_hub import hf_hub_download, snapshot_download from infx.benchmarks.common import ( read_json, @@ -19,12 +20,13 @@ from infx.benchmarks.spec import RuntimeSpec from infx.srt_slurm.provision import ProvisionConfig, snapshot from infx.srt_slurm.provision_runtime import ( + _GSM_SCRIPT, Commands, ProvisionStepError, - _GSM_SCRIPT, _offline_env, _sites, installer_environment, + materialize_trace, owned_generation, publish_evidence, require_clean_checkout, @@ -90,6 +92,27 @@ def test_private_views_bind_original_snapshot_without_mutating_legacy_refs(asset terminate_grace_seconds=1, ) original_model = snapshot(assets, dataset=False) + resolved_model = Path( + snapshot_download( + assets.model_repository, + revision="main", + cache_dir=env["HF_HUB_CACHE"], + local_files_only=True, + ) + ).resolve() + assert resolved_model == original_model + assert (resolved_model / "one.safetensors").read_bytes() == b"weight" + resolved_trace = Path( + snapshot_download( + assets.dataset_repository, + repo_type="dataset", + revision="main", + cache_dir=env["HF_HUB_CACHE"], + local_files_only=True, + ) + ).resolve() + assert resolved_trace == snapshot(assets, dataset=True) + assert (resolved_trace / "train.parquet").read_bytes() == b"trace" assert ( verify_model_snapshot_assets( runtime, @@ -125,30 +148,35 @@ def test_private_views_bind_original_snapshot_without_mutating_legacy_refs(asset def test_owned_generation_rejects_overlap_and_reuse_and_retains_failure(tmp_path): root = tmp_path.resolve() / "prepared" - with pytest.raises(RuntimeError, match="controlled failure"): - with owned_generation(root, "attempt-1") as generation: - (generation / "evidence/progress.txt").write_text("completed stage") - with pytest.raises(ValueError, match="another provisioning"): - with owned_generation(root, "attempt-2"): - pytest.fail("lock was bypassed") - raise RuntimeError("controlled failure") + with ( + pytest.raises(RuntimeError, match="controlled failure"), + owned_generation(root, "attempt-1") as generation, + ): + (generation / "evidence/progress.txt").write_text("completed stage") + with ( + pytest.raises(ValueError, match="another provisioning"), + owned_generation(root, "attempt-2"), + ): + pytest.fail("lock was bypassed") + raise RuntimeError("controlled failure") assert read_json(generation / "state.json") == { "state": "failed", "error_type": "RuntimeError", "qualification_complete": False, } assert (generation / "evidence/progress.txt").read_text() == "completed stage" - with pytest.raises(FileExistsError): - with owned_generation(root, "attempt-1"): - pytest.fail("failed generation was silently reused") + with pytest.raises(FileExistsError), owned_generation(root, "attempt-1"): + pytest.fail("failed generation was silently reused") assert not (root / "generations/attempt-2").exists() @pytest.mark.parametrize("namespace", ["../escape", "", "/absolute"]) def test_namespace_cannot_escape_owned_root(tmp_path, namespace): - with pytest.raises(ValueError, match="namespace"): - with owned_generation(tmp_path.resolve() / "prepared", namespace): - pytest.fail("unsafe namespace accepted") + with ( + pytest.raises(ValueError, match="namespace"), + owned_generation(tmp_path.resolve() / "prepared", namespace), + ): + pytest.fail("unsafe namespace accepted") def test_child_gets_no_ambient_credentials_and_failure_logs_are_publishable(tmp_path): @@ -279,7 +307,7 @@ def test_installed_client_probes_retain_behavior_and_reject_invalid_tokenization "def snapshot_download(repository, *, revision, local_files_only):\n" " assert revision == 'main' and local_files_only\n" " root=pathlib.Path(os.environ['HF_HUB_CACHE'])/('models--'+repository.replace('/','--'))\n" - " return str(root/'snapshots'/(root/'refs/main').read_text().strip())\n" + " return str(root/'snapshots'/(root/'refs/main').read_text())\n" ) (purelib / "aiperf/common/tokenizer.py").write_text( "import os\n" @@ -288,19 +316,35 @@ def test_installed_client_probes_retain_behavior_and_reject_invalid_tokenization " def from_pretrained(cls, repository, *, trust_remote_code):\n" " assert repository == 'fixture/model' and trust_remote_code\n" " assert os.environ['HF_HUB_OFFLINE'] == '1'\n" + " if 'FIXTURE_INVALID_TOKENIZER' in os.environ:\n" + " raise ValueError('controlled tokenizer construction failure')\n" " return cls()\n" " def encode(self, text): return [17,19]\n" " def decode(self, tokens): return 'decoded text'\n" " def encode_lengths_batch(self, texts):\n" " return [2,2] if 'FIXTURE_INVALID_LENGTHS' not in os.environ else [1,2]\n" ) + (purelib / "lm_eval/models/api_models.py").write_text( + "class JsonChatStr:\n" + " def __init__(self, prompt): self.prompt=prompt\n" + "class TemplateAPI:\n" + " def apply_chat_template(self, messages):\n" + " raise ValueError('packaged compatibility patch was not applied')\n" + ) (purelib / "lm_eval/models/openai_completions.py").write_text( - "class LocalChatCompletion:\n" + "import json\n" + "from lm_eval.models.api_models import TemplateAPI\n" + "class LocalChatCompletion(TemplateAPI):\n" " def __init__(self, **kwargs):\n" " assert kwargs['tokenized_requests'] is False\n" " self.tokenizer_backend=None\n" - " def apply_chat_template(self, messages): return messages\n" - " def create_message(self, batch): return batch[0]\n" + " self.model=kwargs['model']\n" + " def create_message(self, batch): return json.loads(batch[0].prompt)\n" + " def _create_payload(self, messages, *, generate, gen_kwargs, eos):\n" + " assert generate and eos == ''\n" + " assert gen_kwargs.pop('do_sample') is False\n" + " return dict(messages=messages, model=self.model, seed=1234,\n" + " stop=gen_kwargs.pop('until'), **gen_kwargs)\n" ) with owned_generation(Path(assets.shared_root), "client-probes") as generation: env = _offline_env(generation) @@ -318,6 +362,16 @@ def test_installed_client_probes_retain_behavior_and_reject_invalid_tokenization {"role": "user", "content": "InferenceX client preparation."} ] assert evaluation["tokenizer_backend"] is None + assert evaluation["payload"] == { + "messages": [{"role": "user", "content": "InferenceX client preparation."}], + "model": "fixture/model", + "max_tokens": 12288, + "temperature": 0, + "top_p": 1, + "stop": ["", "<|im_end|>"], + "seed": 1234, + } + assert evaluation["parsed_generations"] == ["final answer", "reasoning answer"] with pytest.raises(ProvisionStepError, match="offline-agentx-behavior"): verify_clients( commands, assets, clients, {**env, "FIXTURE_INVALID_LENGTHS": "1"} @@ -326,6 +380,21 @@ def test_installed_client_probes_retain_behavior_and_reject_invalid_tokenization "batch lengths differ" in (generation / "logs/03-offline-agentx-behavior.log").read_text() ) + configuration_path = generation / "evidence/agentx-behavior.configuration.json" + configuration_path.unlink() + with pytest.raises(ProvisionStepError, match="offline-agentx-behavior"): + verify_clients( + commands, assets, clients, {**env, "FIXTURE_INVALID_TOKENIZER": "1"} + ) + assert read_json(configuration_path) == { + "repository": "fixture/model", + "snapshot": str(snapshot(assets, dataset=False)), + "configuration": {"config.json": {"model_type": "fixture"}}, + } + assert ( + "controlled tokenizer construction failure" + in (generation / "logs/04-offline-agentx-behavior.log").read_text() + ) def test_gsm_materialization_pins_online_revision_then_checks_nominal_offline_lookup( @@ -392,6 +461,15 @@ def test_gsm_materialization_pins_online_revision_then_checks_nominal_offline_lo "train_documents": 5, "offline": True, } + assert Path( + snapshot_download( + "openai/gsm8k", + repo_type="dataset", + revision="main", + cache_dir=tmp_path / "hub", + local_files_only=True, + ) + ) == tmp_path / "hub/datasets--openai--gsm8k/snapshots" / ("e" * 40) data = read_json(payload) data["test"][0]["answer"] = "changed document" payload.write_text(json.dumps(data)) @@ -409,3 +487,106 @@ def test_gsm_materialization_pins_online_revision_then_checks_nominal_offline_lo ) assert failed.returncode != 0 assert "GSM8K document differs" in failed.stderr + + +def test_trace_materialization_proves_nominal_rows_and_keeps_online_writes_private( + assets, isolated_python +): + python, purelib = isolated_python + source = snapshot(assets, dataset=True) + payload = { + "schema": {"id": "string", "turns": "int64"}, + "rows": [{"id": "first", "turns": 2}, {"id": "second", "turns": 3}], + } + (source / "fixture.json").write_text(json.dumps(payload)) + (purelib / "datasets.py").write_text( + "import json, os, pathlib, types\n" + "class Features(dict):\n" + " def to_dict(self): return dict(self)\n" + "class Schema:\n" + " def __init__(self, fields): self.fields=list(fields.items())\n" + " def equals(self, other, *, check_metadata): return self.fields == other.fields\n" + "class Dataset:\n" + " def __init__(self, payload, cache):\n" + " self.rows=payload['rows']\n" + " self.column_names=list(payload['schema'])\n" + " self.features=Features(payload['schema'])\n" + " self.data=types.SimpleNamespace(schema=Schema(payload['schema']))\n" + " self.cache_files=[{'filename':str(cache)}]\n" + " def __len__(self): return len(self.rows)\n" + " def __iter__(self): return iter(self.rows)\n" + "def load_dataset(target, *, name, split, trust_remote_code, streaming, **kwargs):\n" + " assert name is None and split == 'train' and not trust_remote_code and not streaming\n" + " cache=pathlib.Path(os.environ['HF_DATASETS_CACHE'])/'trace.json'\n" + " if target.startswith('/'):\n" + " data=json.loads((pathlib.Path(target)/'fixture.json').read_text())\n" + " elif os.environ['HF_HUB_OFFLINE'] == '0':\n" + " assert kwargs['revision'] == 'b'*40\n" + " seeded=pathlib.Path(os.environ['HF_HUB_CACHE'])/('datasets--'+target.replace('/','--'))/'snapshots'/kwargs['revision']\n" + " (seeded/'new-metadata.json').write_text('{}')\n" + " data=json.loads((seeded/'fixture.json').read_text())\n" + " cache.write_text(json.dumps(data))\n" + " else:\n" + " assert not kwargs\n" + " data=json.loads(cache.read_text())\n" + " return Dataset(data, cache)\n" + ) + with owned_generation(Path(assets.shared_root), "trace-cache") as generation: + env = _offline_env(generation) + _sites(assets, {"agentx": python, "eval": python}, env) + commands = Commands(generation, installer_environment(generation, os.environ)) + materialize_trace(commands, assets, python, env) + receipt = read_json(generation / "evidence/trace-offline.json") + assert receipt["rows"] == 2 + assert receipt["columns"] == ["id", "turns"] + assert receipt["features"] == {"id": "string", "turns": "int64"} + assert receipt["offline"] is True + assert not (source / "new-metadata.json").exists() + assert (source.parent.parent / "refs/main").read_text() == "c" * 40 + cached_file = Path( + hf_hub_download( + assets.dataset_repository, + "fixture.json", + repo_type="dataset", + revision="b" * 40, + cache_dir=generation / "hf/trace-preparation-hub", + local_files_only=True, + ) + ) + assert json.loads(cached_file.read_text()) == payload + assert cached_file.is_symlink() + assert not cached_file.parent.is_symlink() + cache = Path(env["HF_DATASETS_CACHE"]) / "trace.json" + script = generation / "materialize-trace.py" + for changed, message in ( + ({**payload, "rows": list(reversed(payload["rows"]))}, "row 0 differs"), + ({**payload, "rows": payload["rows"][:1]}, "row count differs"), + ( + {**payload, "schema": {"id": "string", "turns": "float64"}}, + "schema differs", + ), + ): + cache.write_text(json.dumps(changed)) + with pytest.raises(ProvisionStepError, match="trace-recheck"): + commands.run( + "trace-recheck", + [ + str(python), + "-I", + str(script), + "offline", + assets.dataset_repository, + assets.dataset_revision, + str(source), + str(generation / "recheck.json"), + ], + cwd=generation, + timeout=10, + environment=env, + ) + assert ( + message + in ( + generation / "logs" / f"{commands.number:02d}-trace-recheck.log" + ).read_text() + ) From bc0710934f3ca2dbdd37940aa93c91a5002e5061 Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sat, 19 Sep 2026 22:15:28 -0400 Subject: [PATCH 12/16] fix: use native-safe cancellation qualification intents MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 使用符合原生约束的取消验证 intent,保留错误证据并清晰报告原生命令失败,覆盖实际固定运行时边界。 --- infx/srt_slurm/qualify_cancellation.py | 6 +- utils/test_qualify_cancellation.py | 85 ++++++++++++++++++++++++++ 2 files changed, 90 insertions(+), 1 deletion(-) diff --git a/infx/srt_slurm/qualify_cancellation.py b/infx/srt_slurm/qualify_cancellation.py index 187dd55efa..822b6f8042 100644 --- a/infx/srt_slurm/qualify_cancellation.py +++ b/infx/srt_slurm/qualify_cancellation.py @@ -144,6 +144,8 @@ def invoke(self, name: str, argv: list[str], *, timeout: int) -> dict[str, Any]: write_json(path, {"state": "interrupted", "error_type": type(error).__name__}) raise write_json(path, result) + if result.get("state") == "error": + raise NativeCommandError(result, f"native {name} failed: {result}") return result @@ -535,7 +537,7 @@ def interrupted(signum: int, _frame: Any) -> None: ) != str(directory / "native-output"): raise ValueError("native prepared paths escaped the owned qualification generation") report["prepared"] = prepared - intent = "cancellation-qualification:" + namespace + intent = "cancellation-qualification-" + namespace location = native.run( "intent-path", "--intent", @@ -545,6 +547,8 @@ def interrupted(signum: int, _frame: Any) -> None: "--journal-dir", str(directory / "journal"), ) + if location.get("state") != "intent" or not isinstance(location.get("receipt_path"), str): + raise ValueError(f"native intent-path did not return an ownership path: {location}") receipt_path = Path(location["receipt_path"]) if receipt_path.resolve() != receipt_path or not receipt_path.is_relative_to( directory / "journal" diff --git a/utils/test_qualify_cancellation.py b/utils/test_qualify_cancellation.py index 164ed80bd2..277b9f69ca 100644 --- a/utils/test_qualify_cancellation.py +++ b/utils/test_qualify_cancellation.py @@ -81,6 +81,16 @@ def load_receipt(): return json.loads(receipt_path().read_text()) output({"state": "prepared", "prepared_dir": state["prepared_dir"], "resources": resources, "output_root": state["output_root"], "manifest_sha256": "a" * 64}) if command == "intent-path": + if control["scenario"] == "intent-error": + output({"schema": 1, "state": "error", "error": "native rejected the intent"}, 2) + if control["scenario"] == "missing-intent-path": + output({"schema": 1, "state": "intent"}) + if control.get("native_intent_python"): + result = subprocess.run([control["native_intent_python"], *args], + capture_output=True, text=True, timeout=15) + payload = json.loads(result.stdout) + if payload.get("receipt_path"): state["receipt"] = payload["receipt_path"] + output(payload, result.returncode) path = Path(value("--journal-dir")) / "owned-intent/receipt.json" if control["scenario"] == "escaping-intent": path = control_path.parent / "foreign-receipt.json" state["receipt"] = str(path) @@ -377,6 +387,81 @@ def test_native_preflight_rejects_wrong_demand_or_unowned_journal_before_submit( ) +@pytest.mark.parametrize( + "scenario, error_type, message, payload", + [ + ( + "intent-error", + "NativeCommandError", + "native rejected the intent", + {"schema": 1, "state": "error", "error": "native rejected the intent"}, + ), + ( + "missing-intent-path", + "ValueError", + "did not return an ownership path", + {"schema": 1, "state": "intent"}, + ), + ], +) +def test_native_intent_failure_retains_original_payload_without_submission( + pilot, scenario, error_type, message, payload +): + root, _, _ = pilot + with pytest.raises(RuntimeError, match="inspect"): + run_probe(pilot, scenario=scenario) + report = json.loads((root / "artifacts/qualification.json").read_text()) + assert report["error_type"] == error_type + assert message in report["error"] + assert report["lifecycle_qualified"] is False + assert ( + json.loads( + (root / "artifacts/evidence/commands/0003-intent-path.json").read_text() + ) + == payload + ) + state = json.loads((root / "state.json").read_text()) + assert "submit-prepared" not in state["calls"] + assert "cancel-known" not in state["calls"] + + +def test_installed_native_intent_accepts_maximum_cancellation_namespace(pilot): + native_python = os.environ.get("INFX_NATIVE_PHASE1_PYTHON") + if not native_python: + pytest.skip( + "set INFX_NATIVE_PHASE1_PYTHON to exercise installed native intent-path" + ) + root, _, draft = pilot + control = json.loads((root / "control.json").read_text()) + store(root / "control.json", {**control, "native_intent_python": native_python}) + namespace = "n" * 96 + report = run_probe(pilot, namespace=namespace) + assert report["state"] == "passed" + assert report["cleanup"]["closure"]["terminal"] is True + location = json.loads( + (root / "artifacts/evidence/commands/0003-intent-path.json").read_text() + ) + assert location["state"] == "intent" + receipt = Path(location["receipt_path"]) + assert receipt == Path(report["receipt_path"]) + journal = ( + Path(draft["shared_root"]) + / "cancellation-qualification" + / namespace + / "journal" + ) + assert receipt.is_relative_to(journal / draft["cluster"]) + assert json.loads(receipt.read_text())["accepted_ids"] == ["71"] + + +def test_oversized_namespace_fails_before_native_preparation(pilot): + root, _, _ = pilot + with pytest.raises(ValueError, match="safe path component"): + run_probe(pilot, namespace="n" * 97) + assert not (root / "state.json").exists() + assert not (root / "artifacts").exists() + + def test_unqualified_pin_and_deployment_pins_fail_before_native_submission(pilot): root, _, draft = pilot with pytest.raises(ValueError, match="Extra inputs"): From 366842b6a89eb3f61ed2122060d276838511e0fe Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sat, 19 Sep 2026 22:29:44 -0400 Subject: [PATCH 13/16] fix: match cancellation probes to the real c28 eval server MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 复用实际测量点的服务参数渲染逻辑,固定 c28 eval 的序列与图捕获上限,并记录真实失败及清理证据;不放宽生命周期验收要求。 --- docs/srt-slurm-phase1.md | 10 ++- docs/srt-slurm-phase1_zh.md | 10 ++- infx/srt_slurm/qualify_cancellation.py | 5 +- infx/srt_slurm/render.py | 38 +++++++---- utils/test_qualify_cancellation.py | 88 +++++++++++++++++++++++++- 5 files changed, 136 insertions(+), 15 deletions(-) diff --git a/docs/srt-slurm-phase1.md b/docs/srt-slurm-phase1.md index ee196b8525..788719bb15 100644 --- a/docs/srt-slurm-phase1.md +++ b/docs/srt-slurm-phase1.md @@ -118,6 +118,10 @@ Successful publication requires both native terminal success and a closed client Before accepting a sweep, qualify real cancellation through the E2E `cancel-startup` and `cancel-client` site operations. Supply the actual provisioned `phase1-site-draft` path. They use separate diagnostic output/journals, the pinned image and TP8 worker, and native ownership-aware cancellation; they do not fabricate deployment pins or produce accepted benchmark manifests. Startup uses a 300-second allocation with a 240-second observation budget. Client interruption uses a 3,600-second allocation with a 3,300-second observation budget so model readiness can complete. Each has a separate 180-second cleanup budget. Preserve the resulting `qualification.json` and native terminal/writer evidence; a cancellation RPC or `COMPLETING` alone is insufficient. +Both probes use the real c28 eval serving settings through the same `apply_serving_point` renderer as benchmark execution: `max-num-seqs: 56`, `max-cudagraph-capture-size: 512`, and DSpark with real block rejection and adaptive verification. The model, image, TP8 topology, `max-model-len: 1048576` and `max-num-batched-tokens: 4096` remain unchanged. The diagnostic client is still a literal Python signal-aware writer. `qualification.json` records this serving point; the probe does not run or publish an eval. Matching these settings corrects the previous reliance on vLLM's sequence defaults, but does not prove the observed CUDA initialization failure is fixed. + +The [H100 observer run 35483966784](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35483966784) confirmed Slurm `25.05.7`, `StepMgrEnabled=Yes` for owned job `18325`, and `enable_stepmgr` in the controller configuration. During the preceding startup probe, `squeue --steps` exposed only `batch` and `extern` despite a running aggregate step. Native step discovery is being updated to query `scontrol --oneliner show steps ` and validate the returned job, step name and running state. This is a native observation/cleanup correction; it does not relax the requirement for an actual aggregate step plus worker identity. The observer ran after job `18325` ended, so its empty step listing is not live-worker verification. The updated native runtime still requires an actual rerun. + The native pin now requires `prepared-direct-listener-ownership-v1`. Before client traffic, it verifies the listening socket belongs to the recorded worker process tree, using PID start times and PID/network namespace identities. It repeats ownership checks during the client and before accepting exit 0. A foreign/replaced listener, reused PID or inaccessible ownership evidence fails the job and closes the client. Actual Pyxis namespace/proc visibility is part of cluster qualification. The complete source contract is eight throughput points and one real c28 eval. GSM8K requires all 1,319 documents and both filters (2,638 scored rows), the preserved 16,384 context / 12,288 generation budgets, finite scores and complete sample identities. Aggregate eval metadata has `disagg:false`, `is_multinode:false`, eight serving GPUs and zero prefill/decode worker counts. @@ -138,13 +142,17 @@ Current readiness: at the user’s request, Vercel Instant Rollback restored pro [H100 inventory run 35477700047](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35477700047) passed on an actual login runner at `14d56f1bbf8f3c867ea79ae97a2f716f304aaaa2`. Both pinned snapshots and all indexed model shards exist at the configured canonical shared paths; the serving squash file is 21,390,860,288 bytes, and the four required Slurm commands are present. [CI 35477693002](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35477693002) passed the updated topology and inventory behavior plus the native Linux contract. Neither run submitted a GPU benchmark. +The first allocated lifecycle probes used InferenceX `bc0710934f3ca2dbdd37940aa93c91a5002e5061` and native `50c3dacc37def01606ee9e4e0ed873646d4f7cc5`. [Startup run 35483589637](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35483589637) accepted job `18324`; its TP8 worker ran as step `18324.5`. The probe timed out because step discovery missed that worker, then cancelled the exact owned allocation. Retained logs show SIGTERM, and native observation confirms terminal `CANCELLED` with cleanup complete. This proves owned cancellation and closure, but the requested startup trigger did not qualify. + +[Client run 35483590787](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35483590787) accepted job `18325`, whose vLLM worker failed during DSpark initialization with CUDA index-out-of-bounds/device-assert errors. Native observation confirms terminal `FAILED` and completed cleanup; cancellation found the allocation already terminal. No diagnostic writer started, so there is no intended client-interruption or closed-writer evidence. Both runs retain `lifecycle_qualified: false`. The c28 serving-parity and native step-discovery fixes need fresh hardware verification; neither failed run qualifies a lifecycle gate or establishes a sweep outcome. + | Gate | Status / required evidence | | --- | --- | | Native and client behavior | CPU tests and installed-wheel checks; no GPU claim | | Receipt, app and recovery | Reader rolled back and revision variable removed; additive schema retained; native receipt import and publication gated | | H100 throughput | c1,2,4,8,16,20,24,28 pending on the unchanged image | | Real evaluation | New c28 run pending; historical full raw eval passes validator | -| Cancellation/cleanup | Local ownership/race/closure tests; real Slurm signal qualification pending | +| Cancellation/cleanup | Job `18324` reached owned `CANCELLED` closure but missed its trigger; job `18325` failed initialization before the writer; both lifecycle qualifications remain pending | | Measurement equivalence | Compare qualified metrics, failures, warmup/drain, server settings and raw schemas against the retained baseline | | Publication | Trusted receipt, later publication record and refreshed app evidence pending | | Power | Explicit temporary parity exception; measured power not claimed | diff --git a/docs/srt-slurm-phase1_zh.md b/docs/srt-slurm-phase1_zh.md index ea472cd0d9..7f89e29cd4 100644 --- a/docs/srt-slurm-phase1_zh.md +++ b/docs/srt-slurm-phase1_zh.md @@ -118,6 +118,10 @@ Slurm 分配、claim、已接受 ID、调度器观察与取消均由原生运行 接受 sweep 前,先通过 E2E 的 `cancel-startup` 与 `cancel-client` 站点操作验证真实取消行为,并显式提供已准备的 `phase1-site-draft` 路径。它们使用独立诊断输出和 journal、固定镜像与 TP8 worker,以及原生归属校验和取消接口,不编造部署 pin,也不生成可接受的 benchmark manifest。启动期探针申请 300 秒 allocation,观察预算为 240 秒;客户端中断探针申请 3,600 秒 allocation,观察预算为 3,300 秒,以便模型完成就绪。两种模式的清理预算均为额外 180 秒。保留 `qualification.json` 及原生终态/writer 证据;仅有取消 RPC 或 `COMPLETING` 不足以通过验收。 +两种探针通过与 benchmark 执行共用的 `apply_serving_point` 渲染逻辑,采用真实 c28 eval 的服务设置:`max-num-seqs: 56`、`max-cudagraph-capture-size: 512`,以及真实 block rejection 和 adaptive verification 的 DSpark。模型、镜像、TP8 拓扑、`max-model-len: 1048576` 与 `max-num-batched-tokens: 4096` 均保持不变。诊断客户端仍是以字面代码执行、可处理信号的 Python writer。`qualification.json` 记录该服务点;探针不运行或发布 eval。此设置对齐修正了此前依赖 vLLM 默认序列上限的问题,但不能证明已观察到的 CUDA 初始化失败已经修复。 + +[H100 观察运行 35483966784](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35483966784) 确认 Slurm 版本为 `25.05.7`,所属任务 `18325` 的 `StepMgrEnabled=Yes`,且 controller 配置含 `enable_stepmgr`。此前的启动期探针中,实际聚合 step 正在运行,但 `squeue --steps` 仅返回 `batch` 和 `extern`。原生 step 发现正在改为查询 `scontrol --oneliner show steps `,并校验返回的任务、step 名及运行状态。这属于原生观察/清理修正,不放宽“实际聚合 step 加 worker 身份”的要求。观察运行发生在任务 `18325` 结束后,因此当时的空 step 列表不能证明活跃 worker 的发现行为。更新后的原生运行时仍须实际重跑验证。 + 当前原生 pin 要求 `prepared-direct-listener-ownership-v1`。发送客户端流量前,它根据 PID 启动时间及 PID/network namespace 身份,验证监听套接字属于记录的 worker 进程树;客户端运行期间和接受退出码 0 之前也会重新验证。外部或替换监听器、PID 复用或无法读取的归属证据都会使任务失败并关闭客户端。实际 Pyxis 的 namespace/proc 可见性仍需集群验收。 完整源契约为八个吞吐点加一个真实 c28 eval。GSM8K 必须包含全部 1,319 文档及两种 filter(2,638 个评分行),保留 16,384 上下文 / 12,288 生成预算,验证有限分数和完整样本身份。聚合 eval 元数据为 `disagg:false`、`is_multinode:false`、八个服务 GPU、prefill/decode worker 数均为零。 @@ -138,13 +142,17 @@ Merge helper 保留最近明确授权的 `/use RUN_ID` 或 `/reuse-sweep-run RUN [H100 资源检查运行 35477700047](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35477700047) 在实际登录 runner 上通过,提交为 `14d56f1bbf8f3c867ea79ae97a2f716f304aaaa2`。固定的两个快照及全部索引内模型分片均位于配置的规范共享路径,服务 squash 文件大小为 21,390,860,288 字节,四个必需的 Slurm 命令均可用。[CI 35477693002](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35477693002) 通过了更新后的拓扑、资源检查行为和原生 Linux 契约验证。这两次运行均未提交 GPU benchmark。 +首次实际获得资源的生命周期探针使用 InferenceX `bc0710934f3ca2dbdd37940aa93c91a5002e5061` 和原生 `50c3dacc37def01606ee9e4e0ed873646d4f7cc5`。[启动期运行 35483589637](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35483589637) 接受任务 `18324`,其 TP8 worker 以 step `18324.5` 运行。探针因 step 发现未识别该 worker 而超时,随后准确取消所属 allocation。保留日志显示 SIGTERM,原生观察确认终态 `CANCELLED` 且清理完成。这证明了所属资源的取消和关闭,但指定的启动期触发条件未通过验收。 + +[客户端运行 35483590787](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35483590787) 接受任务 `18325`;其 vLLM worker 在 DSpark 初始化期间出现 CUDA 索引越界/device-assert 错误并失败。原生观察确认终态 `FAILED` 且清理完成;取消操作发现 allocation 已经终止。诊断 writer 未启动,因此没有预期客户端中断或 writer 关闭证据。两次运行均保留 `lifecycle_qualified: false`。c28 服务参数对齐和原生 step 发现修正需要新的硬件验证;这两次失败运行均不能关闭生命周期验收门禁,也不能证明 sweep 的最终结果。 + | Gate | 状态 / 所需证据 | | --- | --- | | 原生及客户端行为 | CPU 测试、已安装 wheel 检查;不声称 GPU 验收 | | 回执、app、恢复 | reader 已回滚,revision 变量已删除;新增 schema 保留;原生回执导入与发布仍受门禁限制 | | H100 吞吐 | 原镜像 c1、2、4、8、16、20、24、28 待运行 | | 真实 eval | 新 c28 待运行;历史完整原始 eval 通过新 validator | -| 取消与清理 | 本地所属关系/race/closure 测试;实际 Slurm 信号验收待做 | +| 取消与清理 | 任务 `18324` 已确认所属资源以 `CANCELLED` 关闭,但未满足触发条件;任务 `18325` 在 writer 启动前初始化失败;两项生命周期验收仍待完成 | | 测量等价性 | 对照保留基线比较指标、失败、warmup/drain、服务设置及原始 schema | | 发布 | 受信任回执、后续发布记录及刷新后 app 证据待完成 | | 功耗 | 明确临时一致性例外;不声称实测功耗 | diff --git a/infx/srt_slurm/qualify_cancellation.py b/infx/srt_slurm/qualify_cancellation.py index 822b6f8042..cdff445799 100644 --- a/infx/srt_slurm/qualify_cancellation.py +++ b/infx/srt_slurm/qualify_cancellation.py @@ -20,7 +20,7 @@ from infx.srt_slurm.job import file_digest, read_json from infx.srt_slurm.launch import NativeCommandError, RuntimeLock, checked_json from infx.srt_slurm.provision_runtime import NATIVE_LOCK -from infx.srt_slurm.render import PreparedSite +from infx.srt_slurm.render import PreparedSite, apply_serving_point RECIPE = ( "benchmarks/multi_node/srt-slurm-recipes/dsv41flash/vllm/h100-fp4/agentx/agg-tp8-dspark5.yaml" @@ -197,6 +197,8 @@ def render_probe( or recipe["model"]["container"] != site.image_reference ): raise ValueError("qualification requires the selected exclusive H100 aggregate TP8 recipe") + # Exercise the actual c28 eval server before interrupting the harmless writer. + apply_serving_point(role["args"], concurrency=28, evaluation=True, root=root, policy=None) hours, remainder = divmod(walltime_seconds, 3600) minutes, seconds = divmod(remainder, 60) walltime = f"{hours:02d}:{minutes:02d}:{seconds:02d}" @@ -492,6 +494,7 @@ def qualify( "lifecycle_qualified": False, "directory": str(directory), "native_revision": lock.revision, + "serving_point": {"mode": "eval", "concurrency": 28}, "walltime_seconds": walltime_seconds, "observation_timeout_seconds": observation_timeout_seconds, "cleanup_timeout_seconds": cleanup_timeout_seconds, diff --git a/infx/srt_slurm/render.py b/infx/srt_slurm/render.py index 1801d2bda5..8d7f0fd320 100644 --- a/infx/srt_slurm/render.py +++ b/infx/srt_slurm/render.py @@ -194,6 +194,30 @@ def client_spec( ) +def apply_serving_point( + args: dict[str, Any], + *, + concurrency: int, + evaluation: bool, + root: Path, + policy: ClientPolicy | None, +) -> None: + """Apply the benchmark point's serving limits and DSpark verification mode.""" + args["max-num-seqs"] = 2 * concurrency + args["max-cudagraph-capture-size"] = min(2048, 1 << (12 * concurrency - 1).bit_length()) + spec = args["speculative-config"] + if spec.get("synthetic_acceptance_length") is not None: + raise ValueError("recipe must not hard-code a synthetic acceptance value") + spec["enable_adaptive_verification"] = evaluation + spec["rejection_sample_method"] = "block" if evaluation else "synthetic" + if not evaluation: + if policy is None: + raise ValueError("throughput serving requires the selected client policy") + spec["synthetic_acceptance_length"] = golden_acceptance( + root, policy, spec["num_speculative_tokens"] + ) + + def render_recipe( job: JobSpec, root: Path, @@ -232,17 +256,9 @@ def render_recipe( raise ValueError(f"engine {name} differs from the requested aggregate topology") if normalized.get("enable-expert-parallel", False) is not False: raise ValueError("aggregate pilot requires expert parallelism disabled") - args["max-num-seqs"] = 2 * job.row.conc - args["max-cudagraph-capture-size"] = min(2048, 1 << (12 * job.row.conc - 1).bit_length()) - spec = args["speculative-config"] - if spec.get("synthetic_acceptance_length") is not None: - raise ValueError("recipe must not hard-code a synthetic acceptance value") - spec["enable_adaptive_verification"] = job.mode == "eval" - spec["rejection_sample_method"] = "block" if job.mode == "eval" else "synthetic" - if job.mode != "eval": - spec["synthetic_acceptance_length"] = golden_acceptance( - root, policy, spec["num_speculative_tokens"] - ) + apply_serving_point( + args, concurrency=job.row.conc, evaluation=job.mode == "eval", root=root, policy=policy + ) recipe["model"]["path"] = site.model_snapshot recipe["model"]["container"] = site.image.path recipe["identity"] = { diff --git a/utils/test_qualify_cancellation.py b/utils/test_qualify_cancellation.py index 277b9f69ca..fab815f931 100644 --- a/utils/test_qualify_cancellation.py +++ b/utils/test_qualify_cancellation.py @@ -3,6 +3,7 @@ import hashlib import json import os +import shutil import signal import subprocess import sys @@ -12,12 +13,15 @@ import pytest import yaml +from infx.srt_slurm.contracts import load_mapping +from infx.srt_slurm.job import parse_job from infx.srt_slurm.qualify_cancellation import ( DraftSite, qualify, render_probe, verify_closed_writer, ) +from infx.srt_slurm.render import ClientPolicy, render_recipe # This executable replaces only the external native/Slurm boundary. Real # preparation, path validation, trigger selection, cleanup orchestration and @@ -220,7 +224,7 @@ def pilot(tmp_path): "engine": "vllm", "slurm": {"time_limit": "08:00:00"}, "model": { - "path": "controlled/model", + "path": "deepseek-ai/DeepSeek-V4.1-Flash", "container": "controlled/image:retained", "precision": "fp4", }, @@ -235,6 +239,13 @@ def pilot(tmp_path): "tensor-parallel-size": 8, "max-model-len": 8192, "max-num-batched-tokens": 256, + "speculative-config": { + "method": "dspark", + "num_speculative_tokens": 5, + "draft_sample_method": "probabilistic", + "rejection_sample_method": "block", + "enable_adaptive_verification": True, + }, }, } }, @@ -299,7 +310,17 @@ def test_startup_cancels_the_entered_worker_and_preserves_diagnostic_evidence(pi "tensor-parallel-size": 8, "max-model-len": 8192, "max-num-batched-tokens": 256, + "max-num-seqs": 56, + "max-cudagraph-capture-size": 512, + "speculative-config": { + "method": "dspark", + "num_speculative_tokens": 5, + "draft_sample_method": "probabilistic", + "rejection_sample_method": "block", + "enable_adaptive_verification": True, + }, } + assert report["serving_point"] == {"mode": "eval", "concurrency": 28} assert state["profile"]["default_account"] == "controlled-account" assert state["profile"]["default_partition"] == "controlled-partition" assert state["profile"]["default_time_limit"] == "00:05:00" @@ -335,6 +356,71 @@ def test_client_cancellation_requires_a_real_signal_closed_writer(pilot): ) +def test_probe_serving_arguments_match_the_real_c28_eval_renderer(pilot): + root, checkout, draft = pilot + fixtures = Path(__file__).parent / "fixtures/native_pilot" + policy_path = "benchmarks/multi_node/srt-slurm-recipes/configs/client-policy.json" + shutil.copyfile(fixtures / "client-policy.json", checkout / policy_path) + golden = checkout / "golden_al_distribution/dsv41flash_dspark.yaml" + golden.parent.mkdir() + shutil.copyfile(fixtures / "golden.yaml", golden) + row = { + "image": draft["image_reference"], + "model": "deepseek-ai/DeepSeek-V4.1-Flash", + "model-prefix": "dsv41flash", + "precision": "fp4", + "framework": "vllm", + "runner": "cluster:h100-dgxc", + "tp": 8, + "pp": 1, + "dcp-size": 1, + "pcp-size": 1, + "ep": 1, + "dp-attn": False, + "spec-decoding": "mtp", + "conc": 28, + "kv-offloading": "none", + "total-cpu-dram-gb": 0, + "duration": 3600, + "exp-name": "controlled-c28-eval", + "scenario-type": "agentic-coding", + "run-eval": True, + "eval-only": True, + "eval-framework": "lm-eval", + "execution": { + "runtime": "srt-slurm", + "contract-version": 1, + "recipe": "benchmarks/multi_node/srt-slurm-recipes/dsv41flash/vllm/h100-fp4/agentx/agg-tp8-dspark5.yaml", + "profile": "runners/srt-slurm/h100-phase1.yaml", + "runtime-lock": "benchmarks/multi_node/srt-slurm-recipes/configs/prepared-runtime-lock.json", + "client-policy": policy_path, + }, + } + job = parse_job( + row, checkout, {"priority": "0", "queue-token": "controlled", "node-count": 1} + ) + site = DraftSite.model_validate(draft) + measured, _ = render_recipe( + job, + checkout, + site, + ClientPolicy.model_validate(load_mapping(checkout / policy_path)), + root / "client.json", + root / "client-output", + ) + diagnostic, _ = render_probe(checkout, site, root / "probe", "parity", 3600) + assert diagnostic["roles"] == measured["roles"] + assert diagnostic["model"] == measured["model"] + assert diagnostic["frontend"] == measured["frontend"] + assert diagnostic["roles"]["agg"]["args"]["max-num-seqs"] == 56 + assert diagnostic["roles"]["agg"]["args"]["max-cudagraph-capture-size"] == 512 + assert diagnostic["roles"]["agg"]["args"]["max-model-len"] == 8192 + assert diagnostic["roles"]["agg"]["args"]["max-num-batched-tokens"] == 256 + assert diagnostic["roles"]["agg"]["env"] == {"UNCHANGED": "retained"} + assert diagnostic["benchmark"]["argv"][1:3] == ["-I", "-c"] + assert measured["benchmark"]["argv"][1:4] == ["-I", "-m", "infx.benchmarks.eval"] + + @pytest.mark.parametrize( "scenario, expected_cancelled, message", [ From 17423228f1f4ad225233364ba53b3f36f90a818b Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sat, 19 Sep 2026 22:39:30 -0400 Subject: [PATCH 14/16] fix: pin Step Manager aware native H100 execution MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 固定已验证源码的原生运行时,保留全部历史 changelog 字节及完整九点负载,并同步中英文候选验证状态。 --- .../srt-slurm-recipes/configs/prepared-runtime-lock.json | 2 +- docs/srt-slurm-phase1.md | 2 +- docs/srt-slurm-phase1_zh.md | 2 +- perf-changelog.yaml | 9 +++++++++ 4 files changed, 12 insertions(+), 3 deletions(-) diff --git a/benchmarks/multi_node/srt-slurm-recipes/configs/prepared-runtime-lock.json b/benchmarks/multi_node/srt-slurm-recipes/configs/prepared-runtime-lock.json index afcf7a6d98..857a77b518 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/configs/prepared-runtime-lock.json +++ b/benchmarks/multi_node/srt-slurm-recipes/configs/prepared-runtime-lock.json @@ -1,7 +1,7 @@ { "schema_version": 1, "repository": "https://github.com/SemiAnalysisAI/srt-slurm.git", - "revision": "50c3dacc37def01606ee9e4e0ed873646d4f7cc5", + "revision": "62beb5ec4f8c33abc26851ba0adaded29957ca5b", "uv_lock_sha256": "f7c3ef25605ebe27bac9c6a54aa2acef7332210321fff918c3fa72b7049d4dea", "capabilities": [ "prepared-v1", diff --git a/docs/srt-slurm-phase1.md b/docs/srt-slurm-phase1.md index 788719bb15..ce4f133e4e 100644 --- a/docs/srt-slurm-phase1.md +++ b/docs/srt-slurm-phase1.md @@ -120,7 +120,7 @@ Before accepting a sweep, qualify real cancellation through the E2E `cancel-star Both probes use the real c28 eval serving settings through the same `apply_serving_point` renderer as benchmark execution: `max-num-seqs: 56`, `max-cudagraph-capture-size: 512`, and DSpark with real block rejection and adaptive verification. The model, image, TP8 topology, `max-model-len: 1048576` and `max-num-batched-tokens: 4096` remain unchanged. The diagnostic client is still a literal Python signal-aware writer. `qualification.json` records this serving point; the probe does not run or publish an eval. Matching these settings corrects the previous reliance on vLLM's sequence defaults, but does not prove the observed CUDA initialization failure is fixed. -The [H100 observer run 35483966784](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35483966784) confirmed Slurm `25.05.7`, `StepMgrEnabled=Yes` for owned job `18325`, and `enable_stepmgr` in the controller configuration. During the preceding startup probe, `squeue --steps` exposed only `batch` and `extern` despite a running aggregate step. Native step discovery is being updated to query `scontrol --oneliner show steps ` and validate the returned job, step name and running state. This is a native observation/cleanup correction; it does not relax the requirement for an actual aggregate step plus worker identity. The observer ran after job `18325` ended, so its empty step listing is not live-worker verification. The updated native runtime still requires an actual rerun. +The [H100 observer run 35483966784](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35483966784) confirmed Slurm `25.05.7`, `StepMgrEnabled=Yes` for owned job `18325`, and `enable_stepmgr` in the controller configuration. During the preceding startup probe, `squeue --steps` exposed only `batch` and `extern` despite a running aggregate step. Candidate native pin `62beb5ec4f8c33abc26851ba0adaded29957ca5b` queries `scontrol --oneliner show steps ` and validates the returned job, step name and running state. This is a native observation/cleanup correction; it does not relax the requirement for an actual aggregate step plus worker identity. The observer ran after job `18325` ended, so its empty step listing is not live-worker verification. The candidate pin and c28 probe settings still require an actual rerun. The native pin now requires `prepared-direct-listener-ownership-v1`. Before client traffic, it verifies the listening socket belongs to the recorded worker process tree, using PID start times and PID/network namespace identities. It repeats ownership checks during the client and before accepting exit 0. A foreign/replaced listener, reused PID or inaccessible ownership evidence fails the job and closes the client. Actual Pyxis namespace/proc visibility is part of cluster qualification. diff --git a/docs/srt-slurm-phase1_zh.md b/docs/srt-slurm-phase1_zh.md index 7f89e29cd4..0cdc70851b 100644 --- a/docs/srt-slurm-phase1_zh.md +++ b/docs/srt-slurm-phase1_zh.md @@ -120,7 +120,7 @@ Slurm 分配、claim、已接受 ID、调度器观察与取消均由原生运行 两种探针通过与 benchmark 执行共用的 `apply_serving_point` 渲染逻辑,采用真实 c28 eval 的服务设置:`max-num-seqs: 56`、`max-cudagraph-capture-size: 512`,以及真实 block rejection 和 adaptive verification 的 DSpark。模型、镜像、TP8 拓扑、`max-model-len: 1048576` 与 `max-num-batched-tokens: 4096` 均保持不变。诊断客户端仍是以字面代码执行、可处理信号的 Python writer。`qualification.json` 记录该服务点;探针不运行或发布 eval。此设置对齐修正了此前依赖 vLLM 默认序列上限的问题,但不能证明已观察到的 CUDA 初始化失败已经修复。 -[H100 观察运行 35483966784](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35483966784) 确认 Slurm 版本为 `25.05.7`,所属任务 `18325` 的 `StepMgrEnabled=Yes`,且 controller 配置含 `enable_stepmgr`。此前的启动期探针中,实际聚合 step 正在运行,但 `squeue --steps` 仅返回 `batch` 和 `extern`。原生 step 发现正在改为查询 `scontrol --oneliner show steps `,并校验返回的任务、step 名及运行状态。这属于原生观察/清理修正,不放宽“实际聚合 step 加 worker 身份”的要求。观察运行发生在任务 `18325` 结束后,因此当时的空 step 列表不能证明活跃 worker 的发现行为。更新后的原生运行时仍须实际重跑验证。 +[H100 观察运行 35483966784](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35483966784) 确认 Slurm 版本为 `25.05.7`,所属任务 `18325` 的 `StepMgrEnabled=Yes`,且 controller 配置含 `enable_stepmgr`。此前的启动期探针中,实际聚合 step 正在运行,但 `squeue --steps` 仅返回 `batch` 和 `extern`。候选原生 pin `62beb5ec4f8c33abc26851ba0adaded29957ca5b` 查询 `scontrol --oneliner show steps `,并校验返回的任务、step 名及运行状态。这属于原生观察/清理修正,不放宽“实际聚合 step 加 worker 身份”的要求。观察运行发生在任务 `18325` 结束后,因此当时的空 step 列表不能证明活跃 worker 的发现行为。候选 pin 与 c28 探针设置仍须实际重跑验证。 当前原生 pin 要求 `prepared-direct-listener-ownership-v1`。发送客户端流量前,它根据 PID 启动时间及 PID/network namespace 身份,验证监听套接字属于记录的 worker 进程树;客户端运行期间和接受退出码 0 之前也会重新验证。外部或替换监听器、PID 复用或无法读取的归属证据都会使任务失败并关闭客户端。实际 Pyxis 的 namespace/proc 可见性仍需集群验收。 diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 46c6eb1d37..8d0f82a5c0 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -8473,3 +8473,12 @@ - "Require owned direct-vLLM listeners and explicitly disable unprovisioned native telemetry for the H100 Phase 1 pilot; qualify the complete unchanged throughput/eval workload using isolated PR-only artifacts while app publication is unavailable." - "H100 阶段 1 试点要求直连 vLLM 监听端口的进程归属可验证,并显式关闭尚未部署的原生遥测;在 app 发布能力不可用期间,使用隔离的 PR 验收产物运行完整且参数不变的吞吐与 eval 工作负载。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3299 + +- config-keys: + - dsv41flash-fp4-h100-vllm-agentic-dspark + scenario-type: + - agentic-coding + description: + - "Pin the H100 prepared runtime to native 62beb5ec4f8c33abc26851ba0adaded29957ca5b for job-scoped StepMgr step discovery and owned cleanup. Align disposable cancellation probes with the real c28 eval serving settings (56 sequences, graph capture 512); retain the same image, TP8 topology, eight full AgentX points and separate full c28 GSM8K evaluation. Hardware lifecycle reruns remain required." + - "将 H100 预备式运行时固定至原生 62beb5ec4f8c33abc26851ba0adaded29957ca5b,按所属任务发现 StepMgr step 并清理资源。一次性取消探针对齐真实 c28 eval 服务设置(56 个序列、graph capture 512);保留相同镜像、TP8 拓扑、八个完整 AgentX 点及独立完整 c28 GSM8K eval。硬件生命周期仍须重跑验收。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3299 From 6fc59329afd2873019623be3f8ca7e12c3f5f009 Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sat, 19 Sep 2026 22:52:31 -0400 Subject: [PATCH 15/16] docs: show shared Phase 1 serving argument calls MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 更新中英文函数与文件调用图,标明正式 sweep 和取消探针共用服务参数渲染函数。 --- docs/srt-slurm-phase1.md | 5 +++++ docs/srt-slurm-phase1_zh.md | 5 +++++ 2 files changed, 10 insertions(+) diff --git a/docs/srt-slurm-phase1.md b/docs/srt-slurm-phase1.md index ce4f133e4e..3ac6126855 100644 --- a/docs/srt-slurm-phase1.md +++ b/docs/srt-slurm-phase1.md @@ -40,6 +40,10 @@ flowchart TD SP --> SI[inspect_assets: observed shared files] SP --> SR[infx.srt_slurm.provision_runtime.provision] E --> CQ[infx.srt_slurm.qualify_cancellation.qualify] + CQ --> CR[infx.srt_slurm.qualify_cancellation.render_probe] + CR --> AS[infx.srt_slurm.render.apply_serving_point] + CR --> Y + CR --> H CQ --> NI SR --> SD[Private runtimes, offline cache and site draft] SD --> PQ[PreparedSite: same-repository PR qualification] @@ -51,6 +55,7 @@ flowchart TD F --> P[infx.srt_slurm.launch.prepare] P --> CP[infx.benchmarks.prepare.prepare] P --> R[infx.srt_slurm.render.render_recipe] + R --> AS R --> Y["dsv41flash/vllm/h100-fp4/agentx/
agg-tp8-dspark5.yaml"] R --> H[runners/srt-slurm/h100-phase1.yaml] P --> NP[srtctl prepare] diff --git a/docs/srt-slurm-phase1_zh.md b/docs/srt-slurm-phase1_zh.md index 0cdc70851b..a5f276a990 100644 --- a/docs/srt-slurm-phase1_zh.md +++ b/docs/srt-slurm-phase1_zh.md @@ -40,6 +40,10 @@ flowchart TD SP --> SI[inspect_assets:实际共享文件] SP --> SR[infx.srt_slurm.provision_runtime.provision] E --> CQ[infx.srt_slurm.qualify_cancellation.qualify] + CQ --> CR[infx.srt_slurm.qualify_cancellation.render_probe] + CR --> AS[infx.srt_slurm.render.apply_serving_point] + CR --> Y + CR --> H CQ --> NI SR --> SD[专用运行环境、离线缓存与站点草稿] SD --> PQ[PreparedSite:同仓库 PR 验收] @@ -51,6 +55,7 @@ flowchart TD F --> P[infx.srt_slurm.launch.prepare] P --> CP[infx.benchmarks.prepare.prepare] P --> R[infx.srt_slurm.render.render_recipe] + R --> AS R --> Y["dsv41flash/vllm/h100-fp4/agentx/
agg-tp8-dspark5.yaml"] R --> H[runners/srt-slurm/h100-phase1.yaml] P --> NP[srtctl prepare] From c3db6c7df5c0dbe5222eccff9a4403707c5c1bed Mon Sep 17 00:00:00 2001 From: functionstackx <47992694+functionstackx@users.noreply.github.com> Date: Sat, 19 Sep 2026 22:56:54 -0400 Subject: [PATCH 16/16] docs: record corrected Phase 1 CI and sweep replacement MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 同步中英文验证账本,保留已取消运行的真实证据及新环境准备和硬件验收的待完成状态。 --- docs/srt-slurm-phase1.md | 8 +++++++- docs/srt-slurm-phase1_zh.md | 8 +++++++- 2 files changed, 14 insertions(+), 2 deletions(-) diff --git a/docs/srt-slurm-phase1.md b/docs/srt-slurm-phase1.md index 3ac6126855..699d8d8e27 100644 --- a/docs/srt-slurm-phase1.md +++ b/docs/srt-slurm-phase1.md @@ -151,9 +151,15 @@ The first allocated lifecycle probes used InferenceX `bc0710934f3ca2dbdd37940aa9 [Client run 35483590787](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35483590787) accepted job `18325`, whose vLLM worker failed during DSpark initialization with CUDA index-out-of-bounds/device-assert errors. Native observation confirms terminal `FAILED` and completed cleanup; cancellation found the allocation already terminal. No diagnostic writer started, so there is no intended client-interruption or closed-writer evidence. Both runs retain `lifecycle_qualified: false`. The c28 serving-parity and native step-discovery fixes need fresh hardware verification; neither failed run qualifies a lifecycle gate or establishes a sweep outcome. +[Candidate CI 35484873112](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35484873112) passed on InferenceX `17423228f1f4ad225233364ba53b3f36f90a818b` with native `62beb5ec4f8c33abc26851ba0adaded29957ca5b`: 1,925 producer tests passed with three skipped; 2,507 native Linux tests passed with two skipped and six actual-GPU integration tests deselected. The source/lock/lineage checks, installed-runtime contract covering all nine prepared points, lint and MCP construction/discovery checks passed. [Strict Zizmor 35484875668](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35484875668) also passed on that candidate. These checks do not establish H100 or lifecycle qualification. + +[Old full sweep 35483266492](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35483266492), using producer `065cb56de373dc89343f17307e56851cd0a4df24` and native `50c3dacc37def01606ee9e4e0ed873646d4f7cc5`, was deliberately cancelled during observed asset preparation because it cannot qualify the final corrected native pin and wrapper. The sole sweep-enabling label was removed, and all nine GitHub point jobs ended `cancelled`; no point qualified. All nine launch logs show interruption in the initial `collect_assets` hash pass, before bundle creation or native submission. The [post-cancellation observer 35485241189](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35485241189) succeeded at `2026-09-20T02:56:21Z`: all nine exact paths had no client runtime, executable bundle or native receipt, and its bounded process scan found no matching workflow processes. Together with the terminal logs and submission ordering, this confirms that the cancelled sweep submitted no Slurm allocation; no GPU benchmark ran. + +[Replacement provisioning 35484558552](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35484558552) is in progress using the actual candidate branch SHA `17423228f1f4ad225233364ba53b3f36f90a818b`, not the final PR merge SHA. Adopting its generation requires preserving that recorded provisioning source and verifying the installed wrapper bytes and runtime inputs against the actual final PR merge checkout. If equivalence does not hold, provision a fresh generation from that exact merge SHA. Provisioning success, final-source equivalence, both lifecycle gates and the complete H100 sweep remain pending. + | Gate | Status / required evidence | | --- | --- | -| Native and client behavior | CPU tests and installed-wheel checks; no GPU claim | +| Native and client behavior | Candidate `17423228` / native `62beb5ec` Linux CI, installed nine-point contract and strict workflow audit passed; no GPU claim | | Receipt, app and recovery | Reader rolled back and revision variable removed; additive schema retained; native receipt import and publication gated | | H100 throughput | c1,2,4,8,16,20,24,28 pending on the unchanged image | | Real evaluation | New c28 run pending; historical full raw eval passes validator | diff --git a/docs/srt-slurm-phase1_zh.md b/docs/srt-slurm-phase1_zh.md index a5f276a990..5d123172da 100644 --- a/docs/srt-slurm-phase1_zh.md +++ b/docs/srt-slurm-phase1_zh.md @@ -151,9 +151,15 @@ Merge helper 保留最近明确授权的 `/use RUN_ID` 或 `/reuse-sweep-run RUN [客户端运行 35483590787](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35483590787) 接受任务 `18325`;其 vLLM worker 在 DSpark 初始化期间出现 CUDA 索引越界/device-assert 错误并失败。原生观察确认终态 `FAILED` 且清理完成;取消操作发现 allocation 已经终止。诊断 writer 未启动,因此没有预期客户端中断或 writer 关闭证据。两次运行均保留 `lifecycle_qualified: false`。c28 服务参数对齐和原生 step 发现修正需要新的硬件验证;这两次失败运行均不能关闭生命周期验收门禁,也不能证明 sweep 的最终结果。 +[候选 CI 35484873112](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35484873112) 在 InferenceX `17423228f1f4ad225233364ba53b3f36f90a818b` 和原生 `62beb5ec4f8c33abc26851ba0adaded29957ca5b` 上通过:1,925 项 producer 测试通过、三项跳过;2,507 项原生 Linux 测试通过、两项跳过,另有六项实际 GPU 集成测试未选择执行。源代码/锁文件/版本沿袭校验、覆盖全部九个准备点的已安装运行时契约、lint 及 MCP 构造/发现检查均通过。[严格 Zizmor 35484875668](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35484875668) 也在该候选提交上通过。这些检查不能证明 H100 或生命周期验收完成。 + +[旧完整 sweep 35483266492](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35483266492) 使用 producer `065cb56de373dc89343f17307e56851cd0a4df24` 和原生 `50c3dacc37def01606ee9e4e0ed873646d4f7cc5`。由于它无法验收最终修正后的原生 pin 与 wrapper,已在观察到的资源准备阶段主动取消。唯一的 sweep 启用标签已移除,九个 GitHub 测试点任务均以 `cancelled` 结束,没有测试点通过验收。九个启动日志均显示在最初的 `collect_assets` 哈希阶段中断,此时尚未创建 bundle 或进行原生提交。[取消后观察运行 35485241189](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35485241189) 于 `2026-09-20T02:56:21Z` 成功:九个准确路径均没有客户端 runtime、可执行 bundle 或原生 receipt,有界进程扫描未发现匹配的 workflow 进程。结合终止日志和提交顺序,可确认被取消的 sweep 未提交任何 Slurm allocation,也未运行 GPU benchmark。 + +[替代环境准备运行 35484558552](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35484558552) 正在进行,实际使用候选分支 SHA `17423228f1f4ad225233364ba53b3f36f90a818b`,并非最终 PR merge SHA。采用该 generation 时,必须保留其记录的准备源提交,并针对实际最终 PR merge checkout 验证已安装 wrapper 字节与运行时输入是否等价。如果不等价,须从该精确 merge SHA 重新准备独立 generation。环境准备成功、最终源代码等价性、两项生命周期门禁和完整 H100 sweep 均仍待验证。 + | Gate | 状态 / 所需证据 | | --- | --- | -| 原生及客户端行为 | CPU 测试、已安装 wheel 检查;不声称 GPU 验收 | +| 原生及客户端行为 | 候选 `17423228` / 原生 `62beb5ec` 的 Linux CI、已安装九点契约及严格 workflow 审计通过;不声称 GPU 验收 | | 回执、app、恢复 | reader 已回滚,revision 变量已删除;新增 schema 保留;原生回执导入与发布仍受门禁限制 | | H100 吞吐 | 原镜像 c1、2、4、8、16、20、24、28 待运行 | | 真实 eval | 新 c28 待运行;历史完整原始 eval 通过新 validator |