From 55ed3c56f5e583bc7c9178e66a19778e2fdd0e17 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 22 Sep 2026 12:52:22 +0000 Subject: [PATCH 01/14] feat: Validate Actor input and apply input-schema defaults A build now records the input schema from the Actor's pushed source, and every run started against that build has its input validated against it with the schema's defaults applied - the last piece of run-start behaviour that was listed as unsupported. Resolution at build time follows what apify-cli looks for locally: the `input` field of `.actor/actor.json` (inline object or a path relative to `.actor/`), then `.actor/INPUT_SCHEMA.json`, then `INPUT_SCHEMA.json` at the Actor root, matched case-insensitively with every outcome stated in the build log. A schema that is itself invalid fails the build rather than producing an image whose every later run would silently skip validation. At run start, a build with a schema gets the platform's own behaviour: defaults fill every omitted field (nested objects and array items included), the filled-in input is what lands in the run's INPUT record - so a call with no input at all still runs on the defaults - and an input the schema rejects answers 400 without creating a run or a container. Validation itself is the platform's, through @apify/input_schema and the same AJV configuration the API uses, so the message a developer reads locally is the one the real API would give. Two deliberate local differences, both documented: proxy group availability is not checked (no proxy groups are emulated here), and encrypted secret input fields stay unsupported. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_011U9yUTUn78bqGDn5obYSXs --- README.md | 17 ++ package.json | 2 + pnpm-lock.yaml | 97 ++++++ requirements/actor-driver.md | 38 +++ requirements/api.md | 15 +- requirements/test.md | 1 + requirements/unsupported.md | 1 - skills/actor-runtime/SKILL.md | 22 ++ src/api/routes/actors.ts | 34 ++- src/services/builds.ts | 24 ++ src/services/input-schema-location.ts | 187 ++++++++++++ src/services/input-schema.ts | 313 ++++++++++++++++++++ src/storage/entities.ts | 19 ++ test/e2e/actor-dev-loop.test.ts | 26 ++ test/e2e/helpers/apify-cli.ts | 18 ++ test/integration/input-schema.test.ts | 273 +++++++++++++++++ test/unit/input-schema-location.test.ts | 200 +++++++++++++ test/unit/input-schema.test.ts | 375 ++++++++++++++++++++++++ 18 files changed, 1655 insertions(+), 7 deletions(-) create mode 100644 src/services/input-schema-location.ts create mode 100644 src/services/input-schema.ts create mode 100644 test/integration/input-schema.test.ts create mode 100644 test/unit/input-schema-location.test.ts create mode 100644 test/unit/input-schema.test.ts diff --git a/README.md b/README.md index 973e164..527b629 100644 --- a/README.md +++ b/README.md @@ -46,6 +46,23 @@ In a build or run log, everything the runtime itself has to say - dev-folder not attach line, the browser-view URL, migration markers, a run that could not be started - opens with a blue `[actor-runtime]` prefix. Your Actor's own output is passed through byte for byte. +## Input schema: defaults and validation + +An Actor that declares an input schema (the `input` field of `.actor/actor.json`, `.actor/INPUT_SCHEMA.json`, +or `INPUT_SCHEMA.json` at its root) gets the platform's behaviour locally: the schema's defaults are filled into +every run's input, and an input the schema rejects fails the call with the API's own message instead of starting a +container. + +```bash +apify call # runs on the schema's defaults +apify call --input '{"maxPages":0}' # 400 Input is not valid: Field input.maxPages must be >= 1 +``` + +The schema is read at build time, so editing it locally needs an `apify push` even under a registered dev folder, +and a schema the Apify meta-schema rejects fails the build with the defect in its log. Proxy group availability is +not checked locally, and encrypted secret input fields stay unsupported - see +`requirements/actor-driver.md`'s "Input schema, validation and defaults". + ## Running with Podman instead of Docker The runtime talks to the container engine only through its Docker-compatible API socket, and Podman diff --git a/package.json b/package.json index cb3fbc6..0e6c2a5 100644 --- a/package.json +++ b/package.json @@ -26,9 +26,11 @@ "test:watch": "vitest" }, "dependencies": { + "@apify/input_schema": "^3.29.2", "@crawlee/core": "4.0.0-beta.145", "@crawlee/fs-storage": "4.0.0-beta.145", "@novnc/novnc": "1.7.0", + "ajv": "^8.20.0", "dockerode": "^4.0.5", "express": "^5.1.0", "json5": "^2.2.3", diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 8127ff0..5c528b0 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -11,6 +11,9 @@ importers: .: dependencies: + '@apify/input_schema': + specifier: ^3.29.2 + version: 3.29.2(ajv@8.20.0) '@crawlee/core': specifier: 4.0.0-beta.145 version: 4.0.0-beta.145 @@ -20,6 +23,9 @@ importers: '@novnc/novnc': specifier: 1.7.0 version: 1.7.0 + ajv: + specifier: ^8.20.0 + version: 8.20.0 dockerode: specifier: ^4.0.5 version: 4.0.12 @@ -90,12 +96,29 @@ packages: '@apify/consts@2.57.2': resolution: {integrity: sha512-iCAltDfZqEZazrMFoGIp3EEkvd7GTlgs9y74QZ+JTz+zH3GqBvOD2DfW4WYsqI1rfKlLyo5z3S7wipWpkSNNJQ==} + '@apify/consts@2.58.1': + resolution: {integrity: sha512-pKnoPincN2eqiSANmX2u19FihJATX3f1ucMhxss7t2epUmB/UhSjGM7G4immjC7Q/CXrNmbETBbOGmR0/EH4/Q==} + '@apify/datastructures@2.0.6': resolution: {integrity: sha512-hI/Q5cCjXXI/g4OPwX+/xuk+JsOtXRhKNXl2wnq/nKmdrji3nYCc9vGsAEc1iXVbs16xQMZy5uRhoY3XTABovA==} + '@apify/input_schema@3.29.2': + resolution: {integrity: sha512-hKW/7w1B7nu3v1gXeWAXhXV/qCQnZ1d/NdXUSVB4Ryb162Lp5Tn+lOjDbYpb7/+Qc5Y5JvOx/fLLg6/qdQ6WzQ==} + peerDependencies: + ajv: ^8.0.0 + + '@apify/input_secrets@1.2.60': + resolution: {integrity: sha512-KfrnFFTy7pozNxgX6MBdeCbVIhxM/t6t8gDroRbADJyYmaGtkqsxk9t4/xw9ri+ZObiWdWQrlU10xF+NUqDQBg==} + + '@apify/json_schemas@0.17.2': + resolution: {integrity: sha512-nN1QiBX+FJJyOG2OTjE6X6mG1EPYz5Je5TA4JmQ8JyWmWilQ/6hBDB6mUyg1JmvS3QDd+swtZS0G3dJ7sP+6uQ==} + '@apify/log@2.5.50': resolution: {integrity: sha512-obSTa6SUP4Ywy9EqcCMoHZXL0kH4lGTgajqlqTVkkdchLKAGNv+hsNLRSXz15N5aHJXwykqdWxidWJrRNawuxA==} + '@apify/log@2.5.54': + resolution: {integrity: sha512-EZRHrKCWnet0+6WaTx14Mpjxrn/PRudmRk6/N1dCTz+jtHN+Pbdvw5FFtIrCUngPhm4+bdI+kKVYUBmvTwrjvg==} + '@apify/ps-tree@1.2.0': resolution: {integrity: sha512-VHIswI7rD/R4bToeIDuJ9WJXt+qr5SdhfoZ9RzdjmCs9mgy7l0P4RugQEUCcU+WB4sfImbd4CKwzXcn0uYx1yw==} engines: {node: '>= 0.10'} @@ -104,6 +127,9 @@ packages: '@apify/timeout@0.4.10': resolution: {integrity: sha512-RlsqjT47MX5W5DC8yG1sBq38qZCYRGia6gCmmqiio1Cg6TWjIRlrUfFCY3lVuxn9MOPCOcHqxsEqsVvFX5990g==} + '@apify/utilities@2.35.11': + resolution: {integrity: sha512-od9Axa/+OSZB5JO8aljAnN7lS8vDb5KqUEAkUiZZh95K5xIrC0GzN45DNcK8oItnJ3QT/oqtHdaa22TipinDvQ==} + '@apify/utilities@2.35.7': resolution: {integrity: sha512-UUJkBCCWpOdTPpGenGtkNnnUbwmgANwxJDZoDPfa27+V23V8UfurE/lGA0LCGDH6DWYYJkQoEAtehxYJHR0U9g==} @@ -919,6 +945,10 @@ packages: peerDependencies: acorn: ^6.0.0 || ^7.0.0 || ^8.0.0 + acorn-loose@8.5.2: + resolution: {integrity: sha512-PPvV6g8UGMGgjrMu+n/f9E/tCSkNQ2Y97eFvuVdJfG11+xdIeDcLyNdC8SHcrHbRqkfwLASdplyR6B6sKM1U4A==} + engines: {node: '>=0.4.0'} + acorn@8.18.0: resolution: {integrity: sha512-lGq+9yr1/GuAWaVYIHRjvvySG5/4VfKIvC8EWxStPdcDh/Ka7FG3twP6v4d5BkravUilhIAsG4Qj83t02LWUPQ==} engines: {node: '>=0.4.0'} @@ -935,6 +965,9 @@ packages: ajv@6.15.0: resolution: {integrity: sha512-fgFx7Hfoq60ytK2c7DhnF8jIvzYgOMxfugjLOSMHjLIPgenqa7S7oaagATUq99mV6IYvN2tRmC0wnTYX6iPbMw==} + ajv@8.20.0: + resolution: {integrity: sha512-Thbli+OlOj+iMPYFBVBfJ3OmCAnaSyNn4M1vz9T6Gka5Jt9ba/HIR56joy65tY6kx/FCF5VXNB819Y7/GUrBGA==} + ansi-colors@4.1.3: resolution: {integrity: sha512-/6w/C21Pm1A7aZitlI5Ni/2J6FFQN8i1Cvz3kHABAAbw93v/NlvKdVOqz7CCWz/3iv/JplRSEEZ83XION15ovw==} engines: {node: '>=6'} @@ -1139,6 +1172,9 @@ packages: resolution: {integrity: sha512-yki5XnKuf750l50uGTllt6kKILY4nQ1eNIQatoXEByZ5dWgnKqbnqmTrBE5B4N7lrMJKQ2ytWMiTO2o0v6Ew/w==} engines: {node: '>= 0.6'} + countries-list@3.4.1: + resolution: {integrity: sha512-sKLGAZD5EAdzQwQ19vOp/9u33MTjmZ0CkdXPYtp/An6cyIlJ/E8TwGf2FcPe1AnzHO+h7nbTfoHyXIi58Hpa+Q==} + cpu-features@0.0.10: resolution: {integrity: sha512-9IkYqtX3YHPCzoVg1Py+o9057a3i0fp7S530UWokCSaFVTc7CwXPRiOjRjBQQ18ZCNafx78YfnG+HALxtVmOGA==} engines: {node: '>=10.0.0'} @@ -1385,6 +1421,9 @@ packages: fast-levenshtein@2.0.6: resolution: {integrity: sha512-DCXu6Ifhqcks7TZKY3Hxp3y6qphY5SJZmrWMDrKcERSOXWQdMhU9Ig/PYrzyw/ul9jOIyh0N4M0tbC5hodg8dw==} + fast-uri@3.1.8: + resolution: {integrity: sha512-GZMtZUTNRpOVIECoXwLNZS5xUGE+mVNbTB8h/7Rwh2TFWcBQiPzTgyZi05BF9UMZKkLJv8XBRJTlU7zg8+ZfMg==} + fdir@6.5.0: resolution: {integrity: sha512-tIbYtZbucOs0BRGqPJkshJUYdL+SDH7dVM8gjy+ERp3WAUjLEFJE+02kanyHtwjWOnwrKYBiwAmM0p4kLJAnXg==} engines: {node: '>=12.0.0'} @@ -1585,6 +1624,9 @@ packages: json-schema-traverse@0.4.1: resolution: {integrity: sha512-xbbCH5dCYU5T8LcEhhuh7HJ88HXuW3qsI3Y0zOZFKfZEHcpWiHU/Jxzk629Brsab/mMiHQti9wMP+845RPe3Vg==} + json-schema-traverse@1.0.0: + resolution: {integrity: sha512-NM8/P9n3XjXhIZn1lLhkFaACTOURQXjWhV4BA/RnOv8xvgqtqpAX9IO4mRQxSx1Rlo4tqzeqb0sOlruaOy3dug==} + json-stable-stringify-without-jsonify@1.0.1: resolution: {integrity: sha512-Bdboy+l7tA3OGW6FjyFHWkP5LuByj1Tk33Ljyq0axyzdk9//JSi2u3fP1QSmd1KNwq6VOKYGlAu87CisVir6Pw==} @@ -1832,6 +1874,10 @@ packages: resolution: {integrity: sha512-fGxEI7+wsG9xrvdjsrlmL22OMTTiHRwAMroiEeMgq8gzoLC/PQr7RsRDSTLUg/bZAZtF+TVIkHc6/4RIKrui+Q==} engines: {node: '>=0.10.0'} + require-from-string@2.0.2: + resolution: {integrity: sha512-Xf0nWe6RseziFMu+Ap9biiUbmplq6S9/p+7w7YXP/JBHhrUDDUhwa+vANyubuqfZWTveU//DYVGsDG7RKL/vEw==} + engines: {node: '>=0.10.0'} + resolve-from@4.0.0: resolution: {integrity: sha512-pb/MYmXstAkysRFx8piNI1tGFNQIFA3vkE3Gq4EuA1dF6gHp/+vgZqsCGJapvy8N3Q+4o7FwvquPJcnZ7RYy4g==} engines: {node: '>=4'} @@ -2245,19 +2291,51 @@ snapshots: '@apify/consts@2.57.2': {} + '@apify/consts@2.58.1': {} + '@apify/datastructures@2.0.6': {} + '@apify/input_schema@3.29.2(ajv@8.20.0)': + dependencies: + '@apify/consts': 2.58.1 + '@apify/input_secrets': 1.2.60 + '@apify/json_schemas': 0.17.2 + acorn-loose: 8.5.2 + ajv: 8.20.0 + countries-list: 3.4.1 + + '@apify/input_secrets@1.2.60': + dependencies: + '@apify/log': 2.5.54 + '@apify/utilities': 2.35.11 + ow: 0.28.2 + + '@apify/json_schemas@0.17.2': + dependencies: + '@apify/consts': 2.58.1 + ajv: 8.20.0 + '@apify/log@2.5.50': dependencies: '@apify/consts': 2.57.2 ansi-colors: 4.1.3 + '@apify/log@2.5.54': + dependencies: + '@apify/consts': 2.58.1 + ansi-colors: 4.1.3 + '@apify/ps-tree@1.2.0': dependencies: event-stream: 3.3.4 '@apify/timeout@0.4.10': {} + '@apify/utilities@2.35.11': + dependencies: + '@apify/consts': 2.58.1 + '@apify/log': 2.5.54 + '@apify/utilities@2.35.7': dependencies: '@apify/consts': 2.57.2 @@ -2937,6 +3015,10 @@ snapshots: dependencies: acorn: 8.18.0 + acorn-loose@8.5.2: + dependencies: + acorn: 8.18.0 + acorn@8.18.0: {} agent-base@6.0.2: @@ -2954,6 +3036,13 @@ snapshots: json-schema-traverse: 0.4.1 uri-js: 4.4.1 + ajv@8.20.0: + dependencies: + fast-deep-equal: 3.1.3 + fast-uri: 3.1.8 + json-schema-traverse: 1.0.0 + require-from-string: 2.0.2 + ansi-colors@4.1.3: {} ansi-regex@5.0.1: {} @@ -3174,6 +3263,8 @@ snapshots: cookie@0.7.2: {} + countries-list@3.4.1: {} + cpu-features@0.0.10: dependencies: buildcheck: 0.0.7 @@ -3515,6 +3606,8 @@ snapshots: fast-levenshtein@2.0.6: {} + fast-uri@3.1.8: {} + fdir@6.5.0(picomatch@4.0.5): optionalDependencies: picomatch: 4.0.5 @@ -3711,6 +3804,8 @@ snapshots: json-schema-traverse@0.4.1: {} + json-schema-traverse@1.0.0: {} + json-stable-stringify-without-jsonify@1.0.1: {} json5@2.2.3: {} @@ -3958,6 +4053,8 @@ snapshots: require-directory@2.1.1: {} + require-from-string@2.0.2: {} + resolve-from@4.0.0: {} retry@0.13.1: {} diff --git a/requirements/actor-driver.md b/requirements/actor-driver.md index e1ef33c..f620635 100644 --- a/requirements/actor-driver.md +++ b/requirements/actor-driver.md @@ -34,6 +34,44 @@ 4. the platform's bundled default Dockerfile, for that build only - the pushed source itself is unchanged. - Matching is case-insensitive, exact-case wins ties, and every outcome is stated in the build log. +# Input schema, validation and defaults + +- **A build records the input schema found in the Actor's pushed source**, and every run started against + that build has its input validated against it and the schema's defaults applied. A build that has no + input schema runs every input through untouched, exactly as the caller sent it. +- Resolution order at build time, stopping at the first hit, matching what `apify-cli` looks for + locally: + 1. the `input` field of `.actor/actor.json` - either an inline schema object, or a path relative to + `.actor/`; a path escaping the Actor root fails the build with "points outside the Actor root + directory", a value that is neither a string nor an object fails with `"input" must be a string +or an object`, and a path naming no pushed file falls through instead of failing. + 2. `.actor/INPUT_SCHEMA.json` + 3. `INPUT_SCHEMA.json` at the Actor root + - Matching is case-insensitive, exact-case wins ties, and every outcome is stated in the build log. + Schema files are parsed as JSON5, like `.actor/actor.json` already is. +- **A schema that is itself invalid fails the build**, with the defect in both the build log and the + build's status message - rather than producing an image whose every later run silently skips + validation. Validity is judged by the Apify input-schema meta-schema, the platform's own. +- **At run start** (`POST /v2/actors/:actorId/runs`), for a build that has a schema: + - The input must be `application/json` and must parse to a JSON object; `null` counts as no input. + - The schema's defaults fill in every field the caller left out, nested objects and the items of + arrays included, and the result is what is stored as the run's `INPUT` record. A run started with + no input at all therefore still gets an `INPUT` record holding the defaults. + - A field that is `required` but has a `default` is satisfied by that default; a required array must + hold at least one item. + - An input the schema rejects answers `400` and starts nothing - no run record, no container + (`api.md`). + - The schema used is that of the build the run actually resolved, never another tag's more recently + pushed one. +- **Known differences from the platform**: proxy group availability is not checked, since this runtime + emulates no proxy groups - a `proxy` field is still validated for shape, custom proxy URLs and + country code, but any `apifyProxyGroups` selection is accepted. Encrypted secret input fields stay + unsupported (`unsupported.md`), so a value for an `isSecret` field is validated as the plain value it + locally is. +- The schema comes from the **build**, so an Actor running from a registered dev folder validates + against the schema of its last build: editing `INPUT_SCHEMA.json` locally needs an `apify push` to + take effect, unlike a source edit. + # Bind mount volumes with Actor source code - To let an Actor be re-run with source changes and no rebuild, the Actor's registered local dev diff --git a/requirements/api.md b/requirements/api.md index 9fcbe67..f13280c 100644 --- a/requirements/api.md +++ b/requirements/api.md @@ -17,6 +17,15 @@ - `DELETE /v2/actor-builds/:buildId` and `DELETE /v2/actor-runs/:runId` on a **non-terminal** build/run are rejected, not aborted-then-deleted: `400` with error type `deleting-unfinished-build` (builds) or `cannot-remove-running-run` (runs), matching the Apify platform. +- `POST /v2/actors/:actorId/runs` validates the input against the resolved build's input schema, when it + has one (`actor-driver.md`'s "Input schema, validation and defaults"), and answers `400` with the + platform's own error types and messages when it does not pass - starting nothing at all: + - `invalid-input` - `Actor input must have content type "application/json".`, `Cannot parse input +JSON body: ...`, `The input JSON must be object, got "" instead.`, or `Input is not valid: +`. + - `invalid-input-schema` - `Input schema is not valid: ...`, for a stored schema that cannot be + compiled at all (a build made by an older runtime, before builds validated their schemas). + - A build with no input schema accepts any body, unvalidated, exactly as before. - Three endpoints are exceptions to the `{data}` envelope: - `GET /v2/logs/:buildOrRunId` (and its `actor-builds`/`actor-runs` aliases): the body is plain text, never `{data}`-wrapped, matching apify-client-js's `log().get()`. @@ -283,9 +292,9 @@ This runtime emulates that observable experience on demand: - `fallbackNotFoundEnabled` covers a request that reaches a route this runtime does serve, but whose specific record id doesn't exist locally (`record-not-found`, see "Response envelopes" above). - - Every other error type - `invalid-request`, `user-not-authenticated`, - `cannot-remove-running-run`, `deleting-unfinished-build`, any `dev-folder-*` type, - `internal-error` - is never relayed, regardless of either toggle's state. + - Every other error type - `invalid-request`, `invalid-input`, `invalid-input-schema`, + `user-not-authenticated`, `cannot-remove-running-run`, `deleting-unfinished-build`, any + `dev-folder-*` type, `internal-error` - is never relayed, regardless of either toggle's state. - **All HTTP methods are eligible for both toggles, writes included**: a `POST`/`PUT`/`DELETE` that would otherwise 404/501 locally is relayed exactly like a `GET` when its toggle is on - and, if the platform accepts it, becomes a real write against the caller's real account. This is a deliberate diff --git a/requirements/test.md b/requirements/test.md index 495fadb..ee4d19b 100644 --- a/requirements/test.md +++ b/requirements/test.md @@ -36,6 +36,7 @@ Test case must verify full Actor development flow: - Push and build Actor in local actor runtime `apify push` - Run each sample Actor in the local actor runtime with `apify call --input '{"maxPages":N}'` for at least two different values of `N`, waiting for each run to finish - Assert via `apify datasets info ` that the default dataset's `itemCount` tracks `N` - the assertion is input-dependent, not just "some items exist" +- Cover the input schema through the CLI too (`actor-driver.md`'s "Input schema, validation and defaults"): a `apify call` with no `--input` at all must run on the schema's defaults, with the same input-dependent `itemCount` assertion, and a `apify call` whose input the schema rejects must fail, naming the offending field, with no run started ## Browser view diff --git a/requirements/unsupported.md b/requirements/unsupported.md index 2d5fcef..d79de87 100644 --- a/requirements/unsupported.md +++ b/requirements/unsupported.md @@ -25,7 +25,6 @@ real account, not in the runtime. - Ad-hoc webhooks on run start - Run usage, cost and compute units - Last-run shortcuts (`runs/last`) -- Input validation and defaults from the input schema - Encrypted secret input fields - Actor-level default run options - Dynamic and bounded memory from `.actor/actor.json` diff --git a/skills/actor-runtime/SKILL.md b/skills/actor-runtime/SKILL.md index 086b1a9..729744d 100644 --- a/skills/actor-runtime/SKILL.md +++ b/skills/actor-runtime/SKILL.md @@ -53,6 +53,28 @@ both `apify/actor-node-playwright*` and `apify/actor-python-playwright*` are - a those is retried for `linux/amd64`, the architecture the Apify platform builds and runs on, with the reason in the build log. The engine emulates it (Rosetta on Apple Silicon), so it works, just slower. +## Input schema: defaults and validation + +If the Actor declares an input schema - the `input` field of `.actor/actor.json`, or +`.actor/INPUT_SCHEMA.json`, or `INPUT_SCHEMA.json` at its root - the runtime uses it the way the +platform does: + +```sh +apify call # no input: the run gets the schema's defaults +apify call --input '{"maxPages":0}' # rejected before anything starts, if the schema forbids it +``` + +- Every field left out of `--input` is filled from the schema's `default`, and that filled-in input is + what the run reads back as `INPUT`. +- An input the schema rejects fails the call with the same message the real API gives (`Input is not +valid: Field input.maxPages must be >= 1`); no run is created and no container starts. +- The schema comes from the **build**, so a locally edited `INPUT_SCHEMA.json` needs an `apify push` to + take effect - unlike a source edit under a registered dev folder. +- A schema the Apify input-schema meta-schema rejects fails the **build**, with the defect in the build + log, instead of being silently ignored. +- Two local differences: Apify Proxy groups are not checked (any `apifyProxyGroups` selection is + accepted), and encrypted secret input fields are not supported. + ## Iterate without rebuilding (dev folder) After that first `apify push`, the runtime registers the pushed directory as the Actor's **dev diff --git a/src/api/routes/actors.ts b/src/api/routes/actors.ts index 93890db..a9dfc2e 100644 --- a/src/api/routes/actors.ts +++ b/src/api/routes/actors.ts @@ -3,7 +3,7 @@ import type { Router } from 'express'; import { requireUser } from '../auth.js'; import { paginate, sendData, sendPaginated, sortByTimestamp } from '../envelope.js'; -import { recordNotFound, invalidRequest } from '../errors.js'; +import { ApiError, recordNotFound, invalidRequest } from '../errors.js'; import { h, jsonBody, paginationParams, queryBoolean, queryNumber, queryString, rawBody } from '../handler.js'; import { addOrReplaceVersion, @@ -25,10 +25,16 @@ import { import { listOwnedRuns, startRun, waitForRunFinish } from '../../services/runs.js'; import { getRegistries } from '../../storage/registries.js'; import { actorDto, buildDto, runDto } from '../dto/actors.js'; -import type { ActorVersionRecord } from '../../storage/entities.js'; +import type { ActorVersionRecord, BuildRecord } from '../../storage/entities.js'; import type { ApiServerDeps } from '../server.js'; import { CONTAINER_API_BASE_URL } from '../../config.js'; import { resolveProxyPassword } from '../../services/users.js'; +import { + describeInputProcessingFailure, + processActorInput, + type ActorInput, + type InputProcessingResult, +} from '../../services/input-schema.js'; export function mountActors(router: Router, deps: ApiServerDeps): void { router.get( @@ -272,6 +278,27 @@ export function mountActors(router: Router, deps: ApiServerDeps): void { }), ); + /** Maps a non-`ok` `InputProcessingResult` to the `ApiError` this route throws - the HTTP status and + * error `type` are the route's concern, the message text the service's (`services/input-schema.ts`), + * the same split `api/routes/dev-folder.ts` already uses. Both types, and both messages, are the + * real platform's for the same input (`@apify-packages/errors`'s `actor.inputNotValid` and friends). */ + function inputApiError(result: Exclude): ApiError { + const message = describeInputProcessingFailure(result); + const type = result.kind === 'invalid-schema' ? 'invalid-input-schema' : 'invalid-input'; + return new ApiError(400, type, message); + } + + /** The run's effective input: the caller's bytes as-is for a build with no input schema, or - for one + * that has a schema - the input with the schema's defaults applied, validated, and re-serialized + * (`services/input-schema.ts`). A schema turns "no input at all" into "the defaults", which is why + * this can return an input for a request that carried no body. */ + function resolveRunInput(build: BuildRecord, raw: ActorInput | undefined): ActorInput | undefined { + if (!build.inputSchema) return raw; + const processed = processActorInput(raw, build.inputSchema); + if (processed.kind !== 'ok') throw inputApiError(processed); + return processed.input; + } + router.post( '/actors/:actorId/runs', h(async (req, res) => { @@ -298,8 +325,9 @@ export function mountActors(router: Router, deps: ApiServerDeps): void { const build = lookup.build; const body = rawBody(req); - const input = + const rawInput = body.length > 0 ? { body, contentType: req.header('content-type') ?? 'application/json' } : undefined; + const input = resolveRunInput(build, rawInput); // `resolveProxyPassword(requireUser(req))` is the *run owner's* proxy password, not just "the // caller's": `actor` was resolved via `resolveOwnedActor(requireUser(req).id, ...)` above, so diff --git a/src/services/builds.ts b/src/services/builds.ts index 98b6089..55319cc 100644 --- a/src/services/builds.ts +++ b/src/services/builds.ts @@ -8,6 +8,7 @@ import { DriverTimedOutError } from '../driver/types.js'; import { normalizeEntryName } from '../driver/tar-entry-name.js'; import { qualifyDockerfileImageReferences } from './dockerfile-image-refs.js'; import { resolveDockerfileLocation } from './dockerfile-location.js'; +import { resolveInputSchemaLocation } from './input-schema-location.js'; import { appendLog, appendRuntimeLog, flushLog, markLogTerminal } from './logs.js'; import { isTerminalJobStatus, transitionJobStatus } from './job-status.js'; @@ -214,6 +215,25 @@ export async function runBuildInBackground( return; } for (const line of dockerfileResolution.logLines) appendRuntimeLog(record.id, line); + + // Resolved before the image is built, not after: an Actor whose declared input contract cannot be + // read (or is itself invalid) fails the build with that reason stated, rather than building an image + // whose every later run would silently skip validation. Same failure shape as the Dockerfile + // resolution above, and, like it, nothing here ever touches the pushed source itself. + const inputSchemaResolution = resolveInputSchemaLocation(version.sourceFiles); + if (inputSchemaResolution.outcome === 'failure') { + appendRuntimeLog(record.id, inputSchemaResolution.message); + await flushLog(record.id); + markLogTerminal(record.id); + await transitionJobStatus(builds, record.id, 'FAILED', { + finishedAt: new Date().toISOString(), + statusMessage: inputSchemaResolution.message, + }); + return; + } + for (const line of inputSchemaResolution.logLines) appendRuntimeLog(record.id, line); + const inputSchema = inputSchemaResolution.outcome === 'resolved' ? inputSchemaResolution.schema : undefined; + const sourceFiles: SourceFile[] = qualifyDockerfileImages( dockerfileResolution.outcome === 'default' ? [...version.sourceFiles, dockerfileResolution.extraSourceFile] @@ -269,6 +289,10 @@ export async function runBuildInBackground( ...(outcome.imageWorkingDirectory !== undefined ? { imageWorkingDirectory: outcome.imageWorkingDirectory } : {}), + // Omitted entirely (rather than written as `undefined`) for an Actor with no input schema, + // same reason as `imageWorkingDirectory` right above: "never present on a non-SUCCEEDED + // build" stays true of the value too, there is simply nothing to record for this build. + ...(inputSchema !== undefined ? { inputSchema } : {}), }); if (succeeded?.status !== 'SUCCEEDED') { await updateActor(actor.id, (current) => { diff --git a/src/services/input-schema-location.ts b/src/services/input-schema-location.ts new file mode 100644 index 0000000..075f91c --- /dev/null +++ b/src/services/input-schema-location.ts @@ -0,0 +1,187 @@ +/** + * Resolves which input schema a build should carry from the version's `sourceFiles` - the build-time + * half of input validation (`actor-driver.md`'s "Input schema, validation and defaults"). The run-time + * half (defaults, validation) lives in `services/input-schema.ts` and reads only the schema this module + * resolved, never the source files again. + * + * Candidate order, stopping at the first hit, matching what `apify-cli` itself looks for locally (its + * `readInputSchema`), which is also what the platform's builder records on the build: + * 1. the `input` field of `.actor/actor.json` - an inline schema object, or a path relative to + * `.actor/` + * 2. `.actor/INPUT_SCHEMA.json` + * 3. `INPUT_SCHEMA.json` at the Actor root + * Matching is case-insensitive with an exact-case win, exactly like `services/dockerfile-location.ts` + * (so `.actor/input_schema.json` and `.actor/INPUT_SCHEMA.json` are the same candidate - which is why + * the CLI's four-entry list collapses to the two locations above here), and every outcome is stated in + * the build log. + * + * `.actor/actor.json` and the schema file alike are parsed as JSON5, matching how this runtime already + * parses `.actor/actor.json` for the Dockerfile lookup; every valid JSON file is also valid JSON5, so + * this only ever accepts more than the platform, never less. + */ +import * as path from 'node:path'; +import JSON5 from 'json5'; + +import { normalizeEntryName } from '../driver/tar-entry-name.js'; +import type { InputSchema, SourceFile } from '../storage/entities.js'; +import { describeInputSchemaDefect } from './input-schema.js'; + +const ACTOR_DIR = '.actor'; +const ACTOR_JSON_NAME = `${ACTOR_DIR}/actor.json`; +/** Checked case-insensitively, so the `INPUT_SCHEMA.json` spelling matches these too. */ +const DEFAULT_SCHEMA_CANDIDATES = [`${ACTOR_DIR}/input_schema.json`, 'input_schema.json'] as const; + +/** Why input-schema resolution failed. Each one fails the build, the same way the Dockerfile + * resolution's own failures do - an Actor whose declared input contract cannot be read is not built + * with that contract silently dropped. */ +export type InputSchemaResolutionFailureReason = + 'escapes-actor-root' | 'invalid-input-field' | 'unparseable-input-schema' | 'invalid-input-schema'; + +/** `resolveInputSchemaLocation`'s outcomes: `resolved` (a candidate matched and the schema is valid), + * `none` (no candidate at all - the Actor simply has no input schema, which is not an error), or + * `failure`. */ +export type InputSchemaResolution = + | { outcome: 'resolved'; schema: InputSchema; source: string; logLines: string[] } + | { outcome: 'none'; logLines: string[] } + | { outcome: 'failure'; reason: InputSchemaResolutionFailureReason; message: string }; + +function sourceFileToText(file: SourceFile): string { + return file.format === 'BASE64' ? Buffer.from(file.content, 'base64').toString('utf8') : file.content; +} + +/** Exact-case match wins; otherwise the first match in `sourceFiles` order - same rule as + * `services/dockerfile-location.ts`'s own lookup. */ +function findCaseInsensitive(sourceFiles: SourceFile[], candidate: string): SourceFile | undefined { + const lowerCandidate = candidate.toLowerCase(); + let firstMatch: SourceFile | undefined; + for (const file of sourceFiles) { + const normalizedName = normalizeEntryName(file.name); + if (normalizedName.toLowerCase() !== lowerCandidate) continue; + if (normalizedName === candidate) return file; + firstMatch ??= file; + } + return firstMatch; +} + +/** `.actor/actor.json`'s own path is not case-folded, unlike the schema candidates. */ +function findExact(sourceFiles: SourceFile[], normalizedTarget: string): SourceFile | undefined { + return sourceFiles.find((file) => normalizeEntryName(file.name) === normalizedTarget); +} + +function escapesActorRootFailure(rawField: string): InputSchemaResolution { + return { + outcome: 'failure', + reason: 'escapes-actor-root', + message: `Input schema path "${rawField}" in .actor/actor.json points outside the Actor root directory.`, + }; +} + +/** Meta-validates a resolved candidate, turning it into the `resolved`/`failure` outcome. Kept in one + * place so an inline `input` object and a schema file are held to exactly the same standard. */ +function acceptSchema(schema: unknown, source: string, logLines: string[]): InputSchemaResolution { + const defect = describeInputSchemaDefect(schema); + if (defect) { + return { + outcome: 'failure', + reason: 'invalid-input-schema', + message: `Input schema from ${source} is not valid: ${defect}`, + }; + } + return { + outcome: 'resolved', + schema: schema as InputSchema, + source, + logLines: [...logLines, `Using the input schema from ${source}.\n`], + }; +} + +export function resolveInputSchemaLocation(sourceFiles: SourceFile[]): InputSchemaResolution { + const logLines: string[] = []; + + const actorJsonFile = findExact(sourceFiles, ACTOR_JSON_NAME); + let actorSpecification: unknown; + if (actorJsonFile) { + try { + actorSpecification = JSON5.parse(sourceFileToText(actorJsonFile)); + } catch { + // Deliberately not a failure of its own: `services/dockerfile-location.ts` runs first on the + // very same file and already fails the build with its own "Could not parse .actor/actor.json" + // message, so reporting it twice (in two different wordings) would only be noise. + actorSpecification = undefined; + } + } + + if (actorSpecification !== null && typeof actorSpecification === 'object' && 'input' in actorSpecification) { + const field: unknown = actorSpecification.input; + + // An inline schema object, the other shape the Actor specification allows for this field. + if (field !== null && typeof field === 'object' && !Array.isArray(field)) { + return acceptSchema(field, 'the "input" field in .actor/actor.json', logLines); + } + + if (typeof field !== 'string') { + return { + outcome: 'failure', + reason: 'invalid-input-field', + message: '.actor/actor.json has invalid format: "input" must be a string or an object.', + }; + } + + if (field === '') { + logLines.push( + 'Warning: "" (from the "input" field in .actor/actor.json) is not in the pushed source; falling back to the default locations.\n', + ); + } else if (field.startsWith('/')) { + return escapesActorRootFailure(field); + } else { + const joined = normalizeEntryName(path.posix.join(ACTOR_DIR, field)); + if (joined === '..' || joined.startsWith('../')) { + return escapesActorRootFailure(field); + } + + const match = findCaseInsensitive(sourceFiles, joined); + if (match) { + const parsed = parseSchemaFile(match); + if (parsed.outcome === 'failure') return parsed; + return acceptSchema( + parsed.schema, + `"${normalizeEntryName(match.name)}" (the "input" field in .actor/actor.json)`, + logLines, + ); + } + + // A value naming no pushed file falls through to the default locations instead of failing - + // the same tolerance `apify-cli` shows locally (it warns and keeps looking), and the same one + // the Dockerfile field already has here. + logLines.push( + `Warning: "${joined}" (from the "input" field in .actor/actor.json) is not in the pushed source; falling back to the default locations.\n`, + ); + } + } + + for (const candidate of DEFAULT_SCHEMA_CANDIDATES) { + const match = findCaseInsensitive(sourceFiles, normalizeEntryName(candidate)); + if (!match) continue; + const parsed = parseSchemaFile(match); + if (parsed.outcome === 'failure') return parsed; + return acceptSchema(parsed.schema, `"${normalizeEntryName(match.name)}"`, logLines); + } + + // No input schema at all is the normal state of an Actor that never declared one - runs of such a + // build take their input exactly as the caller sent it, with nothing validated and nothing added. + return { outcome: 'none', logLines }; +} + +type SchemaFileParse = { outcome: 'parsed'; schema: unknown } | Extract; + +function parseSchemaFile(file: SourceFile): SchemaFileParse { + try { + return { outcome: 'parsed', schema: JSON5.parse(sourceFileToText(file)) as unknown }; + } catch (error) { + return { + outcome: 'failure', + reason: 'unparseable-input-schema', + message: `Could not parse the input schema "${normalizeEntryName(file.name)}": ${(error as Error).message}`, + }; + } +} diff --git a/src/services/input-schema.ts b/src/services/input-schema.ts new file mode 100644 index 0000000..68a98dc --- /dev/null +++ b/src/services/input-schema.ts @@ -0,0 +1,313 @@ +/** + * Input validation and defaults from the Actor's input schema - the run-time half of the feature whose + * build-time half (finding the schema in the pushed source) is `services/input-schema-location.ts`. + * + * Validation itself is the platform's own: `@apify/input_schema`'s `validateInputSchema` (the schema + * against the Apify input-schema meta-schema) and `validateInputUsingValidator` (the input against the + * schema, plus the proxy/`requestListSources`/pattern checks AJV alone cannot express), driven by the + * same AJV configuration the platform's API uses. That is deliberate: a developer must see locally the + * very same message the real API would answer with, not an approximation of it. + * + * Two behaviours the platform has and this runtime deliberately does not: + * - **Proxy group availability.** The platform passes the caller's proxy groups into the validator, + * which then rejects a group the account cannot use. This runtime emulates no proxy groups at all + * (`unsupported.md`), so it passes none: a `proxy` field is still checked for shape, custom proxy + * URLs and country code, but any `apifyProxyGroups` selection is accepted. + * - **Encrypted secret input fields** stay unsupported (`unsupported.md`); a schema may declare + * `isSecret`, and the value is then validated as the plain value it locally is. + */ +import ajvPackage from 'ajv'; +import ajv2019Package from 'ajv/dist/2019.js'; +import { validateInputSchema, validateInputUsingValidator } from '@apify/input_schema'; + +import type { InputSchema } from '../storage/entities.js'; + +// AJV ships as CommonJS, so under this package's ESM resolution its class arrives as the module's +// `default` rather than as the import binding itself. `2019` is the same library built with +// draft-2019-09 support, which the Apify input-schema meta-schema requires and the plain build lacks. +const Ajv = ajvPackage.default; +const Ajv2019 = ajv2019Package.default; +type InputValidator = ReturnType['compile']>; + +/** The Actor input as it travels from the API route into the run: the exact bytes stored as the run's + * `INPUT` record, with the content type they are stored under. */ +export interface ActorInput { + body: Buffer; + contentType: string; +} + +/** Every way `processActorInput` can end, `ok` included - a discriminated union the API route maps to + * its own `ApiError` types, the same split `services/dev-folder.ts` uses. */ +export type InputProcessingResult = + | { kind: 'ok'; input: ActorInput } + | { kind: 'not-json' } + | { kind: 'unparseable-json'; parseError: string } + | { kind: 'not-object'; actualType: string } + | { kind: 'invalid'; message: string } + | { kind: 'invalid-schema'; message: string }; + +/** The messages the real Apify API answers with, word for word (`@apify-packages/errors`'s + * `actor.inputNotJson`/`inputNotValidJson`/`inputNotObject`/`inputNotValid`/`invalidInputSchema`) - + * the whole point of routing validation through the platform's own validator is that the text a + * developer reads locally is the text they would read against the platform. */ +export function describeInputProcessingFailure(result: Exclude): string { + switch (result.kind) { + case 'not-json': + return 'Actor input must have content type "application/json".'; + case 'unparseable-json': + return `Cannot parse input JSON body: ${result.parseError}`; + case 'not-object': + return `The input JSON must be object, got "${result.actualType}" instead.`; + case 'invalid': + return `Input is not valid: ${result.message}`; + case 'invalid-schema': + return `Input schema is not valid: ${result.message}`; + } +} + +/** + * Meta-validates an input schema, returning `null` when it is valid or a human-readable defect when it + * is not. Used at build time (`services/input-schema-location.ts`) so a broken schema is reported where + * the developer is already looking - the build log - rather than silently at every later run. + * + * The meta-schema uses JSON Schema draft 2019-09 features, hence AJV's 2019 build here; the input + * validator below deliberately uses the plain one, exactly as the platform does for each. + */ +export function describeInputSchemaDefect(schema: unknown): string | null { + if (schema === null || typeof schema !== 'object' || Array.isArray(schema)) { + return 'Input schema must be an object.'; + } + try { + // A fresh instance per call: AJV caches every schema it compiles in a Map it never evicts, and + // this runs once per build, not per request. + const ajv = new Ajv2019({ strict: false, unicodeRegExp: false }); + // Cloned because `validateInputSchema` normalizes the object it is handed in place. + validateInputSchema(ajv, structuredClone(schema) as Record); + return null; + } catch (error) { + return (error as Error).message; + } +} + +/** + * Applies the schema's defaults to `input` and validates the result, returning the effective input a + * run should actually be started with. Mirrors the platform's `processInputUsingSchema`: an absent + * input is not "no input" once a schema exists - it is an empty object the defaults are applied to, so + * `apify call` with no input at all still gets the schema's defaults in its `INPUT` record, and a + * required field with no default is still reported missing. + */ +export function processActorInput(input: ActorInput | undefined, schema: InputSchema): InputProcessingResult { + let parsedInput: unknown = {}; + if (input) { + if (!isJsonContentType(input.contentType)) return { kind: 'not-json' }; + try { + parsedInput = JSON.parse(input.body.toString('utf8')) as unknown; + } catch (error) { + return { kind: 'unparseable-json', parseError: (error as Error).message }; + } + // A literal `null` body is the platform's "no input" too, not a type error. + if (parsedInput === null || parsedInput === undefined) parsedInput = {}; + if (!isPlainObject(parsedInput)) { + return { kind: 'not-object', actualType: Array.isArray(parsedInput) ? 'array' : typeof parsedInput }; + } + } + + const withDefaults = mergeDefaultsFromInputSchema(parsedInput as Record, schema); + + let validator; + try { + validator = compileInputSchemaValidator(schema); + } catch (error) { + // Only reachable for a schema that passed the build-time meta-validation but still cannot be + // compiled (or one recorded by an older runtime, before builds validated schemas at all). + return { kind: 'invalid-schema', message: (error as Error).message }; + } + + // No `proxy` options: proxy-group availability is not emulated here (see this module's doc comment). + const validationErrors = validateInputUsingValidator(validator, schema, withDefaults, {}); + if (validationErrors.length > 0) { + // Every message, joined - what the platform returns for an API-origin run (it shortens this to + // the first message only for the web console, which has a single notification line to show it in). + return { kind: 'invalid', message: validationErrors.map(({ message }) => message).join(', ') }; + } + + return { + kind: 'ok', + input: { body: Buffer.from(JSON.stringify(withDefaults), 'utf8'), contentType: DEFAULT_INPUT_CONTENT_TYPE }, + }; +} + +const DEFAULT_INPUT_CONTENT_TYPE = 'application/json'; + +/** `application/json`, with any parameters (`; charset=utf-8`) ignored - the same single media type the + * platform accepts for an Actor input validated against a schema. */ +function isJsonContentType(contentType: string): boolean { + return contentType.split(';')[0]?.trim().toLowerCase() === DEFAULT_INPUT_CONTENT_TYPE; +} + +function isPlainObject(value: unknown): value is Record { + return typeof value === 'object' && value !== null && !Array.isArray(value); +} + +/** + * Merges the schema's default values into `values` and returns a new object - a port of the platform's + * `mergeDefaultsFromInputSchema`, including its recursion into nested objects and into the items of an + * array whose `items` sub-schema carries defaults. + * + * Exported for direct testing; run-start callers go through `processActorInput`. + */ +export function mergeDefaultsFromInputSchema( + values: Record, + schema: InputSchema, +): Record { + const properties = isPlainObject(schema.properties) ? schema.properties : {}; + const defaults: Record = {}; + for (const [key, fieldSchema] of Object.entries(properties)) { + defaults[key] = extractDefaults(fieldSchema); + } + + // Cloned once here so nothing ever assigns a live reference into the schema record itself - a later + // mutation of the returned input (or of the stored schema) can then never reach the other. + const clonedDefaults = structuredClone(defaults); + const target = structuredClone(values); + assignRecursively(target, clonedDefaults, schema); + return target; +} + +/** The `default` of a field, plus - for an object field with `properties` - the defaults of its nested + * fields, with the field's own `default` winning over them key by key. */ +function extractDefaults(fieldSchema: unknown): unknown { + if (!isPlainObject(fieldSchema)) return undefined; + + const rootValue = fieldSchema.default; + const properties = isPlainObject(fieldSchema.properties) ? fieldSchema.properties : undefined; + if (properties) { + const nested: Record = {}; + for (const [key, subSchema] of Object.entries(properties)) { + nested[key] = extractDefaults(subSchema); + } + if (isPlainObject(rootValue)) return deepMerge(nested, rootValue); + if (Object.values(nested).some((value) => value !== undefined)) return nested; + } + return rootValue; +} + +/** The defaults for the items of an array field, from its `items` sub-schema. */ +function extractArrayItemDefaults(fieldSchema: unknown): unknown { + if (!isPlainObject(fieldSchema) || !('items' in fieldSchema)) return undefined; + return extractDefaults(fieldSchema.items); +} + +/** Assigns a default only where the current value is `undefined`, recursing into nested objects and + * into array items. `fillDefinedValues` is set for array items alone, where the platform fills an + * item's fields even though the item itself is present. */ +function assignRecursively( + target: Record, + defaults: Record, + schema: InputSchema | undefined, + fillDefinedValues = false, +): void { + const schemaProperties = isPlainObject(schema?.properties) ? schema.properties : {}; + const keys = new Set([...Object.keys(defaults), ...Object.keys(schemaProperties)]); + + for (const key of keys) { + const defaultValue = defaults[key]; + const currentValue = target[key]; + const fieldSchema = schemaProperties[key]; + + if (currentValue === undefined) { + if (defaultValue !== undefined) target[key] = defaultValue; + continue; + } + + if (isPlainObject(currentValue)) { + const nestedDefaults = isPlainObject(defaultValue) + ? defaultValue + : ((extractDefaults(fieldSchema) as Record | undefined) ?? {}); + assignRecursively( + currentValue, + // Without `fillDefinedValues`, a present nested object keeps exactly the keys it came + // with - only its own sub-schema's defaults for keys it is missing are considered. + fillDefinedValues ? nestedDefaults : {}, + fieldSchema as InputSchema | undefined, + ); + continue; + } + + if (Array.isArray(currentValue)) { + const itemDefaults = extractArrayItemDefaults(fieldSchema); + if (!isPlainObject(itemDefaults)) continue; + const itemSchema = isPlainObject(fieldSchema) ? fieldSchema.items : undefined; + for (const item of currentValue) { + if (!isPlainObject(item)) continue; + assignRecursively(item, itemDefaults, itemSchema as InputSchema | undefined, true); + } + } + } +} + +/** Deep-merges `overrides` onto a copy of `base`, object by object - the one `lodash.merge` behaviour + * `extractDefaults` needs, without the dependency. Arrays and scalars replace wholesale. */ +function deepMerge(base: Record, overrides: Record): Record { + const merged: Record = { ...base }; + for (const [key, value] of Object.entries(overrides)) { + const existing = merged[key]; + merged[key] = isPlainObject(existing) && isPlainObject(value) ? deepMerge(existing, value) : value; + } + return merged; +} + +/** + * Compiled validators, keyed by the schema's serialization - a run start would otherwise recompile the + * same schema on every single call. Bounded and oldest-first evicted, the same reason the platform caps + * its own LRU: a long-lived runtime must not grow a map per schema it has ever seen. + */ +const validatorCache = new Map(); +const VALIDATOR_CACHE_MAX_ENTRIES = 100; + +function compileInputSchemaValidator(schema: InputSchema): InputValidator { + const cacheKey = JSON.stringify(schema); + const cached = validatorCache.get(cacheKey); + if (cached) return cached; + + const validator = new Ajv({ strict: false, unicodeRegExp: false }).compile(prepareSchemaForValidation(schema)); + + if (validatorCache.size >= VALIDATOR_CACHE_MAX_ENTRIES) { + const oldest = validatorCache.keys().next(); + if (!oldest.done) validatorCache.delete(oldest.value); + } + validatorCache.set(cacheKey, validator); + return validator; +} + +/** + * The platform's `getAjvValidator` preparation, ported: a required field that has a default is treated + * as optional (there is always a value for it by the time validation runs), a required array must hold + * at least one item, and `$schema` is dropped because AJV would otherwise try to fetch the Apify + * meta-schema it names and fail to compile the schema at all. + */ +function prepareSchemaForValidation(schema: InputSchema): Record { + const copy = structuredClone(schema) as Record; + const required: string[] = []; + const originalRequired = Array.isArray(schema.required) ? schema.required : []; + const properties = isPlainObject(copy.properties) ? copy.properties : {}; + + for (const [key, fieldSchema] of Object.entries(properties)) { + if (!originalRequired.includes(key)) continue; + if (isPlainObject(fieldSchema) && fieldSchema.default !== undefined) continue; + required.push(key); + if (isPlainObject(fieldSchema) && fieldSchema.type === 'array') { + const minItems = typeof fieldSchema.minItems === 'number' ? fieldSchema.minItems : 0; + fieldSchema.minItems = Math.max(1, minItems); + } + } + + copy.required = required; + delete copy.$schema; + return copy; +} + +/** Test-only: a validator compiled against one test's schema must never be reused by the next. */ +export function resetInputSchemaValidatorCacheForTests(): void { + validatorCache.clear(); +} diff --git a/src/storage/entities.ts b/src/storage/entities.ts index 500c330..f706db1 100644 --- a/src/storage/entities.ts +++ b/src/storage/entities.ts @@ -42,6 +42,18 @@ export interface SourceFile { content: string; } +/** + * An Actor's input schema, exactly as it was pushed (`services/input-schema-location.ts` resolves which + * file or inline object that is). Kept as a loose record rather than a modelled type: it is the + * developer's own document, handed to the platform's own validator verbatim, and the runtime itself + * only ever reads `properties`/`required` out of it. + */ +export interface InputSchema { + [key: string]: unknown; + properties?: Record; + required?: string[]; +} + export interface ActorVersionRecord { versionNumber: string; buildTag: string; @@ -120,6 +132,13 @@ export interface BuildRecord { * inspect failed or the working directory was empty/`/` (mounting over `/` would destroy the * container) - never present on a non-`SUCCEEDED` build. */ imageWorkingDirectory?: string; + /** The input schema resolved from this build's own source files (`services/input-schema-location.ts`), + * recorded in the same status-transition write that records `SUCCEEDED`, exactly like + * `imageWorkingDirectory` above. Build-specific for the same reason that field is: a run validates + * its input against the schema of the build it actually resolved, never against some other tag's + * more recently pushed one. Absent when the Actor declares no input schema at all, and never present + * on a non-`SUCCEEDED` build - such a run then takes its input exactly as the caller sent it. */ + inputSchema?: InputSchema; exitCode?: number; statusMessage?: string; } diff --git a/test/e2e/actor-dev-loop.test.ts b/test/e2e/actor-dev-loop.test.ts index 2675e09..e0c80b0 100644 --- a/test/e2e/actor-dev-loop.test.ts +++ b/test/e2e/actor-dev-loop.test.ts @@ -21,6 +21,7 @@ import { apify, apifyAllOutput, apifyEnv, + apifyExpectingFailure, createIsolatedApifyHome, loginApifyCli, removeIsolatedApifyHome, @@ -133,6 +134,31 @@ describe('full Actor dev loop via apify-cli (requires Docker)', () => { expect(log).toMatch(/Processing/); }); + it("a call with no input at all runs on the input schema's defaults, and one the schema rejects never starts", () => { + const env = apifyEnv(isolatedApifyHome); + const actorDir = join(REPO_ROOT, 'sample_actor_ts'); + + // No `--input`: the Actor's input schema defaults `maxPages` to 2, so the item count is 2 - + // input-dependent, like every other assertion here, just with the input coming from the schema. + const callOutput = apify(['call', '--json'], { cwd: actorDir, env }); + const call = JSON.parse(callOutput) as CallResult; + expect(call.run.status).toBe('SUCCEEDED'); + + const infoOutput = apify(['datasets', 'info', call.storage.defaultDatasetId, '--json'], { + cwd: actorDir, + env, + }); + expect((JSON.parse(infoOutput) as DatasetInfoResult).itemCount).toBe(2); + + // `maxPages` has `minimum: 1` - the run is refused before any container starts, and the CLI + // surfaces the runtime's validation message. + const rejected = apifyExpectingFailure(['call', '--input', JSON.stringify({ maxPages: 0 }), '--json'], { + cwd: actorDir, + env, + }); + expect(rejected).toMatch(/maxPages/); + }); + it('apify api reads back the run and its default dataset (requirements/cli.md: `apify api`)', () => { const env = apifyEnv(isolatedApifyHome); const callOutput = apify(['call', '--input', JSON.stringify({ maxPages: 1 }), '--json'], { diff --git a/test/e2e/helpers/apify-cli.ts b/test/e2e/helpers/apify-cli.ts index 96e5109..91cd0f4 100644 --- a/test/e2e/helpers/apify-cli.ts +++ b/test/e2e/helpers/apify-cli.ts @@ -32,6 +32,24 @@ export function apifyAllOutput(args: string[], options: { cwd: string; env: Node return `${result.stdout}\n${result.stderr}`; } +/** + * Like `apifyAllOutput()`, but for a command that is *expected* to fail: returns its combined output + * instead of throwing, and fails the test if the command unexpectedly succeeded. Used for the run that + * the runtime must refuse (an input its Actor's input schema rejects), where the CLI's non-zero exit is + * itself part of what the test asserts. + */ +export function apifyExpectingFailure(args: string[], options: { cwd: string; env: NodeJS.ProcessEnv }): string { + const result = spawnSync('npx', ['-y', '-p', 'apify-cli', 'apify', ...args], { + cwd: options.cwd, + env: options.env, + encoding: 'utf8', + }); + if (result.status === 0) { + throw new Error(`apify ${args.join(' ')} was expected to fail, but succeeded:\n${result.stdout}`); + } + return `${result.stdout}\n${result.stderr}`; +} + /** Any non-empty value works - the runtime creates a user for this token ad-hoc on its first API * request (`cli.md`'s User bootstrap), the same token throughout this e2e run mapping back to that one * user on every subsequent request. */ diff --git a/test/integration/input-schema.test.ts b/test/integration/input-schema.test.ts new file mode 100644 index 0000000..b27fafc --- /dev/null +++ b/test/integration/input-schema.test.ts @@ -0,0 +1,273 @@ +/** + * Input validation and defaults at run start, over the real HTTP server and a real `apify-client` + * (`actor-driver.md`'s "Input schema, validation and defaults"). Builds are seeded directly, the way + * the rest of the run-start integration tests do, so no Docker daemon is needed: what matters here is + * what the API answers and what lands in the run's `INPUT` record, neither of which involves the driver. + */ +import { afterEach, beforeEach, describe, expect, it } from 'vitest'; + +import { fixedBuildOutcomeDriver, startTestServer, type TestServerHandle } from './helpers/test-server.js'; +import { getRegistries } from '../../src/storage/registries.js'; +import { recordTaggedBuild, updateActor } from '../../src/services/actors.js'; +import { resetInputSchemaValidatorCacheForTests } from '../../src/services/input-schema.js'; +import { runBuildInBackground } from '../../src/services/builds.js'; +import type { InputSchema, SourceFile } from '../../src/storage/entities.js'; + +const SAMPLE_SCHEMA: InputSchema = { + title: 'Sample actor input', + type: 'object', + schemaVersion: 1, + properties: { + startUrl: { + title: 'Start URL', + type: 'string', + editor: 'textfield', + description: 'The page the crawler starts from.', + default: 'https://crawlee.dev/', + }, + maxPages: { + title: 'Max pages', + type: 'integer', + description: 'The maximum number of pages to crawl.', + default: 2, + minimum: 1, + }, + }, + required: [], +}; + +describe('input validation and defaults (via real apify-client)', () => { + let server: TestServerHandle; + + beforeEach(async () => { + server = await startTestServer(); + }); + + afterEach(async () => { + resetInputSchemaValidatorCacheForTests(); + await server.close(); + }); + + /** Seeds a tagged, successful build - with or without an input schema - for `actorId`. */ + async function seedBuild(actorId: string, userId: string, inputSchema?: InputSchema): Promise { + const { builds } = getRegistries(); + const buildId = `seeded${Math.random().toString(36).slice(2, 13)}`.slice(0, 17); + await builds.set(buildId, { + id: buildId, + userId, + actorId, + versionNumber: '0.0', + buildNumber: '0.0.1', + tag: 'latest', + status: 'SUCCEEDED', + startedAt: new Date().toISOString(), + finishedAt: new Date().toISOString(), + imageId: 'fake-image:latest', + ...(inputSchema ? { inputSchema } : {}), + }); + await updateActor(actorId, (current) => recordTaggedBuild(current, 'latest', buildId, '0.0.1')); + } + + async function storedInput(runId: string): Promise { + const run = await server.client.run(runId).get(); + const record = await server.client.keyValueStore(run!.defaultKeyValueStoreId).getRecord('INPUT'); + return record?.value; + } + + it("fills the schema's defaults into the run's INPUT, keeping what the caller sent", async () => { + const actor = await server.client.actors().create({ name: 'defaults-actor' }); + await seedBuild(actor.id, actor.userId, SAMPLE_SCHEMA); + + const run = await server.client.actor(actor.id).start({ maxPages: 7 }); + expect(await storedInput(run.id)).toEqual({ maxPages: 7, startUrl: 'https://crawlee.dev/' }); + }); + + it('writes the defaults even for a run started with no input at all', async () => { + const actor = await server.client.actors().create({ name: 'no-input-actor' }); + await seedBuild(actor.id, actor.userId, SAMPLE_SCHEMA); + + // No body at all on the wire - the raw endpoint, since `apify-client` always sends the object + // it is given. + const response = await fetch(`${server.baseUrl}/v2/actors/${actor.id}/runs?token=${server.token}`, { + method: 'POST', + }); + expect(response.status).toBe(201); + const { data } = (await response.json()) as { data: { id: string } }; + expect(await storedInput(data.id)).toEqual({ startUrl: 'https://crawlee.dev/', maxPages: 2 }); + }); + + it("rejects an input the schema does not accept, with the platform's error type and message", async () => { + const actor = await server.client.actors().create({ name: 'invalid-input-actor' }); + await seedBuild(actor.id, actor.userId, SAMPLE_SCHEMA); + + await expect(server.client.actor(actor.id).start({ maxPages: 0 })).rejects.toMatchObject({ + statusCode: 400, + type: 'invalid-input', + message: 'Input is not valid: Field input.maxPages must be >= 1', + }); + + // Nothing was started: a rejected input never creates a run. + const runs = await server.client.actor(actor.id).runs().list(); + expect(runs.items).toHaveLength(0); + }); + + it('rejects a missing required field that has no default', async () => { + const actor = await server.client.actors().create({ name: 'required-field-actor' }); + await seedBuild(actor.id, actor.userId, { + ...SAMPLE_SCHEMA, + properties: { + ...SAMPLE_SCHEMA.properties, + token: { title: 'Token', type: 'string', editor: 'textfield', description: 'A token.' }, + }, + required: ['token'], + }); + + await expect(server.client.actor(actor.id).start({ maxPages: 1 })).rejects.toMatchObject({ + statusCode: 400, + type: 'invalid-input', + message: 'Input is not valid: Field input.token is required', + }); + }); + + it('rejects a non-object and a non-JSON body against a schema', async () => { + const actor = await server.client.actors().create({ name: 'malformed-input-actor' }); + await seedBuild(actor.id, actor.userId, SAMPLE_SCHEMA); + const url = `${server.baseUrl}/v2/actors/${actor.id}/runs?token=${server.token}`; + + const array = await fetch(url, { + method: 'POST', + headers: { 'content-type': 'application/json' }, + body: '[1,2]', + }); + expect(array.status).toBe(400); + expect(await array.json()).toEqual({ + error: { type: 'invalid-input', message: 'The input JSON must be object, got "array" instead.' }, + }); + + const text = await fetch(url, { + method: 'POST', + headers: { 'content-type': 'text/plain' }, + body: '{"maxPages":1}', + }); + expect(text.status).toBe(400); + expect(await text.json()).toEqual({ + error: { type: 'invalid-input', message: 'Actor input must have content type "application/json".' }, + }); + + const broken = await fetch(url, { + method: 'POST', + headers: { 'content-type': 'application/json' }, + body: '{oops', + }); + expect(broken.status).toBe(400); + expect(((await broken.json()) as { error: { message: string } }).error.message).toContain( + 'Cannot parse input JSON body:', + ); + }); + + it('leaves the input untouched for a build that carries no input schema', async () => { + const actor = await server.client.actors().create({ name: 'schemaless-actor' }); + await seedBuild(actor.id, actor.userId); + + // Not valid against SAMPLE_SCHEMA at all - and accepted, byte for byte, because this build has + // no schema to validate against. + const run = await server.client.actor(actor.id).start({ maxPages: 0, whatever: 'anything' }); + expect(await storedInput(run.id)).toEqual({ maxPages: 0, whatever: 'anything' }); + }); + + it('validates against the schema of the build the run actually resolved, not the newest one', async () => { + const actor = await server.client.actors().create({ name: 'two-tag-actor' }); + const { builds } = getRegistries(); + + // `latest` demands maxPages >= 1; `beta`, built afterwards, has no schema at all. + await seedBuild(actor.id, actor.userId, SAMPLE_SCHEMA); + const betaBuildId = 'betaBuildId12345b'; + await builds.set(betaBuildId, { + id: betaBuildId, + userId: actor.userId, + actorId: actor.id, + versionNumber: '0.1', + buildNumber: '0.1.1', + tag: 'beta', + status: 'SUCCEEDED', + startedAt: new Date().toISOString(), + finishedAt: new Date().toISOString(), + imageId: 'fake-image:beta', + }); + await updateActor(actor.id, (current) => recordTaggedBuild(current, 'beta', betaBuildId, '0.1.1')); + + await expect(server.client.actor(actor.id).start({ maxPages: 0 })).rejects.toMatchObject({ + statusCode: 400, + type: 'invalid-input', + }); + + const betaRun = await server.client.actor(actor.id).start({ maxPages: 0 }, { build: 'beta' }); + expect(await storedInput(betaRun.id)).toEqual({ maxPages: 0 }); + }); + + it('records the schema found in the pushed source on the build, and fails the build on a broken one', async () => { + const actor = await server.client.actors().create({ name: 'building-actor' }); + const { actors, builds } = getRegistries(); + + /** Runs a real build (driver stubbed out at the `docker build` itself) over `sourceFiles`, so the + * schema really is resolved from the pushed source the way a pushed Actor's would be. */ + const build = async (buildId: string, sourceFiles: SourceFile[]): Promise => { + await builds.set(buildId, { + id: buildId, + userId: actor.userId, + actorId: actor.id, + versionNumber: '0.0', + buildNumber: '0.0.1', + tag: 'latest', + status: 'READY', + startedAt: new Date().toISOString(), + }); + await runBuildInBackground( + fixedBuildOutcomeDriver({ imageId: 'built-image:latest' }), + (await actors.get(actor.id))!, + { versionNumber: '0.0', buildTag: 'latest', sourceType: 'SOURCE_FILES', sourceFiles }, + (await builds.get(buildId))!, + { tag: 'latest', useCache: true }, + ); + }; + + const goodBuildId = 'goodBuildId12345g'; + await build(goodBuildId, [ + { name: 'main.js', format: 'TEXT', content: 'console.log(1)' }, + { name: '.actor/input_schema.json', format: 'TEXT', content: JSON.stringify(SAMPLE_SCHEMA) }, + ]); + + expect((await builds.get(goodBuildId))?.status).toBe('SUCCEEDED'); + expect((await builds.get(goodBuildId))?.inputSchema).toEqual(SAMPLE_SCHEMA); + expect(await server.client.log(goodBuildId).get()).toContain('Using the input schema from'); + + // End to end from that build: a run against it is validated, and gets the defaults. + await expect(server.client.actor(actor.id).start({ maxPages: 0 })).rejects.toMatchObject({ + statusCode: 400, + type: 'invalid-input', + }); + const run = await server.client.actor(actor.id).start({ maxPages: 5 }); + expect(await storedInput(run.id)).toEqual({ maxPages: 5, startUrl: 'https://crawlee.dev/' }); + + // A build whose schema is invalid fails, with the reason in both its status message and its log - + // rather than producing an image whose every later run would silently skip validation. + const brokenBuildId = 'brokenBuildId123b'; + await build(brokenBuildId, [ + { + name: '.actor/input_schema.json', + format: 'TEXT', + content: JSON.stringify({ + title: 'Missing a description', + type: 'object', + schemaVersion: 1, + properties: { maxPages: { title: 'Max pages', type: 'integer' } }, + }), + }, + ]); + const broken = await server.client.build(brokenBuildId).get(); + expect(broken?.status).toBe('FAILED'); + expect(broken?.statusMessage).toContain('is not valid'); + expect(await server.client.log(brokenBuildId).get()).toContain('.actor/input_schema.json'); + expect((await builds.get(brokenBuildId))?.inputSchema).toBeUndefined(); + }); +}); diff --git a/test/unit/input-schema-location.test.ts b/test/unit/input-schema-location.test.ts new file mode 100644 index 0000000..07aa738 --- /dev/null +++ b/test/unit/input-schema-location.test.ts @@ -0,0 +1,200 @@ +import { describe, expect, it } from 'vitest'; + +import { resolveInputSchemaLocation } from '../../src/services/input-schema-location.js'; +import type { SourceFile } from '../../src/storage/entities.js'; + +/** A minimal but genuinely valid input schema - every field the Apify meta-schema demands is present, + * so a test that is not about schema validity never trips over it. */ +function validSchema(extra: Record = {}): Record { + return { + title: 'Sample input', + type: 'object', + schemaVersion: 1, + properties: { + maxPages: { title: 'Max pages', type: 'integer', description: 'How many pages.', default: 2 }, + }, + ...extra, + }; +} + +function text(name: string, content: string): SourceFile { + return { name, format: 'TEXT', content }; +} + +function json(name: string, value: unknown): SourceFile { + return text(name, JSON.stringify(value)); +} + +describe('resolveInputSchemaLocation', () => { + it('takes an inline schema object from the "input" field of .actor/actor.json', () => { + const resolution = resolveInputSchemaLocation([ + json('.actor/actor.json', { actorSpecification: 1, name: 'a', input: validSchema() }), + json('.actor/input_schema.json', validSchema({ title: 'The file, which must lose' })), + ]); + + expect(resolution.outcome).toBe('resolved'); + if (resolution.outcome !== 'resolved') return; + expect(resolution.schema.title).toBe('Sample input'); + expect(resolution.source).toContain('"input" field'); + expect(resolution.logLines.join('')).toContain('Using the input schema from'); + }); + + it('follows a path in the "input" field, resolved relative to .actor/', () => { + const resolution = resolveInputSchemaLocation([ + json('.actor/actor.json', { actorSpecification: 1, input: './schemas/custom.json' }), + json('.actor/schemas/custom.json', validSchema({ title: 'From the named file' })), + ]); + + expect(resolution.outcome).toBe('resolved'); + if (resolution.outcome !== 'resolved') return; + expect(resolution.schema.title).toBe('From the named file'); + expect(resolution.source).toContain('.actor/schemas/custom.json'); + }); + + it('falls back to the default locations, with a warning, when the named file is not in the pushed source', () => { + const resolution = resolveInputSchemaLocation([ + json('.actor/actor.json', { actorSpecification: 1, input: './missing.json' }), + json('.actor/input_schema.json', validSchema({ title: 'The fallback' })), + ]); + + expect(resolution.outcome).toBe('resolved'); + if (resolution.outcome !== 'resolved') return; + expect(resolution.schema.title).toBe('The fallback'); + expect(resolution.logLines.join('')).toContain('is not in the pushed source'); + }); + + it('treats an empty "input" field as "not found", falling through with a warning', () => { + const resolution = resolveInputSchemaLocation([ + json('.actor/actor.json', { actorSpecification: 1, input: '' }), + json('input_schema.json', validSchema({ title: 'Root fallback' })), + ]); + + expect(resolution.outcome).toBe('resolved'); + if (resolution.outcome !== 'resolved') return; + expect(resolution.schema.title).toBe('Root fallback'); + expect(resolution.logLines.join('')).toContain('falling back to the default locations'); + }); + + it('rejects an "input" path that escapes the Actor root, absolute or relative', () => { + for (const field of ['/etc/passwd', '../../outside.json']) { + const resolution = resolveInputSchemaLocation([ + json('.actor/actor.json', { actorSpecification: 1, input: field }), + ]); + expect(resolution.outcome).toBe('failure'); + if (resolution.outcome !== 'failure') return; + expect(resolution.reason).toBe('escapes-actor-root'); + expect(resolution.message).toContain('points outside the Actor root directory'); + } + }); + + it('rejects an "input" field that is neither a string nor an object', () => { + const resolution = resolveInputSchemaLocation([ + json('.actor/actor.json', { actorSpecification: 1, input: 42 }), + ]); + + expect(resolution.outcome).toBe('failure'); + if (resolution.outcome !== 'failure') return; + expect(resolution.reason).toBe('invalid-input-field'); + expect(resolution.message).toBe('.actor/actor.json has invalid format: "input" must be a string or an object.'); + }); + + it('prefers .actor/INPUT_SCHEMA.json over the Actor root one, matching case-insensitively', () => { + const resolution = resolveInputSchemaLocation([ + json('INPUT_SCHEMA.json', validSchema({ title: 'Root' })), + json('.actor/INPUT_SCHEMA.json', validSchema({ title: 'Actor dir' })), + ]); + + expect(resolution.outcome).toBe('resolved'); + if (resolution.outcome !== 'resolved') return; + expect(resolution.schema.title).toBe('Actor dir'); + expect(resolution.source).toContain('.actor/INPUT_SCHEMA.json'); + }); + + it('reads a BASE64-encoded schema file the same as a TEXT one', () => { + const resolution = resolveInputSchemaLocation([ + { + name: '.actor/input_schema.json', + format: 'BASE64', + content: Buffer.from(JSON.stringify(validSchema({ title: 'Encoded' })), 'utf8').toString('base64'), + }, + ]); + + expect(resolution.outcome).toBe('resolved'); + if (resolution.outcome !== 'resolved') return; + expect(resolution.schema.title).toBe('Encoded'); + }); + + it('reports no schema at all - not a failure - when the source declares none', () => { + const resolution = resolveInputSchemaLocation([text('main.js', 'console.log(1)')]); + expect(resolution.outcome).toBe('none'); + }); + + it('fails on a schema file that cannot be parsed', () => { + const resolution = resolveInputSchemaLocation([text('.actor/input_schema.json', '{ "title": ')]); + + expect(resolution.outcome).toBe('failure'); + if (resolution.outcome !== 'failure') return; + expect(resolution.reason).toBe('unparseable-input-schema'); + expect(resolution.message).toContain('Could not parse the input schema ".actor/input_schema.json"'); + }); + + it("fails on a schema the platform's own meta-schema rejects, naming where it came from", () => { + const resolution = resolveInputSchemaLocation([ + json('.actor/input_schema.json', { + title: 'Missing a field description', + type: 'object', + schemaVersion: 1, + properties: { maxPages: { title: 'Max pages', type: 'integer' } }, + }), + ]); + + expect(resolution.outcome).toBe('failure'); + if (resolution.outcome !== 'failure') return; + expect(resolution.reason).toBe('invalid-input-schema'); + expect(resolution.message).toContain('.actor/input_schema.json'); + expect(resolution.message).toContain('is not valid'); + }); + + it('rejects a schema that is not an object at all', () => { + const resolution = resolveInputSchemaLocation([json('.actor/input_schema.json', ['not', 'a', 'schema'])]); + + expect(resolution.outcome).toBe('failure'); + if (resolution.outcome !== 'failure') return; + expect(resolution.reason).toBe('invalid-input-schema'); + }); + + it('leaves an unparseable .actor/actor.json to the Dockerfile resolver, falling through to the default locations', () => { + // `services/dockerfile-location.ts` runs first on the same file and already fails that build with + // its own message; this must not add a second, differently-worded report of the same defect. + const resolution = resolveInputSchemaLocation([ + text('.actor/actor.json', '{ not json at all'), + json('.actor/input_schema.json', validSchema({ title: 'Still found' })), + ]); + + expect(resolution.outcome).toBe('resolved'); + if (resolution.outcome !== 'resolved') return; + expect(resolution.schema.title).toBe('Still found'); + }); + + it('parses .actor/actor.json as JSON5, like the Dockerfile lookup already does', () => { + const resolution = resolveInputSchemaLocation([ + text('.actor/actor.json', "{ actorSpecification: 1, input: './input_schema.json' /* a comment */ }"), + json('.actor/input_schema.json', validSchema({ title: 'Via JSON5' })), + ]); + + expect(resolution.outcome).toBe('resolved'); + if (resolution.outcome !== 'resolved') return; + expect(resolution.schema.title).toBe('Via JSON5'); + }); + + it('accepts the $schema field real Actor templates carry', () => { + const resolution = resolveInputSchemaLocation([ + json( + '.actor/input_schema.json', + validSchema({ $schema: 'https://apify.com/schemas/v1/input.ide.json', title: 'Templated' }), + ), + ]); + + expect(resolution.outcome).toBe('resolved'); + }); +}); diff --git a/test/unit/input-schema.test.ts b/test/unit/input-schema.test.ts new file mode 100644 index 0000000..a583a68 --- /dev/null +++ b/test/unit/input-schema.test.ts @@ -0,0 +1,375 @@ +import { afterEach, describe, expect, it } from 'vitest'; + +import { + describeInputProcessingFailure, + describeInputSchemaDefect, + mergeDefaultsFromInputSchema, + processActorInput, + resetInputSchemaValidatorCacheForTests, +} from '../../src/services/input-schema.js'; +import type { InputSchema } from '../../src/storage/entities.js'; + +const SAMPLE_SCHEMA: InputSchema = { + title: 'Sample actor input', + type: 'object', + schemaVersion: 1, + properties: { + startUrl: { + title: 'Start URL', + type: 'string', + editor: 'textfield', + description: 'The page the crawler starts from.', + default: 'https://crawlee.dev/', + }, + maxPages: { + title: 'Max pages', + type: 'integer', + description: 'The maximum number of pages to crawl.', + default: 2, + minimum: 1, + }, + label: { title: 'Label', type: 'string', editor: 'textfield', description: 'A free-form label.' }, + }, + required: ['startUrl'], +}; + +function jsonInput(value: unknown): { body: Buffer; contentType: string } { + return { body: Buffer.from(JSON.stringify(value), 'utf8'), contentType: 'application/json' }; +} + +function effectiveInput(result: ReturnType): Record { + if (result.kind !== 'ok') throw new Error(`expected an accepted input, got ${result.kind}`); + expect(result.input.contentType).toBe('application/json'); + return JSON.parse(result.input.body.toString('utf8')) as Record; +} + +afterEach(() => { + resetInputSchemaValidatorCacheForTests(); +}); + +describe('processActorInput - defaults', () => { + it('fills every missing field from the schema and leaves provided ones alone', () => { + expect(effectiveInput(processActorInput(jsonInput({ maxPages: 7 }), SAMPLE_SCHEMA))).toEqual({ + maxPages: 7, + startUrl: 'https://crawlee.dev/', + }); + }); + + it('turns "no input at all" into the schema\'s defaults, the way the platform does', () => { + expect(effectiveInput(processActorInput(undefined, SAMPLE_SCHEMA))).toEqual({ + startUrl: 'https://crawlee.dev/', + maxPages: 2, + }); + }); + + it('treats a literal null body as no input, not as a type error', () => { + const result = processActorInput( + { body: Buffer.from('null', 'utf8'), contentType: 'application/json' }, + SAMPLE_SCHEMA, + ); + expect(effectiveInput(result)).toEqual({ startUrl: 'https://crawlee.dev/', maxPages: 2 }); + }); + + it("keeps a provided value even when it equals the type's zero value", () => { + const schema: InputSchema = { + title: 'Flags', + type: 'object', + schemaVersion: 1, + properties: { + enabled: { title: 'Enabled', type: 'boolean', description: 'A flag.', default: true }, + count: { title: 'Count', type: 'integer', description: 'A count.', default: 5 }, + }, + }; + expect(effectiveInput(processActorInput(jsonInput({ enabled: false, count: 0 }), schema))).toEqual({ + enabled: false, + count: 0, + }); + }); + + it('accepts application/json with parameters, such as a charset', () => { + const result = processActorInput( + { body: Buffer.from('{"maxPages":3}', 'utf8'), contentType: 'application/json; charset=utf-8' }, + SAMPLE_SCHEMA, + ); + expect(effectiveInput(result)).toMatchObject({ maxPages: 3 }); + }); +}); + +describe('mergeDefaultsFromInputSchema', () => { + const NESTED_SCHEMA: InputSchema = { + title: 'Nested', + type: 'object', + schemaVersion: 1, + properties: { + config: { + title: 'Config', + type: 'object', + editor: 'json', + description: 'A nested object.', + properties: { + retries: { title: 'Retries', type: 'integer', description: 'How many.', default: 3 }, + verbose: { title: 'Verbose', type: 'boolean', description: 'Chatty?', default: false }, + }, + }, + items: { + title: 'Items', + type: 'array', + editor: 'json', + description: 'A list of objects.', + items: { type: 'object', properties: { keep: { type: 'boolean', default: true } } }, + }, + }, + }; + + it("builds a nested object out of its fields' defaults when the whole object is missing", () => { + expect(mergeDefaultsFromInputSchema({}, NESTED_SCHEMA)).toEqual({ + config: { retries: 3, verbose: false }, + }); + }); + + it('leaves a provided nested object exactly as it came, adding no keys to it', () => { + expect(mergeDefaultsFromInputSchema({ config: { retries: 9 } }, NESTED_SCHEMA)).toEqual({ + config: { retries: 9 }, + }); + }); + + it('fills defaults into every item of a provided array, without overwriting what the item has', () => { + expect(mergeDefaultsFromInputSchema({ items: [{}, { keep: false }] }, NESTED_SCHEMA)).toEqual({ + config: { retries: 3, verbose: false }, + items: [{ keep: true }, { keep: false }], + }); + }); + + it('never hands out a reference into the schema, so a later edit of the input cannot corrupt it', () => { + const merged = mergeDefaultsFromInputSchema({}, NESTED_SCHEMA) as { config: Record }; + merged.config.retries = 999; + expect(mergeDefaultsFromInputSchema({}, NESTED_SCHEMA)).toEqual({ config: { retries: 3, verbose: false } }); + }); + + it("applies the field's own default over the defaults of its nested fields", () => { + const schema: InputSchema = { + title: 'Merged', + type: 'object', + schemaVersion: 1, + properties: { + config: { + title: 'Config', + type: 'object', + editor: 'json', + description: 'A nested object with its own default.', + default: { retries: 10 }, + properties: { + retries: { title: 'Retries', type: 'integer', description: 'How many.', default: 3 }, + verbose: { title: 'Verbose', type: 'boolean', description: 'Chatty?', default: false }, + }, + }, + }, + }; + expect(mergeDefaultsFromInputSchema({}, schema)).toEqual({ config: { retries: 10, verbose: false } }); + }); +}); + +describe('processActorInput - validation', () => { + it("reports the platform's message for a value below a minimum", () => { + const result = processActorInput(jsonInput({ maxPages: 0 }), SAMPLE_SCHEMA); + expect(result).toEqual({ kind: 'invalid', message: 'Field input.maxPages must be >= 1' }); + expect(describeInputProcessingFailure(result as never)).toBe( + 'Input is not valid: Field input.maxPages must be >= 1', + ); + }); + + it("reports the platform's message for a value of the wrong type", () => { + expect(processActorInput(jsonInput({ label: 5 }), SAMPLE_SCHEMA)).toEqual({ + kind: 'invalid', + message: 'Field input.label must be string', + }); + }); + + it('treats a required field that has a default as satisfied by that default', () => { + // `startUrl` is required and has a default - the platform relaxes exactly this case, since there + // is always a value for such a field by the time validation runs. + expect(effectiveInput(processActorInput(jsonInput({}), SAMPLE_SCHEMA))).toMatchObject({ + startUrl: 'https://crawlee.dev/', + }); + }); + + it('still reports a required field that has no default, even for a run started with no input', () => { + const schema: InputSchema = { + ...SAMPLE_SCHEMA, + required: ['startUrl', 'label'], + }; + expect(processActorInput(undefined, schema)).toEqual({ + kind: 'invalid', + message: 'Field input.label is required', + }); + }); + + it('joins every validation error, the way the API-origin platform response does', () => { + // Two errors of different provenance: the first from AJV (which, configured as the platform + // configures it, reports one at a time), the second from the schema-aware checks that run after it. + const schema: InputSchema = { + title: 'Two problems', + type: 'object', + schemaVersion: 1, + properties: { + a: { title: 'A', type: 'integer', description: 'A number.', minimum: 10 }, + startUrls: { + title: 'Start URLs', + type: 'array', + editor: 'requestListSources', + description: 'Where to start.', + }, + }, + }; + const result = processActorInput(jsonInput({ a: 1, startUrls: [{ url: 'not a url' }] }), schema); + expect(result.kind).toBe('invalid'); + if (result.kind !== 'invalid') return; + expect(result.message).toContain('Field input.a must be >= 10'); + expect(result.message).toContain('input.startUrls'); + expect(result.message).toContain(', '); + }); + + it('requires at least one item in a required array field', () => { + const schema: InputSchema = { + title: 'Sources', + type: 'object', + schemaVersion: 1, + properties: { + startUrls: { + title: 'Start URLs', + type: 'array', + editor: 'requestListSources', + description: 'Where to start.', + }, + }, + required: ['startUrls'], + }; + expect(processActorInput(jsonInput({ startUrls: [] }), schema)).toMatchObject({ kind: 'invalid' }); + expect( + effectiveInput(processActorInput(jsonInput({ startUrls: [{ url: 'https://crawlee.dev/' }] }), schema)), + ).toEqual({ startUrls: [{ url: 'https://crawlee.dev/' }] }); + }); + + it('runs the checks AJV alone cannot express, such as requestListSources entries', () => { + const schema: InputSchema = { + title: 'Sources', + type: 'object', + schemaVersion: 1, + properties: { + startUrls: { + title: 'Start URLs', + type: 'array', + editor: 'requestListSources', + description: 'Where to start.', + }, + }, + }; + expect(processActorInput(jsonInput({ startUrls: [{ url: 'not a url' }] }), schema)).toMatchObject({ + kind: 'invalid', + }); + }); + + it('accepts any apifyProxyGroups selection, since proxy groups are not emulated locally', () => { + const schema: InputSchema = { + title: 'Proxy', + type: 'object', + schemaVersion: 1, + properties: { + proxyConfiguration: { + title: 'Proxy configuration', + type: 'object', + editor: 'proxy', + description: 'Proxy settings.', + default: { useApifyProxy: true, apifyProxyGroups: ['RESIDENTIAL'] }, + }, + }, + }; + expect(effectiveInput(processActorInput(undefined, schema))).toEqual({ + proxyConfiguration: { useApifyProxy: true, apifyProxyGroups: ['RESIDENTIAL'] }, + }); + // Shape checks still apply: a custom proxy URL that is not one is still rejected. + expect( + processActorInput( + jsonInput({ proxyConfiguration: { useApifyProxy: false, proxyUrls: ['nonsense'] } }), + schema, + ), + ).toMatchObject({ kind: 'invalid' }); + }); + + it('compiles a schema carrying the $schema field templates ship with', () => { + const schema: InputSchema = { ...SAMPLE_SCHEMA, $schema: 'https://apify.com/schemas/v1/input.ide.json' }; + expect(effectiveInput(processActorInput(jsonInput({ maxPages: 4 }), schema))).toMatchObject({ maxPages: 4 }); + }); +}); + +describe('processActorInput - malformed requests', () => { + it('rejects a body that is not sent as JSON', () => { + const result = processActorInput({ body: Buffer.from('{}', 'utf8'), contentType: 'text/plain' }, SAMPLE_SCHEMA); + expect(result).toEqual({ kind: 'not-json' }); + expect(describeInputProcessingFailure(result as never)).toBe( + 'Actor input must have content type "application/json".', + ); + }); + + it('rejects a body that is not parseable JSON', () => { + const result = processActorInput( + { body: Buffer.from('{oops', 'utf8'), contentType: 'application/json' }, + SAMPLE_SCHEMA, + ); + expect(result.kind).toBe('unparseable-json'); + expect(describeInputProcessingFailure(result as never)).toContain('Cannot parse input JSON body:'); + }); + + it('rejects a JSON body that is not an object, naming what it got instead', () => { + const array = processActorInput(jsonInput([1, 2]), SAMPLE_SCHEMA); + expect(array).toEqual({ kind: 'not-object', actualType: 'array' }); + expect(describeInputProcessingFailure(array as never)).toBe( + 'The input JSON must be object, got "array" instead.', + ); + + expect(processActorInput(jsonInput('a string'), SAMPLE_SCHEMA)).toEqual({ + kind: 'not-object', + actualType: 'string', + }); + }); + + it('reports a schema that cannot be compiled at all, rather than throwing', () => { + const broken: InputSchema = { title: 'Broken', type: 'object', properties: { a: { type: 'not-a-type' } } }; + const result = processActorInput(jsonInput({}), broken); + expect(result.kind).toBe('invalid-schema'); + expect(describeInputProcessingFailure(result as never)).toContain('Input schema is not valid:'); + }); +}); + +describe('describeInputSchemaDefect', () => { + it('accepts a valid schema', () => { + expect(describeInputSchemaDefect(SAMPLE_SCHEMA)).toBeNull(); + }); + + it('names the defect of an invalid one', () => { + const defect = describeInputSchemaDefect({ + title: 'No description', + type: 'object', + schemaVersion: 1, + properties: { a: { title: 'A', type: 'string', editor: 'textfield' } }, + }); + expect(defect).toContain('description'); + }); + + it('rejects a required field that no property defines', () => { + const defect = describeInputSchemaDefect({ + title: 'Ghost', + type: 'object', + schemaVersion: 1, + properties: {}, + required: ['ghost'], + }); + expect(defect).toContain('ghost'); + }); + + it('rejects anything that is not an object', () => { + expect(describeInputSchemaDefect(['a'])).toBe('Input schema must be an object.'); + expect(describeInputSchemaDefect('a schema')).toBe('Input schema must be an object.'); + expect(describeInputSchemaDefect(null)).toBe('Input schema must be an object.'); + }); +}); From f6b92eb371be5ab4e5a2285b9ebda603279baded Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 22 Sep 2026 15:30:36 +0000 Subject: [PATCH 02/14] docs: Tighten input-schema docs and comments to the new conventions Master's contribution conventions landed after this branch's feature commit: requirements describe what, not how, and comments carry the missing context rather than restating the code. Applied to the input-schema spec sections and the modules they describe; no behaviour change. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_011U9yUTUn78bqGDn5obYSXs --- requirements/actor-driver.md | 58 +++++++-------- requirements/api.md | 15 ++-- src/api/routes/actors.ts | 12 ++-- src/services/builds.ts | 11 ++- src/services/input-schema-location.ts | 44 ++++-------- src/services/input-schema.ts | 100 ++++++++++---------------- src/storage/entities.ts | 17 ++--- 7 files changed, 99 insertions(+), 158 deletions(-) diff --git a/requirements/actor-driver.md b/requirements/actor-driver.md index f620635..a2e1213 100644 --- a/requirements/actor-driver.md +++ b/requirements/actor-driver.md @@ -36,41 +36,31 @@ # Input schema, validation and defaults -- **A build records the input schema found in the Actor's pushed source**, and every run started against - that build has its input validated against it and the schema's defaults applied. A build that has no - input schema runs every input through untouched, exactly as the caller sent it. -- Resolution order at build time, stopping at the first hit, matching what `apify-cli` looks for - locally: - 1. the `input` field of `.actor/actor.json` - either an inline schema object, or a path relative to - `.actor/`; a path escaping the Actor root fails the build with "points outside the Actor root - directory", a value that is neither a string nor an object fails with `"input" must be a string -or an object`, and a path naming no pushed file falls through instead of failing. - 2. `.actor/INPUT_SCHEMA.json` - 3. `INPUT_SCHEMA.json` at the Actor root - - Matching is case-insensitive, exact-case wins ties, and every outcome is stated in the build log. - Schema files are parsed as JSON5, like `.actor/actor.json` already is. -- **A schema that is itself invalid fails the build**, with the defect in both the build log and the - build's status message - rather than producing an image whose every later run silently skips - validation. Validity is judged by the Apify input-schema meta-schema, the platform's own. -- **At run start** (`POST /v2/actors/:actorId/runs`), for a build that has a schema: - - The input must be `application/json` and must parse to a JSON object; `null` counts as no input. - - The schema's defaults fill in every field the caller left out, nested objects and the items of - arrays included, and the result is what is stored as the run's `INPUT` record. A run started with - no input at all therefore still gets an `INPUT` record holding the defaults. - - A field that is `required` but has a `default` is satisfied by that default; a required array must - hold at least one item. - - An input the schema rejects answers `400` and starts nothing - no run record, no container - (`api.md`). - - The schema used is that of the build the run actually resolved, never another tag's more recently - pushed one. -- **Known differences from the platform**: proxy group availability is not checked, since this runtime - emulates no proxy groups - a `proxy` field is still validated for shape, custom proxy URLs and - country code, but any `apifyProxyGroups` selection is accepted. Encrypted secret input fields stay - unsupported (`unsupported.md`), so a value for an `isSecret` field is validated as the plain value it +- **An Actor's input schema is read at build time**, and every run of that build has its input + validated against it with the schema's defaults applied. A build with no input schema takes every + input exactly as the caller sent it. +- The schema is looked for where `apify-cli` looks for it locally, stopping at the first hit: the + `input` field of `.actor/actor.json` (an inline schema, or a path relative to `.actor/`), then + `.actor/INPUT_SCHEMA.json`, then `INPUT_SCHEMA.json` at the Actor root. Matching is case-insensitive, + exact-case wins ties, and every outcome is stated in the build log. +- **A build fails** when the `input` field names a path outside the Actor root or a value that is + neither a path nor a schema, when the schema cannot be parsed, or when it is not a valid Apify input + schema - with the reason in the build log and the build's status message. A path naming no pushed + file only warns, and the default locations are tried instead. +- **A run with a rejected input never starts**: no run record, no container, and a `400` naming the + offending field (`api.md`). +- **Defaults are filled in** for every field the input leaves out, nested objects and array items + included, and the run's `INPUT` record holds the result - so a run started with no input at all + still runs on the schema's defaults. A required field that has a default is satisfied by it; a + required array must hold at least one item. +- A run is validated against the schema of the build it resolved, never another tag's. An Actor run + from a registered dev folder therefore uses its last build's schema: unlike a source edit, an edited + input schema needs an `apify push` to take effect. +- **Known differences from the platform**: Apify Proxy group availability is not checked, since no + proxy groups are emulated here - any `apifyProxyGroups` selection is accepted, while a `proxy` + field's shape, custom proxy URLs and country code still are. Encrypted secret input fields stay + unsupported (`unsupported.md`), so an `isSecret` field's value is validated as the plain value it locally is. -- The schema comes from the **build**, so an Actor running from a registered dev folder validates - against the schema of its last build: editing `INPUT_SCHEMA.json` locally needs an `apify push` to - take effect, unlike a source edit. # Bind mount volumes with Actor source code diff --git a/requirements/api.md b/requirements/api.md index 33b5d2a..a8135d2 100644 --- a/requirements/api.md +++ b/requirements/api.md @@ -17,15 +17,12 @@ - `DELETE /v2/actor-builds/:buildId` and `DELETE /v2/actor-runs/:runId` on a **non-terminal** build/run are rejected, not aborted-then-deleted: `400` with error type `deleting-unfinished-build` (builds) or `cannot-remove-running-run` (runs), matching the Apify platform. -- `POST /v2/actors/:actorId/runs` validates the input against the resolved build's input schema, when it - has one (`actor-driver.md`'s "Input schema, validation and defaults"), and answers `400` with the - platform's own error types and messages when it does not pass - starting nothing at all: - - `invalid-input` - `Actor input must have content type "application/json".`, `Cannot parse input -JSON body: ...`, `The input JSON must be object, got "" instead.`, or `Input is not valid: -`. - - `invalid-input-schema` - `Input schema is not valid: ...`, for a stored schema that cannot be - compiled at all (a build made by an older runtime, before builds validated their schemas). - - A build with no input schema accepts any body, unvalidated, exactly as before. +- `POST /v2/actors/:actorId/runs` validates the input against the resolved build's input schema, when + it has one (`actor-driver.md`), and rejects it with `400` and the platform's own error types and + messages, starting nothing: `invalid-input` for a body that is not `application/json`, is not + parseable JSON, is not a JSON object, or that the schema rejects (every validation error, + comma-separated); `invalid-input-schema` for a stored schema that cannot be compiled at all. A build + with no input schema accepts any body, unvalidated. - Three endpoints are exceptions to the `{data}` envelope: - `GET /v2/logs/:buildOrRunId` (and its `actor-builds`/`actor-runs` aliases): the body is plain text, never `{data}`-wrapped, matching apify-client-js's `log().get()`. diff --git a/src/api/routes/actors.ts b/src/api/routes/actors.ts index 387c012..4da508a 100644 --- a/src/api/routes/actors.ts +++ b/src/api/routes/actors.ts @@ -230,20 +230,16 @@ export function mountActors(router: Router, deps: ApiServerDeps): void { }), ); - /** Maps a non-`ok` `InputProcessingResult` to the `ApiError` this route throws - the HTTP status and - * error `type` are the route's concern, the message text the service's (`services/input-schema.ts`), - * the same split `api/routes/dev-folder.ts` already uses. Both types, and both messages, are the - * real platform's for the same input (`@apify-packages/errors`'s `actor.inputNotValid` and friends). */ + /** Status and error `type` are the route's concern, the message text the service's - the split + * `api/routes/dev-folder.ts` already uses. Both are the real platform's for the same input. */ function inputApiError(result: Exclude): ApiError { const message = describeInputProcessingFailure(result); const type = result.kind === 'invalid-schema' ? 'invalid-input-schema' : 'invalid-input'; return new ApiError(400, type, message); } - /** The run's effective input: the caller's bytes as-is for a build with no input schema, or - for one - * that has a schema - the input with the schema's defaults applied, validated, and re-serialized - * (`services/input-schema.ts`). A schema turns "no input at all" into "the defaults", which is why - * this can return an input for a request that carried no body. */ + /** The caller's bytes as-is for a build with no input schema; otherwise the validated input with + * the schema's defaults applied - which is why a request carrying no body can still yield one. */ function resolveRunInput(build: BuildRecord, raw: ActorInput | undefined): ActorInput | undefined { if (!build.inputSchema) return raw; const processed = processActorInput(raw, build.inputSchema); diff --git a/src/services/builds.ts b/src/services/builds.ts index 55319cc..d18a0ff 100644 --- a/src/services/builds.ts +++ b/src/services/builds.ts @@ -216,10 +216,8 @@ export async function runBuildInBackground( } for (const line of dockerfileResolution.logLines) appendRuntimeLog(record.id, line); - // Resolved before the image is built, not after: an Actor whose declared input contract cannot be - // read (or is itself invalid) fails the build with that reason stated, rather than building an image - // whose every later run would silently skip validation. Same failure shape as the Dockerfile - // resolution above, and, like it, nothing here ever touches the pushed source itself. + // Before the image is built, not after: an Actor whose declared input contract cannot be read fails + // the build with that reason, rather than producing an image whose every run skips validation. const inputSchemaResolution = resolveInputSchemaLocation(version.sourceFiles); if (inputSchemaResolution.outcome === 'failure') { appendRuntimeLog(record.id, inputSchemaResolution.message); @@ -289,9 +287,8 @@ export async function runBuildInBackground( ...(outcome.imageWorkingDirectory !== undefined ? { imageWorkingDirectory: outcome.imageWorkingDirectory } : {}), - // Omitted entirely (rather than written as `undefined`) for an Actor with no input schema, - // same reason as `imageWorkingDirectory` right above: "never present on a non-SUCCEEDED - // build" stays true of the value too, there is simply nothing to record for this build. + // Omitted rather than written as `undefined` for an Actor with no input schema, same reason + // as `imageWorkingDirectory` above. ...(inputSchema !== undefined ? { inputSchema } : {}), }); if (succeeded?.status !== 'SUCCEEDED') { diff --git a/src/services/input-schema-location.ts b/src/services/input-schema-location.ts index 075f91c..a0e9720 100644 --- a/src/services/input-schema-location.ts +++ b/src/services/input-schema-location.ts @@ -1,23 +1,15 @@ /** - * Resolves which input schema a build should carry from the version's `sourceFiles` - the build-time - * half of input validation (`actor-driver.md`'s "Input schema, validation and defaults"). The run-time - * half (defaults, validation) lives in `services/input-schema.ts` and reads only the schema this module - * resolved, never the source files again. + * Which input schema a build carries, resolved from the version's `sourceFiles` + * (`actor-driver.md`'s "Input schema, validation and defaults"). The run-time half - defaults and + * validation - is `services/input-schema.ts`, which reads only the resolved schema, never the source + * files again. * - * Candidate order, stopping at the first hit, matching what `apify-cli` itself looks for locally (its - * `readInputSchema`), which is also what the platform's builder records on the build: - * 1. the `input` field of `.actor/actor.json` - an inline schema object, or a path relative to - * `.actor/` - * 2. `.actor/INPUT_SCHEMA.json` - * 3. `INPUT_SCHEMA.json` at the Actor root - * Matching is case-insensitive with an exact-case win, exactly like `services/dockerfile-location.ts` - * (so `.actor/input_schema.json` and `.actor/INPUT_SCHEMA.json` are the same candidate - which is why - * the CLI's four-entry list collapses to the two locations above here), and every outcome is stated in - * the build log. + * The candidate order is `apify-cli`'s own `readInputSchema`, so a developer's local `apify + * validate-schema` and this build agree on which file is the Actor's schema. Case-insensitive matching + * (as in `services/dockerfile-location.ts`) is why the CLI's four candidates collapse to two here. * - * `.actor/actor.json` and the schema file alike are parsed as JSON5, matching how this runtime already - * parses `.actor/actor.json` for the Dockerfile lookup; every valid JSON file is also valid JSON5, so - * this only ever accepts more than the platform, never less. + * Schema files are parsed as JSON5, like `.actor/actor.json` already is: every valid JSON file is + * valid JSON5, so this only ever accepts more than the platform, never less. */ import * as path from 'node:path'; import JSON5 from 'json5'; @@ -31,15 +23,12 @@ const ACTOR_JSON_NAME = `${ACTOR_DIR}/actor.json`; /** Checked case-insensitively, so the `INPUT_SCHEMA.json` spelling matches these too. */ const DEFAULT_SCHEMA_CANDIDATES = [`${ACTOR_DIR}/input_schema.json`, 'input_schema.json'] as const; -/** Why input-schema resolution failed. Each one fails the build, the same way the Dockerfile - * resolution's own failures do - an Actor whose declared input contract cannot be read is not built - * with that contract silently dropped. */ +/** Why resolution failed. Each one fails the build: an Actor whose declared input contract cannot be + * read must not be built with that contract silently dropped. */ export type InputSchemaResolutionFailureReason = 'escapes-actor-root' | 'invalid-input-field' | 'unparseable-input-schema' | 'invalid-input-schema'; -/** `resolveInputSchemaLocation`'s outcomes: `resolved` (a candidate matched and the schema is valid), - * `none` (no candidate at all - the Actor simply has no input schema, which is not an error), or - * `failure`. */ +/** `none` means the Actor declares no input schema, which is not an error. */ export type InputSchemaResolution = | { outcome: 'resolved'; schema: InputSchema; source: string; logLines: string[] } | { outcome: 'none'; logLines: string[] } @@ -49,8 +38,8 @@ function sourceFileToText(file: SourceFile): string { return file.format === 'BASE64' ? Buffer.from(file.content, 'base64').toString('utf8') : file.content; } -/** Exact-case match wins; otherwise the first match in `sourceFiles` order - same rule as - * `services/dockerfile-location.ts`'s own lookup. */ +/** Exact-case match wins; otherwise the first match in `sourceFiles` order - the same rule as the + * Dockerfile lookup's. */ function findCaseInsensitive(sourceFiles: SourceFile[], candidate: string): SourceFile | undefined { const lowerCandidate = candidate.toLowerCase(); let firstMatch: SourceFile | undefined; @@ -76,8 +65,7 @@ function escapesActorRootFailure(rawField: string): InputSchemaResolution { }; } -/** Meta-validates a resolved candidate, turning it into the `resolved`/`failure` outcome. Kept in one - * place so an inline `input` object and a schema file are held to exactly the same standard. */ +/** One place, so an inline `input` object and a schema file are held to the same standard. */ function acceptSchema(schema: unknown, source: string, logLines: string[]): InputSchemaResolution { const defect = describeInputSchemaDefect(schema); if (defect) { @@ -167,8 +155,6 @@ export function resolveInputSchemaLocation(sourceFiles: SourceFile[]): InputSche return acceptSchema(parsed.schema, `"${normalizeEntryName(match.name)}"`, logLines); } - // No input schema at all is the normal state of an Actor that never declared one - runs of such a - // build take their input exactly as the caller sent it, with nothing validated and nothing added. return { outcome: 'none', logLines }; } diff --git a/src/services/input-schema.ts b/src/services/input-schema.ts index 68a98dc..f1e3662 100644 --- a/src/services/input-schema.ts +++ b/src/services/input-schema.ts @@ -1,20 +1,10 @@ /** - * Input validation and defaults from the Actor's input schema - the run-time half of the feature whose - * build-time half (finding the schema in the pushed source) is `services/input-schema-location.ts`. + * Input validation and defaults at run start; `services/input-schema-location.ts` is the build-time + * half that finds the schema in the pushed source. * - * Validation itself is the platform's own: `@apify/input_schema`'s `validateInputSchema` (the schema - * against the Apify input-schema meta-schema) and `validateInputUsingValidator` (the input against the - * schema, plus the proxy/`requestListSources`/pattern checks AJV alone cannot express), driven by the - * same AJV configuration the platform's API uses. That is deliberate: a developer must see locally the - * very same message the real API would answer with, not an approximation of it. - * - * Two behaviours the platform has and this runtime deliberately does not: - * - **Proxy group availability.** The platform passes the caller's proxy groups into the validator, - * which then rejects a group the account cannot use. This runtime emulates no proxy groups at all - * (`unsupported.md`), so it passes none: a `proxy` field is still checked for shape, custom proxy - * URLs and country code, but any `apifyProxyGroups` selection is accepted. - * - **Encrypted secret input fields** stay unsupported (`unsupported.md`); a schema may declare - * `isSecret`, and the value is then validated as the plain value it locally is. + * Validation goes through the platform's own `@apify/input_schema`, driven by the same AJV + * configuration the platform's API uses, so a developer reads locally the very message the real API + * would answer with rather than an approximation of it. */ import ajvPackage from 'ajv'; import ajv2019Package from 'ajv/dist/2019.js'; @@ -23,8 +13,8 @@ import { validateInputSchema, validateInputUsingValidator } from '@apify/input_s import type { InputSchema } from '../storage/entities.js'; // AJV ships as CommonJS, so under this package's ESM resolution its class arrives as the module's -// `default` rather than as the import binding itself. `2019` is the same library built with -// draft-2019-09 support, which the Apify input-schema meta-schema requires and the plain build lacks. +// `default`. `2019` is the same library with draft-2019-09 support, which the Apify input-schema +// meta-schema requires and the plain build lacks. const Ajv = ajvPackage.default; const Ajv2019 = ajv2019Package.default; type InputValidator = ReturnType['compile']>; @@ -46,10 +36,8 @@ export type InputProcessingResult = | { kind: 'invalid'; message: string } | { kind: 'invalid-schema'; message: string }; -/** The messages the real Apify API answers with, word for word (`@apify-packages/errors`'s - * `actor.inputNotJson`/`inputNotValidJson`/`inputNotObject`/`inputNotValid`/`invalidInputSchema`) - - * the whole point of routing validation through the platform's own validator is that the text a - * developer reads locally is the text they would read against the platform. */ +/** The real Apify API's messages, word for word (`@apify-packages/errors`'s `actor.inputNotJson` and + * its neighbours). */ export function describeInputProcessingFailure(result: Exclude): string { switch (result.kind) { case 'not-json': @@ -66,22 +54,18 @@ export function describeInputProcessingFailure(result: Exclude); return null; } catch (error) { @@ -90,11 +74,9 @@ export function describeInputSchemaDefect(schema: unknown): string | null { } /** - * Applies the schema's defaults to `input` and validates the result, returning the effective input a - * run should actually be started with. Mirrors the platform's `processInputUsingSchema`: an absent - * input is not "no input" once a schema exists - it is an empty object the defaults are applied to, so - * `apify call` with no input at all still gets the schema's defaults in its `INPUT` record, and a - * required field with no default is still reported missing. + * The effective input a run should start with. Mirrors the platform's `processInputUsingSchema`: once + * a schema exists, an absent input is an empty object the defaults are applied to, not "no input" - so + * a call with no input still gets the defaults, and a required field with no default is still missing. */ export function processActorInput(input: ActorInput | undefined, schema: InputSchema): InputProcessingResult { let parsedInput: unknown = {}; @@ -118,16 +100,17 @@ export function processActorInput(input: ActorInput | undefined, schema: InputSc try { validator = compileInputSchemaValidator(schema); } catch (error) { - // Only reachable for a schema that passed the build-time meta-validation but still cannot be - // compiled (or one recorded by an older runtime, before builds validated schemas at all). + // Reachable only for a schema recorded before builds validated them, or one that meta-validates + // yet still will not compile. return { kind: 'invalid-schema', message: (error as Error).message }; } - // No `proxy` options: proxy-group availability is not emulated here (see this module's doc comment). + // No `proxy` options: this runtime emulates no proxy groups (`unsupported.md`), so any + // `apifyProxyGroups` selection is accepted, while the rest of a proxy field is still checked. const validationErrors = validateInputUsingValidator(validator, schema, withDefaults, {}); if (validationErrors.length > 0) { - // Every message, joined - what the platform returns for an API-origin run (it shortens this to - // the first message only for the web console, which has a single notification line to show it in). + // Joined, as the platform answers an API-origin run; it shortens this to the first message only + // for the web console, which has one notification line to show it in. return { kind: 'invalid', message: validationErrors.map(({ message }) => message).join(', ') }; } @@ -150,11 +133,8 @@ function isPlainObject(value: unknown): value is Record { } /** - * Merges the schema's default values into `values` and returns a new object - a port of the platform's - * `mergeDefaultsFromInputSchema`, including its recursion into nested objects and into the items of an - * array whose `items` sub-schema carries defaults. - * - * Exported for direct testing; run-start callers go through `processActorInput`. + * A port of the platform's `mergeDefaultsFromInputSchema`, recursion into nested objects and array + * items included. Exported for direct testing; run-start callers go through `processActorInput`. */ export function mergeDefaultsFromInputSchema( values: Record, @@ -166,8 +146,8 @@ export function mergeDefaultsFromInputSchema( defaults[key] = extractDefaults(fieldSchema); } - // Cloned once here so nothing ever assigns a live reference into the schema record itself - a later - // mutation of the returned input (or of the stored schema) can then never reach the other. + // Cloned so no reference into the stored schema is ever assigned into the returned input, where a + // later mutation of either would reach the other. const clonedDefaults = structuredClone(defaults); const target = structuredClone(values); assignRecursively(target, clonedDefaults, schema); @@ -198,9 +178,8 @@ function extractArrayItemDefaults(fieldSchema: unknown): unknown { return extractDefaults(fieldSchema.items); } -/** Assigns a default only where the current value is `undefined`, recursing into nested objects and - * into array items. `fillDefinedValues` is set for array items alone, where the platform fills an - * item's fields even though the item itself is present. */ +/** `fillDefinedValues` is set for array items alone, where the platform fills an item's fields even + * though the item itself is present. */ function assignRecursively( target: Record, defaults: Record, @@ -226,8 +205,7 @@ function assignRecursively( : ((extractDefaults(fieldSchema) as Record | undefined) ?? {}); assignRecursively( currentValue, - // Without `fillDefinedValues`, a present nested object keeps exactly the keys it came - // with - only its own sub-schema's defaults for keys it is missing are considered. + // A present nested object otherwise keeps exactly the keys it came with. fillDefinedValues ? nestedDefaults : {}, fieldSchema as InputSchema | undefined, ); @@ -246,8 +224,8 @@ function assignRecursively( } } -/** Deep-merges `overrides` onto a copy of `base`, object by object - the one `lodash.merge` behaviour - * `extractDefaults` needs, without the dependency. Arrays and scalars replace wholesale. */ +/** The one `lodash.merge` behaviour `extractDefaults` needs, without the dependency: objects merge + * key by key, arrays and scalars replace wholesale. */ function deepMerge(base: Record, overrides: Record): Record { const merged: Record = { ...base }; for (const [key, value] of Object.entries(overrides)) { @@ -258,9 +236,9 @@ function deepMerge(base: Record, overrides: Record(); const VALIDATOR_CACHE_MAX_ENTRIES = 100; @@ -281,10 +259,10 @@ function compileInputSchemaValidator(schema: InputSchema): InputValidator { } /** - * The platform's `getAjvValidator` preparation, ported: a required field that has a default is treated - * as optional (there is always a value for it by the time validation runs), a required array must hold - * at least one item, and `$schema` is dropped because AJV would otherwise try to fetch the Apify - * meta-schema it names and fail to compile the schema at all. + * The platform's `getAjvValidator` preparation, ported: a required field with a default is optional + * (there is always a value for it by then), a required array must hold at least one item, and + * `$schema` is dropped - AJV would otherwise try to fetch the Apify meta-schema it names and fail to + * compile at all. */ function prepareSchemaForValidation(schema: InputSchema): Record { const copy = structuredClone(schema) as Record; diff --git a/src/storage/entities.ts b/src/storage/entities.ts index f706db1..c23e0da 100644 --- a/src/storage/entities.ts +++ b/src/storage/entities.ts @@ -43,10 +43,9 @@ export interface SourceFile { } /** - * An Actor's input schema, exactly as it was pushed (`services/input-schema-location.ts` resolves which - * file or inline object that is). Kept as a loose record rather than a modelled type: it is the - * developer's own document, handed to the platform's own validator verbatim, and the runtime itself - * only ever reads `properties`/`required` out of it. + * An Actor's input schema, exactly as it was pushed. A loose record rather than a modelled type: it is + * the developer's own document, handed to the platform's validator verbatim, and the runtime itself + * reads only `properties`/`required` out of it. */ export interface InputSchema { [key: string]: unknown; @@ -132,12 +131,10 @@ export interface BuildRecord { * inspect failed or the working directory was empty/`/` (mounting over `/` would destroy the * container) - never present on a non-`SUCCEEDED` build. */ imageWorkingDirectory?: string; - /** The input schema resolved from this build's own source files (`services/input-schema-location.ts`), - * recorded in the same status-transition write that records `SUCCEEDED`, exactly like - * `imageWorkingDirectory` above. Build-specific for the same reason that field is: a run validates - * its input against the schema of the build it actually resolved, never against some other tag's - * more recently pushed one. Absent when the Actor declares no input schema at all, and never present - * on a non-`SUCCEEDED` build - such a run then takes its input exactly as the caller sent it. */ + /** The input schema this build's own source files declared, written with `SUCCEEDED` like + * `imageWorkingDirectory` above, and build-specific for the same reason: a run validates against the + * schema of the build it resolved, never another tag's more recently pushed one. Absent when the + * Actor declares none - such a run takes its input exactly as the caller sent it. */ inputSchema?: InputSchema; exitCode?: number; statusMessage?: string; From ec796ffe95e38c17ebb1586de2abbbee7f1aabc0 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 23 Sep 2026 08:28:39 +0000 Subject: [PATCH 03/14] test: Assert the schema defaults land in the run's INPUT record The e2e case asserted only the dataset item count for a call with no input, which the sample Actor's own internal fallback for a missing maxPages produces just as well. It now reads back the run's INPUT record through the CLI, which only the runtime writes, so it really distinguishes the schema's defaults from the Actor defaulting on its own. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_011U9yUTUn78bqGDn5obYSXs --- test/e2e/actor-dev-loop.test.ts | 14 ++++++++++++-- 1 file changed, 12 insertions(+), 2 deletions(-) diff --git a/test/e2e/actor-dev-loop.test.ts b/test/e2e/actor-dev-loop.test.ts index e0c80b0..2df095e 100644 --- a/test/e2e/actor-dev-loop.test.ts +++ b/test/e2e/actor-dev-loop.test.ts @@ -138,12 +138,22 @@ describe('full Actor dev loop via apify-cli (requires Docker)', () => { const env = apifyEnv(isolatedApifyHome); const actorDir = join(REPO_ROOT, 'sample_actor_ts'); - // No `--input`: the Actor's input schema defaults `maxPages` to 2, so the item count is 2 - - // input-dependent, like every other assertion here, just with the input coming from the schema. const callOutput = apify(['call', '--json'], { cwd: actorDir, env }); const call = JSON.parse(callOutput) as CallResult; expect(call.run.status).toBe('SUCCEEDED'); + // The run's own INPUT record, which only the runtime ever writes: every field of the schema, at + // its default. This is what distinguishes the schema's defaults from the Actor's own internal + // fallback for a missing field - the item count alone cannot tell the two apart. + const storedInput = apify(['key-value-stores', 'get-value', call.storage.defaultKeyValueStoreId, 'INPUT'], { + cwd: actorDir, + env, + }); + expect(JSON.parse(storedInput) as Record).toEqual({ + startUrl: 'https://crawlee.dev/', + maxPages: 2, + }); + const infoOutput = apify(['datasets', 'info', call.storage.defaultDatasetId, '--json'], { cwd: actorDir, env, From ea791633e43ceef8219782761dc599b4e8a88927 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 23 Sep 2026 09:04:39 +0000 Subject: [PATCH 04/14] refactor: Share the pushed-source lookup and trim the input-schema code The input-schema resolver was a near copy of the Dockerfile one, down to the Actor-root containment check. Both now read the pushed source through one module, so a traversal fix cannot land in only one of them, and an unparseable .actor/actor.json is reported by the resolver itself instead of relying on the Dockerfile resolver having run first. Also: one fail-the-build path in place of three copies, the two platform error types moved to the shared error factories, the schema decision moved out of the HTTP route into the service, defaults no longer clone an input nobody else holds, and the comments that restated their own code are gone. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_011U9yUTUn78bqGDn5obYSXs --- src/api/errors.ts | 11 ++ src/api/routes/actors.ts | 49 +++--- src/services/actor-source-files.ts | 92 ++++++++++++ src/services/builds.ts | 45 +++--- src/services/dockerfile-location.ts | 109 ++++---------- src/services/input-schema-location.ts | 189 +++++++++++------------- src/services/input-schema.ts | 175 ++++++++-------------- test/e2e/helpers/apify-cli.ts | 36 ++--- test/integration/input-schema.test.ts | 57 ++----- test/unit/input-schema-location.test.ts | 178 +++++++++++----------- test/unit/input-schema.test.ts | 110 +++++--------- 11 files changed, 472 insertions(+), 579 deletions(-) create mode 100644 src/services/actor-source-files.ts diff --git a/src/api/errors.ts b/src/api/errors.ts index 00b7ab4..2d0f6dc 100644 --- a/src/api/errors.ts +++ b/src/api/errors.ts @@ -30,6 +30,17 @@ export function endpointNotFound(message: string): ApiError { return new ApiError(404, 'not-found', message); } +/** Matches the real platform's rejection of an input its Actor's input schema does not accept + * (`@apify-packages/errors`'s `actor.inputNotValid` and its neighbours), message included. */ +export function invalidInput(message: string): ApiError { + return new ApiError(400, 'invalid-input', message); +} + +/** Matches the real platform's `actor.invalidInputSchema`. */ +export function invalidInputSchema(message: string): ApiError { + return new ApiError(400, 'invalid-input-schema', message); +} + /** * Matches the real Apify platform exactly: `DELETE /v2/actor-runs/:runId` on a non-terminal run is * rejected rather than aborted-then-deleted (the public API answers 400 `cannot-remove-running-run`), diff --git a/src/api/routes/actors.ts b/src/api/routes/actors.ts index 3610d4b..418e336 100644 --- a/src/api/routes/actors.ts +++ b/src/api/routes/actors.ts @@ -3,7 +3,14 @@ import type { Router } from 'express'; import { requireUser } from '../auth.js'; import { paginate, sendData, sendPaginated, sortByTimestamp } from '../envelope.js'; -import { ApiError, cannotSetPricingOnCreate, recordNotFound, invalidRequest } from '../errors.js'; +import { + ApiError, + cannotSetPricingOnCreate, + invalidInput, + invalidInputSchema, + invalidRequest, + recordNotFound, +} from '../errors.js'; import { h, jsonBody, paginationParams, queryBoolean, queryNumber, queryString, rawBody } from '../handler.js'; import { addOrReplaceVersion, @@ -25,17 +32,12 @@ import { import { listOwnedRuns, startRun, waitForRunFinish } from '../../services/runs.js'; import { getRegistries } from '../../storage/registries.js'; import { actorDto, buildDto, runDto } from '../dto/actors.js'; -import type { ActorPricingInfoRecord, ActorRecord, ActorVersionRecord, BuildRecord } from '../../storage/entities.js'; +import type { ActorPricingInfoRecord, ActorRecord, ActorVersionRecord } from '../../storage/entities.js'; import type { ApiServerDeps } from '../server.js'; import { CONTAINER_API_BASE_URL } from '../../config.js'; import { resolveProxyPassword } from '../../services/users.js'; import { validatePricingInfosUpdate } from '../../services/pricing.js'; -import { - describeInputProcessingFailure, - processActorInput, - type ActorInput, - type InputProcessingResult, -} from '../../services/input-schema.js'; +import { resolveBuildInput } from '../../services/input-schema.js'; /** * `undefined` when the body does not mention the field; a body that does but is invalid throws, with the @@ -255,23 +257,6 @@ export function mountActors(router: Router, deps: ApiServerDeps): void { }), ); - /** Status and error `type` are the route's concern, the message text the service's - the split - * `api/routes/dev-folder.ts` already uses. Both are the real platform's for the same input. */ - function inputApiError(result: Exclude): ApiError { - const message = describeInputProcessingFailure(result); - const type = result.kind === 'invalid-schema' ? 'invalid-input-schema' : 'invalid-input'; - return new ApiError(400, type, message); - } - - /** The caller's bytes as-is for a build with no input schema; otherwise the validated input with - * the schema's defaults applied - which is why a request carrying no body can still yield one. */ - function resolveRunInput(build: BuildRecord, raw: ActorInput | undefined): ActorInput | undefined { - if (!build.inputSchema) return raw; - const processed = processActorInput(raw, build.inputSchema); - if (processed.kind !== 'ok') throw inputApiError(processed); - return processed.input; - } - router.post( '/actors/:actorId/runs', h(async (req, res) => { @@ -298,9 +283,15 @@ export function mountActors(router: Router, deps: ApiServerDeps): void { const build = lookup.build; const body = rawBody(req); - const rawInput = - body.length > 0 ? { body, contentType: req.header('content-type') ?? 'application/json' } : undefined; - const input = resolveRunInput(build, rawInput); + const processed = resolveBuildInput( + build, + body.length > 0 ? { body, contentType: req.header('content-type') ?? 'application/json' } : undefined, + ); + if (processed.kind !== 'ok') { + throw processed.kind === 'invalid-input-schema' + ? invalidInputSchema(processed.message) + : invalidInput(processed.message); + } // `resolveProxyPassword(requireUser(req))` is the *run owner's* proxy password, not just "the // caller's": `actor` was resolved via `resolveActorParam(req)` above, so @@ -308,7 +299,7 @@ export function mountActors(router: Router, deps: ApiServerDeps): void { // their own Actor - which makes the two the same user record (`actor-driver.md`'s "one // harvested-per-account password used specifically for each user"). const run = await startRun(deps.driver, actor, build, { - input, + input: processed.input, memoryMbytes: queryNumber(req, 'memory'), timeoutSecs: queryNumber(req, 'timeout'), maxTotalChargeUsd, diff --git a/src/services/actor-source-files.ts b/src/services/actor-source-files.ts new file mode 100644 index 0000000..a6d97dd --- /dev/null +++ b/src/services/actor-source-files.ts @@ -0,0 +1,92 @@ +/** + * What both build-time resolvers (`dockerfile-location.ts`, `input-schema-location.ts`) need from a + * version's pushed `sourceFiles`: the parsed `.actor/actor.json`, and the path fields inside it that + * name another pushed file. + * + * Shared rather than written per field so the Actor-root containment check has one implementation: + * a traversal hole patched in one copy would otherwise stay open in the other. + */ +import * as path from 'node:path'; +import JSON5 from 'json5'; + +import { normalizeEntryName } from '../driver/tar-entry-name.js'; +import type { SourceFile } from '../storage/entities.js'; + +export const ACTOR_DIR = '.actor'; +export const ACTOR_JSON_NAME = `${ACTOR_DIR}/actor.json`; + +export function sourceFileToText(file: SourceFile): string { + return file.format === 'BASE64' ? Buffer.from(file.content, 'base64').toString('utf8') : file.content; +} + +export interface IndexedFile { + normalizedName: string; + lowerName: string; + file: SourceFile; +} + +export function indexSourceFiles(sourceFiles: SourceFile[]): IndexedFile[] { + return sourceFiles.map((file) => { + const normalizedName = normalizeEntryName(file.name); + return { normalizedName, lowerName: normalizedName.toLowerCase(), file }; + }); +} + +/** Exact-case match wins; otherwise the first match in `sourceFiles` order. */ +export function findCaseInsensitive(indexed: IndexedFile[], candidate: string): IndexedFile | undefined { + const lowerCandidate = candidate.toLowerCase(); + let firstMatch: IndexedFile | undefined; + for (const file of indexed) { + if (file.lowerName !== lowerCandidate) continue; + if (file.normalizedName === candidate) return file; // exact case always wins immediately + firstMatch ??= file; + } + return firstMatch; +} + +/** `.actor/actor.json`'s own path is not case-folded, unlike the files its fields name. */ +export function findExact(sourceFiles: SourceFile[], normalizedTarget: string): SourceFile | undefined { + return sourceFiles.find((file) => normalizeEntryName(file.name) === normalizedTarget); +} + +/** `absent` is not an error: an Actor need not push `.actor/actor.json` at all. */ +export type ActorJsonParse = + { outcome: 'parsed'; specification: unknown } | { outcome: 'absent' } | { outcome: 'unparseable'; message: string }; + +export function parseActorJson(sourceFiles: SourceFile[]): ActorJsonParse { + const file = findExact(sourceFiles, ACTOR_JSON_NAME); + if (!file) return { outcome: 'absent' }; + try { + return { outcome: 'parsed', specification: JSON5.parse(sourceFileToText(file)) as unknown }; + } catch (error) { + return { outcome: 'unparseable', message: `Could not parse .actor/actor.json: ${(error as Error).message}` }; + } +} + +/** Where a `.actor/actor.json` path field points. `not-found` carries the path to name in the caller's + * warning - the empty field and a path naming no pushed file are the same outcome, both falling through + * to the default locations rather than failing. */ +export type ActorJsonPathField = + | { outcome: 'match'; file: IndexedFile } + | { outcome: 'not-found'; shownPath: string } + | { outcome: 'escapes-actor-root' }; + +/** `field` is resolved relative to `.actor/`, and may not leave the Actor root. */ +export function resolveActorJsonPathField(indexed: IndexedFile[], field: string): ActorJsonPathField { + if (field === '') return { outcome: 'not-found', shownPath: '' }; + if (field.startsWith('/')) return { outcome: 'escapes-actor-root' }; + + const joined = normalizeEntryName(path.posix.join(ACTOR_DIR, field)); + if (joined === '..' || joined.startsWith('../')) return { outcome: 'escapes-actor-root' }; + + const match = findCaseInsensitive(indexed, joined); + return match ? { outcome: 'match', file: match } : { outcome: 'not-found', shownPath: joined }; +} + +export function fallbackWarningLine(shownPath: string, fieldName: string): string { + return `Warning: "${shownPath}" (from the "${fieldName}" field in .actor/actor.json) is not in the pushed source; falling back to the default locations.\n`; +} + +export function escapesActorRootMessage(rawField: string, subject: string): string { + return `${subject} path "${rawField}" in .actor/actor.json points outside the Actor root directory.`; +} diff --git a/src/services/builds.ts b/src/services/builds.ts index d18a0ff..9832597 100644 --- a/src/services/builds.ts +++ b/src/services/builds.ts @@ -6,6 +6,7 @@ import { recordTaggedBuild, updateActor } from './actors.js'; import type { Driver } from '../driver/types.js'; import { DriverTimedOutError } from '../driver/types.js'; import { normalizeEntryName } from '../driver/tar-entry-name.js'; +import { sourceFileToText } from './actor-source-files.js'; import { qualifyDockerfileImageReferences } from './dockerfile-image-refs.js'; import { resolveDockerfileLocation } from './dockerfile-location.js'; import { resolveInputSchemaLocation } from './input-schema-location.js'; @@ -90,8 +91,7 @@ function qualifyDockerfileImages( ): SourceFile[] { return sourceFiles.map((file) => { if (normalizeEntryName(file.name) !== dockerfilePath) return file; - const text = file.format === 'BASE64' ? Buffer.from(file.content, 'base64').toString('utf8') : file.content; - const { dockerfile, qualified } = qualifyDockerfileImageReferences(text); + const { dockerfile, qualified } = qualifyDockerfileImageReferences(sourceFileToText(file)); if (qualified.length === 0) return file; for (const { from, to } of qualified) { log(`Using "${to}" for FROM "${from}" - a short image name means Docker Hub, as on the platform.\n`); @@ -147,6 +147,19 @@ export async function startBuild( return record; } +/** Fails a build before any image exists, mirroring `services/runs.ts`'s `failBeforeContainer`: the + * reason reaches both the build log and the status message, which default to the same text. */ +async function failBuild(buildId: string, logMessage: string, statusMessage = logMessage): Promise { + const { builds } = getRegistries(); + appendRuntimeLog(buildId, logMessage); + await flushLog(buildId); + markLogTerminal(buildId); + await transitionJobStatus(builds, buildId, 'FAILED', { + finishedAt: new Date().toISOString(), + statusMessage, + }); +} + /** * Exported only for direct testing of the guarded transitions/pre-start abort window (see * `test/integration/job-lifecycle.test.ts`) - not part of the service's public surface for callers @@ -181,13 +194,7 @@ export async function runBuildInBackground( } if (!driver.available) { - appendRuntimeLog(record.id, `Docker is not available: ${driver.unavailableReason}`); - await flushLog(record.id); - markLogTerminal(record.id); - await transitionJobStatus(builds, record.id, 'FAILED', { - finishedAt: new Date().toISOString(), - statusMessage: driver.unavailableReason, - }); + await failBuild(record.id, `Docker is not available: ${driver.unavailableReason}`, driver.unavailableReason); return; } @@ -205,28 +212,16 @@ export async function runBuildInBackground( const dockerfileResolution = resolveDockerfileLocation(version.sourceFiles); if (dockerfileResolution.outcome === 'failure') { - appendRuntimeLog(record.id, dockerfileResolution.message); - await flushLog(record.id); - markLogTerminal(record.id); - await transitionJobStatus(builds, record.id, 'FAILED', { - finishedAt: new Date().toISOString(), - statusMessage: dockerfileResolution.message, - }); + await failBuild(record.id, dockerfileResolution.message); return; } for (const line of dockerfileResolution.logLines) appendRuntimeLog(record.id, line); // Before the image is built, not after: an Actor whose declared input contract cannot be read fails - // the build with that reason, rather than producing an image whose every run skips validation. + // the build, rather than producing an image whose every run would skip validation. const inputSchemaResolution = resolveInputSchemaLocation(version.sourceFiles); if (inputSchemaResolution.outcome === 'failure') { - appendRuntimeLog(record.id, inputSchemaResolution.message); - await flushLog(record.id); - markLogTerminal(record.id); - await transitionJobStatus(builds, record.id, 'FAILED', { - finishedAt: new Date().toISOString(), - statusMessage: inputSchemaResolution.message, - }); + await failBuild(record.id, inputSchemaResolution.message); return; } for (const line of inputSchemaResolution.logLines) appendRuntimeLog(record.id, line); @@ -287,8 +282,6 @@ export async function runBuildInBackground( ...(outcome.imageWorkingDirectory !== undefined ? { imageWorkingDirectory: outcome.imageWorkingDirectory } : {}), - // Omitted rather than written as `undefined` for an Actor with no input schema, same reason - // as `imageWorkingDirectory` above. ...(inputSchema !== undefined ? { inputSchema } : {}), }); if (succeeded?.status !== 'SUCCEEDED') { diff --git a/src/services/dockerfile-location.ts b/src/services/dockerfile-location.ts index da7ea76..0bd530f 100644 --- a/src/services/dockerfile-location.ts +++ b/src/services/dockerfile-location.ts @@ -5,16 +5,19 @@ * case-insensitive; the returned path is always the matched file's own name, never the candidate's * casing - Docker's tar lookup is case-sensitive. An exact-case match wins over a case-differing one. */ -import * as path from 'node:path'; -import JSON5 from 'json5'; - import { normalizeEntryName } from '../driver/tar-entry-name.js'; import type { SourceFile } from '../storage/entities.js'; +import { + ACTOR_DIR, + escapesActorRootMessage, + fallbackWarningLine, + findCaseInsensitive, + indexSourceFiles, + parseActorJson, + resolveActorJsonPathField, +} from './actor-source-files.js'; import { DEFAULT_DOCKERFILE_CONTENT, DEFAULT_DOCKERFILE_NAME } from './default-dockerfile.js'; -const ACTOR_DIR = '.actor'; -const ACTOR_JSON_NAME = `${ACTOR_DIR}/actor.json`; - /** Why Dockerfile resolution failed. */ export type DockerfileResolutionFailureReason = 'escapes-actor-root' | 'invalid-dockerfile-field' | 'unparseable-actor-json'; @@ -26,44 +29,11 @@ export type DockerfileResolution = | { outcome: 'default'; dockerfilePath: string; logLines: string[]; extraSourceFile: SourceFile } | { outcome: 'failure'; reason: DockerfileResolutionFailureReason; message: string }; -function sourceFileToText(file: SourceFile): string { - return file.format === 'BASE64' ? Buffer.from(file.content, 'base64').toString('utf8') : file.content; -} - -interface IndexedFile { - normalizedName: string; - lowerName: string; -} - -function indexSourceFiles(sourceFiles: SourceFile[]): IndexedFile[] { - return sourceFiles.map((file) => { - const normalizedName = normalizeEntryName(file.name); - return { normalizedName, lowerName: normalizedName.toLowerCase() }; - }); -} - -/** Exact-case match wins; otherwise the first match in `sourceFiles` order. */ -function findCaseInsensitive(indexed: IndexedFile[], candidate: string): IndexedFile | undefined { - const lowerCandidate = candidate.toLowerCase(); - let firstMatch: IndexedFile | undefined; - for (const file of indexed) { - if (file.lowerName !== lowerCandidate) continue; - if (file.normalizedName === candidate) return file; // exact case always wins immediately - firstMatch ??= file; - } - return firstMatch; -} - -/** `.actor/actor.json`'s own path is not case-folded, unlike the Dockerfile candidates. */ -function findExact(sourceFiles: SourceFile[], normalizedTarget: string): SourceFile | undefined { - return sourceFiles.find((file) => normalizeEntryName(file.name) === normalizedTarget); -} - function escapesActorRootFailure(rawField: string): DockerfileResolution { return { outcome: 'failure', reason: 'escapes-actor-root', - message: `Dockerfile path "${rawField}" in .actor/actor.json points outside the Actor root directory.`, + message: escapesActorRootMessage(rawField, 'Dockerfile'), }; } @@ -71,22 +41,14 @@ export function resolveDockerfileLocation(sourceFiles: SourceFile[]): Dockerfile const indexed = indexSourceFiles(sourceFiles); const logLines: string[] = []; - const actorJsonFile = findExact(sourceFiles, ACTOR_JSON_NAME); - let actorSpecification: unknown; - if (actorJsonFile) { - try { - actorSpecification = JSON5.parse(sourceFileToText(actorJsonFile)); - } catch (error) { - return { - outcome: 'failure', - reason: 'unparseable-actor-json', - message: `Could not parse .actor/actor.json: ${(error as Error).message}`, - }; - } + const actorJson = parseActorJson(sourceFiles); + if (actorJson.outcome === 'unparseable') { + return { outcome: 'failure', reason: 'unparseable-actor-json', message: actorJson.message }; } + const specification = actorJson.outcome === 'parsed' ? actorJson.specification : undefined; - if (actorSpecification !== null && typeof actorSpecification === 'object' && 'dockerfile' in actorSpecification) { - const field: unknown = actorSpecification.dockerfile; + if (specification !== null && typeof specification === 'object' && 'dockerfile' in specification) { + const field: unknown = specification.dockerfile; if (typeof field !== 'string') { return { outcome: 'failure', @@ -95,34 +57,19 @@ export function resolveDockerfileLocation(sourceFiles: SourceFile[]): Dockerfile }; } - if (field === '') { - logLines.push( - 'Warning: "" (from the "dockerfile" field in .actor/actor.json) is not in the pushed source; falling back to the default locations.\n', - ); - } else if (field.startsWith('/')) { - return escapesActorRootFailure(field); - } else { - const joined = normalizeEntryName(path.posix.join(ACTOR_DIR, field)); - if (joined === '..' || joined.startsWith('../')) { - return escapesActorRootFailure(field); - } - - const match = findCaseInsensitive(indexed, joined); - if (match) { - return { - outcome: 'resolved', - dockerfilePath: match.normalizedName, - logLines: [ - ...logLines, - `Using Dockerfile "${match.normalizedName}" (from the "dockerfile" field in .actor/actor.json).\n`, - ], - }; - } - - logLines.push( - `Warning: "${joined}" (from the "dockerfile" field in .actor/actor.json) is not in the pushed source; falling back to the default locations.\n`, - ); + const resolved = resolveActorJsonPathField(indexed, field); + if (resolved.outcome === 'escapes-actor-root') return escapesActorRootFailure(field); + if (resolved.outcome === 'match') { + return { + outcome: 'resolved', + dockerfilePath: resolved.file.normalizedName, + logLines: [ + ...logLines, + `Using Dockerfile "${resolved.file.normalizedName}" (from the "dockerfile" field in .actor/actor.json).\n`, + ], + }; } + logLines.push(fallbackWarningLine(resolved.shownPath, 'dockerfile')); } const actorDirCandidate = normalizeEntryName(`${ACTOR_DIR}/${DEFAULT_DOCKERFILE_NAME}`); diff --git a/src/services/input-schema-location.ts b/src/services/input-schema-location.ts index a0e9720..e976db2 100644 --- a/src/services/input-schema-location.ts +++ b/src/services/input-schema-location.ts @@ -1,32 +1,44 @@ /** - * Which input schema a build carries, resolved from the version's `sourceFiles` - * (`actor-driver.md`'s "Input schema, validation and defaults"). The run-time half - defaults and - * validation - is `services/input-schema.ts`, which reads only the resolved schema, never the source - * files again. + * Which input schema a build carries, resolved from the version's pushed `sourceFiles` + * (`actor-driver.md`'s "Input schema, validation and defaults"). * - * The candidate order is `apify-cli`'s own `readInputSchema`, so a developer's local `apify - * validate-schema` and this build agree on which file is the Actor's schema. Case-insensitive matching - * (as in `services/dockerfile-location.ts`) is why the CLI's four candidates collapse to two here. - * - * Schema files are parsed as JSON5, like `.actor/actor.json` already is: every valid JSON file is - * valid JSON5, so this only ever accepts more than the platform, never less. + * The candidate order is `apify-cli`'s own `readInputSchema`, so a developer's local + * `apify validate-schema` and this build agree on which file is the Actor's schema. Case-insensitive + * matching is why the CLI's four candidates collapse to the two here. */ -import * as path from 'node:path'; import JSON5 from 'json5'; +import ajv2019Package from 'ajv/dist/2019.js'; +import { validateInputSchema } from '@apify/input_schema'; -import { normalizeEntryName } from '../driver/tar-entry-name.js'; import type { InputSchema, SourceFile } from '../storage/entities.js'; -import { describeInputSchemaDefect } from './input-schema.js'; +import { + ACTOR_DIR, + escapesActorRootMessage, + fallbackWarningLine, + findCaseInsensitive, + indexSourceFiles, + parseActorJson, + resolveActorJsonPathField, + sourceFileToText, + type IndexedFile, +} from './actor-source-files.js'; + +// AJV ships as CommonJS, so under this package's ESM resolution its class arrives as the module's +// `default`. `2019` is the build with draft-2019-09 support, which the Apify input-schema meta-schema +// requires and the plain build lacks. +const Ajv2019 = ajv2019Package.default; -const ACTOR_DIR = '.actor'; -const ACTOR_JSON_NAME = `${ACTOR_DIR}/actor.json`; /** Checked case-insensitively, so the `INPUT_SCHEMA.json` spelling matches these too. */ const DEFAULT_SCHEMA_CANDIDATES = [`${ACTOR_DIR}/input_schema.json`, 'input_schema.json'] as const; /** Why resolution failed. Each one fails the build: an Actor whose declared input contract cannot be * read must not be built with that contract silently dropped. */ export type InputSchemaResolutionFailureReason = - 'escapes-actor-root' | 'invalid-input-field' | 'unparseable-input-schema' | 'invalid-input-schema'; + | 'escapes-actor-root' + | 'invalid-input-field' + | 'unparseable-actor-json' + | 'unparseable-input-schema' + | 'invalid-input-schema'; /** `none` means the Actor declares no input schema, which is not an error. */ export type InputSchemaResolution = @@ -34,35 +46,24 @@ export type InputSchemaResolution = | { outcome: 'none'; logLines: string[] } | { outcome: 'failure'; reason: InputSchemaResolutionFailureReason; message: string }; -function sourceFileToText(file: SourceFile): string { - return file.format === 'BASE64' ? Buffer.from(file.content, 'base64').toString('utf8') : file.content; -} - -/** Exact-case match wins; otherwise the first match in `sourceFiles` order - the same rule as the - * Dockerfile lookup's. */ -function findCaseInsensitive(sourceFiles: SourceFile[], candidate: string): SourceFile | undefined { - const lowerCandidate = candidate.toLowerCase(); - let firstMatch: SourceFile | undefined; - for (const file of sourceFiles) { - const normalizedName = normalizeEntryName(file.name); - if (normalizedName.toLowerCase() !== lowerCandidate) continue; - if (normalizedName === candidate) return file; - firstMatch ??= file; +/** + * `null` for a valid input schema, the defect otherwise. Runs at build time, so a broken schema is + * reported where the developer is already looking rather than silently at every later run. + */ +export function describeInputSchemaDefect(schema: unknown): string | null { + if (schema === null || typeof schema !== 'object' || Array.isArray(schema)) { + return 'Input schema must be an object.'; + } + try { + // A fresh instance per call: AJV never evicts its internal compiled-schema map, and this runs + // once per build, not per request. + const ajv = new Ajv2019({ strict: false, unicodeRegExp: false }); + // `validateInputSchema` normalizes in place, so it must not get the stored schema itself. + validateInputSchema(ajv, structuredClone(schema) as Record); + return null; + } catch (error) { + return (error as Error).message; } - return firstMatch; -} - -/** `.actor/actor.json`'s own path is not case-folded, unlike the schema candidates. */ -function findExact(sourceFiles: SourceFile[], normalizedTarget: string): SourceFile | undefined { - return sourceFiles.find((file) => normalizeEntryName(file.name) === normalizedTarget); -} - -function escapesActorRootFailure(rawField: string): InputSchemaResolution { - return { - outcome: 'failure', - reason: 'escapes-actor-root', - message: `Input schema path "${rawField}" in .actor/actor.json points outside the Actor root directory.`, - }; } /** One place, so an inline `input` object and a schema file are held to the same standard. */ @@ -83,26 +84,34 @@ function acceptSchema(schema: unknown, source: string, logLines: string[]): Inpu }; } +function acceptSchemaFile(match: IndexedFile, source: string, logLines: string[]): InputSchemaResolution { + let parsed: unknown; + try { + parsed = JSON5.parse(sourceFileToText(match.file)); + } catch (error) { + return { + outcome: 'failure', + reason: 'unparseable-input-schema', + message: `Could not parse the input schema "${match.normalizedName}": ${(error as Error).message}`, + }; + } + return acceptSchema(parsed, source, logLines); +} + export function resolveInputSchemaLocation(sourceFiles: SourceFile[]): InputSchemaResolution { + const indexed = indexSourceFiles(sourceFiles); const logLines: string[] = []; - const actorJsonFile = findExact(sourceFiles, ACTOR_JSON_NAME); - let actorSpecification: unknown; - if (actorJsonFile) { - try { - actorSpecification = JSON5.parse(sourceFileToText(actorJsonFile)); - } catch { - // Deliberately not a failure of its own: `services/dockerfile-location.ts` runs first on the - // very same file and already fails the build with its own "Could not parse .actor/actor.json" - // message, so reporting it twice (in two different wordings) would only be noise. - actorSpecification = undefined; - } + const actorJson = parseActorJson(sourceFiles); + if (actorJson.outcome === 'unparseable') { + return { outcome: 'failure', reason: 'unparseable-actor-json', message: actorJson.message }; } + const specification = actorJson.outcome === 'parsed' ? actorJson.specification : undefined; - if (actorSpecification !== null && typeof actorSpecification === 'object' && 'input' in actorSpecification) { - const field: unknown = actorSpecification.input; + if (specification !== null && typeof specification === 'object' && 'input' in specification) { + const field: unknown = specification.input; - // An inline schema object, the other shape the Actor specification allows for this field. + // An inline schema, the other shape the Actor specification allows for this field. if (field !== null && typeof field === 'object' && !Array.isArray(field)) { return acceptSchema(field, 'the "input" field in .actor/actor.json', logLines); } @@ -115,59 +124,27 @@ export function resolveInputSchemaLocation(sourceFiles: SourceFile[]): InputSche }; } - if (field === '') { - logLines.push( - 'Warning: "" (from the "input" field in .actor/actor.json) is not in the pushed source; falling back to the default locations.\n', - ); - } else if (field.startsWith('/')) { - return escapesActorRootFailure(field); - } else { - const joined = normalizeEntryName(path.posix.join(ACTOR_DIR, field)); - if (joined === '..' || joined.startsWith('../')) { - return escapesActorRootFailure(field); - } - - const match = findCaseInsensitive(sourceFiles, joined); - if (match) { - const parsed = parseSchemaFile(match); - if (parsed.outcome === 'failure') return parsed; - return acceptSchema( - parsed.schema, - `"${normalizeEntryName(match.name)}" (the "input" field in .actor/actor.json)`, - logLines, - ); - } - - // A value naming no pushed file falls through to the default locations instead of failing - - // the same tolerance `apify-cli` shows locally (it warns and keeps looking), and the same one - // the Dockerfile field already has here. - logLines.push( - `Warning: "${joined}" (from the "input" field in .actor/actor.json) is not in the pushed source; falling back to the default locations.\n`, - ); + const resolved = resolveActorJsonPathField(indexed, field); + if (resolved.outcome === 'escapes-actor-root') { + return { + outcome: 'failure', + reason: 'escapes-actor-root', + message: escapesActorRootMessage(field, 'Input schema'), + }; } + if (resolved.outcome === 'match') { + const source = `"${resolved.file.normalizedName}" (the "input" field in .actor/actor.json)`; + return acceptSchemaFile(resolved.file, source, logLines); + } + // Falls through instead of failing - the tolerance `apify-cli` shows locally, and the one the + // Dockerfile field already has here. + logLines.push(fallbackWarningLine(resolved.shownPath, 'input')); } for (const candidate of DEFAULT_SCHEMA_CANDIDATES) { - const match = findCaseInsensitive(sourceFiles, normalizeEntryName(candidate)); - if (!match) continue; - const parsed = parseSchemaFile(match); - if (parsed.outcome === 'failure') return parsed; - return acceptSchema(parsed.schema, `"${normalizeEntryName(match.name)}"`, logLines); + const match = findCaseInsensitive(indexed, candidate); + if (match) return acceptSchemaFile(match, `"${match.normalizedName}"`, logLines); } return { outcome: 'none', logLines }; } - -type SchemaFileParse = { outcome: 'parsed'; schema: unknown } | Extract; - -function parseSchemaFile(file: SourceFile): SchemaFileParse { - try { - return { outcome: 'parsed', schema: JSON5.parse(sourceFileToText(file)) as unknown }; - } catch (error) { - return { - outcome: 'failure', - reason: 'unparseable-input-schema', - message: `Could not parse the input schema "${normalizeEntryName(file.name)}": ${(error as Error).message}`, - }; - } -} diff --git a/src/services/input-schema.ts b/src/services/input-schema.ts index f1e3662..46e4c93 100644 --- a/src/services/input-schema.ts +++ b/src/services/input-schema.ts @@ -7,111 +7,86 @@ * would answer with rather than an approximation of it. */ import ajvPackage from 'ajv'; -import ajv2019Package from 'ajv/dist/2019.js'; -import { validateInputSchema, validateInputUsingValidator } from '@apify/input_schema'; +import { validateInputUsingValidator } from '@apify/input_schema'; -import type { InputSchema } from '../storage/entities.js'; +import type { BuildRecord, InputSchema } from '../storage/entities.js'; // AJV ships as CommonJS, so under this package's ESM resolution its class arrives as the module's -// `default`. `2019` is the same library with draft-2019-09 support, which the Apify input-schema -// meta-schema requires and the plain build lacks. +// `default`. const Ajv = ajvPackage.default; -const Ajv2019 = ajv2019Package.default; type InputValidator = ReturnType['compile']>; -/** The Actor input as it travels from the API route into the run: the exact bytes stored as the run's - * `INPUT` record, with the content type they are stored under. */ +/** Stored verbatim as the run's `INPUT` record. */ export interface ActorInput { body: Buffer; contentType: string; } -/** Every way `processActorInput` can end, `ok` included - a discriminated union the API route maps to - * its own `ApiError` types, the same split `services/dev-folder.ts` uses. */ +/** `kind` is the API error type the route answers with; `message` is the real Apify API's own wording + * for the same defect (`@apify-packages/errors`'s `actor.inputNotJson` and its neighbours). */ export type InputProcessingResult = | { kind: 'ok'; input: ActorInput } - | { kind: 'not-json' } - | { kind: 'unparseable-json'; parseError: string } - | { kind: 'not-object'; actualType: string } - | { kind: 'invalid'; message: string } - | { kind: 'invalid-schema'; message: string }; - -/** The real Apify API's messages, word for word (`@apify-packages/errors`'s `actor.inputNotJson` and - * its neighbours). */ -export function describeInputProcessingFailure(result: Exclude): string { - switch (result.kind) { - case 'not-json': - return 'Actor input must have content type "application/json".'; - case 'unparseable-json': - return `Cannot parse input JSON body: ${result.parseError}`; - case 'not-object': - return `The input JSON must be object, got "${result.actualType}" instead.`; - case 'invalid': - return `Input is not valid: ${result.message}`; - case 'invalid-schema': - return `Input schema is not valid: ${result.message}`; - } -} + | { kind: 'invalid-input'; message: string } + | { kind: 'invalid-input-schema'; message: string }; /** - * `null` for a valid input schema, the defect otherwise. Runs at build time, so a broken schema is - * reported where the developer is already looking rather than silently at every later run. + * The input a run against `build` should actually start with: the caller's bytes untouched when the + * build declares no input schema, and the validated input with that schema's defaults applied when it + * does. */ -export function describeInputSchemaDefect(schema: unknown): string | null { - if (schema === null || typeof schema !== 'object' || Array.isArray(schema)) { - return 'Input schema must be an object.'; - } - try { - // A fresh instance per call: AJV never evicts its internal compiled-schema map, and this runs - // once per build, not per request. - const ajv = new Ajv2019({ strict: false, unicodeRegExp: false }); - // `validateInputSchema` normalizes in place, so it must not get the stored schema itself. - validateInputSchema(ajv, structuredClone(schema) as Record); - return null; - } catch (error) { - return (error as Error).message; - } +export function resolveBuildInput(build: BuildRecord, input: ActorInput | undefined): InputProcessingResult { + if (!build.inputSchema) return { kind: 'ok', input: input as ActorInput }; + return processActorInput(input, build.inputSchema); } /** - * The effective input a run should start with. Mirrors the platform's `processInputUsingSchema`: once - * a schema exists, an absent input is an empty object the defaults are applied to, not "no input" - so - * a call with no input still gets the defaults, and a required field with no default is still missing. + * Mirrors the platform's `processInputUsingSchema`: with a schema present, an absent input is an empty + * object the defaults are applied to, not "no input". */ export function processActorInput(input: ActorInput | undefined, schema: InputSchema): InputProcessingResult { - let parsedInput: unknown = {}; + let parsedInput: Record = {}; if (input) { - if (!isJsonContentType(input.contentType)) return { kind: 'not-json' }; + if (!isJsonContentType(input.contentType)) { + return { kind: 'invalid-input', message: 'Actor input must have content type "application/json".' }; + } + let body: unknown; try { - parsedInput = JSON.parse(input.body.toString('utf8')) as unknown; + body = JSON.parse(input.body.toString('utf8')); } catch (error) { - return { kind: 'unparseable-json', parseError: (error as Error).message }; + return { kind: 'invalid-input', message: `Cannot parse input JSON body: ${(error as Error).message}` }; } // A literal `null` body is the platform's "no input" too, not a type error. - if (parsedInput === null || parsedInput === undefined) parsedInput = {}; - if (!isPlainObject(parsedInput)) { - return { kind: 'not-object', actualType: Array.isArray(parsedInput) ? 'array' : typeof parsedInput }; + if (body !== null) { + if (!isPlainObject(body)) { + const actualType = Array.isArray(body) ? 'array' : typeof body; + return { + kind: 'invalid-input', + message: `The input JSON must be object, got "${actualType}" instead.`, + }; + } + parsedInput = body; } } - const withDefaults = mergeDefaultsFromInputSchema(parsedInput as Record, schema); - let validator; try { validator = compileInputSchemaValidator(schema); } catch (error) { // Reachable only for a schema recorded before builds validated them, or one that meta-validates // yet still will not compile. - return { kind: 'invalid-schema', message: (error as Error).message }; + return { kind: 'invalid-input-schema', message: `Input schema is not valid: ${(error as Error).message}` }; } + // `parsedInput` was parsed here and is referenced nowhere else, so the merge may own it. + const withDefaults = assignDefaults(parsedInput, schema); + // No `proxy` options: this runtime emulates no proxy groups (`unsupported.md`), so any // `apifyProxyGroups` selection is accepted, while the rest of a proxy field is still checked. const validationErrors = validateInputUsingValidator(validator, schema, withDefaults, {}); if (validationErrors.length > 0) { - // Joined, as the platform answers an API-origin run; it shortens this to the first message only - // for the web console, which has one notification line to show it in. - return { kind: 'invalid', message: validationErrors.map(({ message }) => message).join(', ') }; + // Joined, as the platform answers an API-origin run. + const message = validationErrors.map(({ message: text }) => text).join(', '); + return { kind: 'invalid-input', message: `Input is not valid: ${message}` }; } return { @@ -122,8 +97,7 @@ export function processActorInput(input: ActorInput | undefined, schema: InputSc const DEFAULT_INPUT_CONTENT_TYPE = 'application/json'; -/** `application/json`, with any parameters (`; charset=utf-8`) ignored - the same single media type the - * platform accepts for an Actor input validated against a schema. */ +/** Parameters such as `; charset=utf-8` are ignored. */ function isJsonContentType(contentType: string): boolean { return contentType.split(';')[0]?.trim().toLowerCase() === DEFAULT_INPUT_CONTENT_TYPE; } @@ -133,29 +107,23 @@ function isPlainObject(value: unknown): value is Record { } /** - * A port of the platform's `mergeDefaultsFromInputSchema`, recursion into nested objects and array - * items included. Exported for direct testing; run-start callers go through `processActorInput`. + * A port of the platform's `mergeDefaultsFromInputSchema`. `values` is filled in place and returned: + * every caller parsed it itself, and cloning it doubles the peak heap of a large input. */ -export function mergeDefaultsFromInputSchema( - values: Record, - schema: InputSchema, -): Record { +export function assignDefaults(values: Record, schema: InputSchema): Record { const properties = isPlainObject(schema.properties) ? schema.properties : {}; const defaults: Record = {}; for (const [key, fieldSchema] of Object.entries(properties)) { defaults[key] = extractDefaults(fieldSchema); } - // Cloned so no reference into the stored schema is ever assigned into the returned input, where a - // later mutation of either would reach the other. - const clonedDefaults = structuredClone(defaults); - const target = structuredClone(values); - assignRecursively(target, clonedDefaults, schema); - return target; + // `extractDefaults` returns references into the stored schema, so the values assigned into the + // input must be copies - otherwise a later mutation of either would reach the other. + assignRecursively(values, structuredClone(defaults), schema); + return values; } -/** The `default` of a field, plus - for an object field with `properties` - the defaults of its nested - * fields, with the field's own `default` winning over them key by key. */ +/** A field's own `default` wins, key by key, over the defaults of its nested fields. */ function extractDefaults(fieldSchema: unknown): unknown { if (!isPlainObject(fieldSchema)) return undefined; @@ -172,12 +140,6 @@ function extractDefaults(fieldSchema: unknown): unknown { return rootValue; } -/** The defaults for the items of an array field, from its `items` sub-schema. */ -function extractArrayItemDefaults(fieldSchema: unknown): unknown { - if (!isPlainObject(fieldSchema) || !('items' in fieldSchema)) return undefined; - return extractDefaults(fieldSchema.items); -} - /** `fillDefinedValues` is set for array items alone, where the platform fills an item's fields even * though the item itself is present. */ function assignRecursively( @@ -200,22 +162,21 @@ function assignRecursively( } if (isPlainObject(currentValue)) { - const nestedDefaults = isPlainObject(defaultValue) - ? defaultValue - : ((extractDefaults(fieldSchema) as Record | undefined) ?? {}); - assignRecursively( - currentValue, - // A present nested object otherwise keeps exactly the keys it came with. - fillDefinedValues ? nestedDefaults : {}, - fieldSchema as InputSchema | undefined, - ); + // A present nested object keeps exactly the keys it came with, so outside an array item + // there is nothing to fill and no defaults to extract. + const nestedDefaults = fillDefinedValues + ? isPlainObject(defaultValue) + ? defaultValue + : ((extractDefaults(fieldSchema) as Record | undefined) ?? {}) + : {}; + assignRecursively(currentValue, nestedDefaults, fieldSchema as InputSchema | undefined); continue; } if (Array.isArray(currentValue)) { - const itemDefaults = extractArrayItemDefaults(fieldSchema); - if (!isPlainObject(itemDefaults)) continue; const itemSchema = isPlainObject(fieldSchema) ? fieldSchema.items : undefined; + const itemDefaults = extractDefaults(itemSchema); + if (!isPlainObject(itemDefaults)) continue; for (const item of currentValue) { if (!isPlainObject(item)) continue; assignRecursively(item, itemDefaults, itemSchema as InputSchema | undefined, true); @@ -224,8 +185,7 @@ function assignRecursively( } } -/** The one `lodash.merge` behaviour `extractDefaults` needs, without the dependency: objects merge - * key by key, arrays and scalars replace wholesale. */ +/** The one `lodash.merge` behaviour `extractDefaults` needs, without the dependency. */ function deepMerge(base: Record, overrides: Record): Record { const merged: Record = { ...base }; for (const [key, value] of Object.entries(overrides)) { @@ -236,9 +196,9 @@ function deepMerge(base: Record, overrides: Record(); const VALIDATOR_CACHE_MAX_ENTRIES = 100; @@ -259,10 +219,8 @@ function compileInputSchemaValidator(schema: InputSchema): InputValidator { } /** - * The platform's `getAjvValidator` preparation, ported: a required field with a default is optional - * (there is always a value for it by then), a required array must hold at least one item, and - * `$schema` is dropped - AJV would otherwise try to fetch the Apify meta-schema it names and fail to - * compile at all. + * Ported from the platform's `getAjvValidator`. `$schema` is dropped because AJV would otherwise try + * to fetch the Apify meta-schema it names and fail to compile the schema at all. */ function prepareSchemaForValidation(schema: InputSchema): Record { const copy = structuredClone(schema) as Record; @@ -284,8 +242,3 @@ function prepareSchemaForValidation(schema: InputSchema): Record> { + return spawnSync('npx', ['-y', '-p', 'apify-cli', 'apify', ...args], { + cwd: options.cwd, + env: options.env, + encoding: 'utf8', + }); +} + /** * Like `apify()`, but returns stdout AND stderr combined. The CLI writes human-readable output — * including the log content of `apify runs log` (see `outputJobLog`'s `process.stderr.write`) — to * stderr, keeping stdout for machine-readable payloads, so log-content assertions must read stderr. */ -export function apifyAllOutput(args: string[], options: { cwd: string; env: NodeJS.ProcessEnv }): string { - const result = spawnSync('npx', ['-y', '-p', 'apify-cli', 'apify', ...args], { - cwd: options.cwd, - env: options.env, - encoding: 'utf8', - }); +export function apifyAllOutput(args: string[], options: ApifyOptions): string { + const result = spawnApify(args, options); if (result.status !== 0) { throw new Error(`apify ${args.join(' ')} exited with ${result.status}:\n${result.stderr}`); } return `${result.stdout}\n${result.stderr}`; } -/** - * Like `apifyAllOutput()`, but for a command that is *expected* to fail: returns its combined output - * instead of throwing, and fails the test if the command unexpectedly succeeded. Used for the run that - * the runtime must refuse (an input its Actor's input schema rejects), where the CLI's non-zero exit is - * itself part of what the test asserts. - */ -export function apifyExpectingFailure(args: string[], options: { cwd: string; env: NodeJS.ProcessEnv }): string { - const result = spawnSync('npx', ['-y', '-p', 'apify-cli', 'apify', ...args], { - cwd: options.cwd, - env: options.env, - encoding: 'utf8', - }); +/** Like `apifyAllOutput()`, but requires a non-zero exit. */ +export function apifyExpectingFailure(args: string[], options: ApifyOptions): string { + const result = spawnApify(args, options); if (result.status === 0) { throw new Error(`apify ${args.join(' ')} was expected to fail, but succeeded:\n${result.stdout}`); } diff --git a/test/integration/input-schema.test.ts b/test/integration/input-schema.test.ts index b27fafc..fd0509c 100644 --- a/test/integration/input-schema.test.ts +++ b/test/integration/input-schema.test.ts @@ -8,8 +8,8 @@ import { afterEach, beforeEach, describe, expect, it } from 'vitest'; import { fixedBuildOutcomeDriver, startTestServer, type TestServerHandle } from './helpers/test-server.js'; import { getRegistries } from '../../src/storage/registries.js'; +import { generateId } from '../../src/storage/ids.js'; import { recordTaggedBuild, updateActor } from '../../src/services/actors.js'; -import { resetInputSchemaValidatorCacheForTests } from '../../src/services/input-schema.js'; import { runBuildInBackground } from '../../src/services/builds.js'; import type { InputSchema, SourceFile } from '../../src/storage/entities.js'; @@ -44,28 +44,33 @@ describe('input validation and defaults (via real apify-client)', () => { }); afterEach(async () => { - resetInputSchemaValidatorCacheForTests(); await server.close(); }); /** Seeds a tagged, successful build - with or without an input schema - for `actorId`. */ - async function seedBuild(actorId: string, userId: string, inputSchema?: InputSchema): Promise { + async function seedBuild( + actorId: string, + userId: string, + inputSchema?: InputSchema, + tag = 'latest', + ): Promise { const { builds } = getRegistries(); - const buildId = `seeded${Math.random().toString(36).slice(2, 13)}`.slice(0, 17); + const buildId = generateId(); + const buildNumber = '0.0.1'; await builds.set(buildId, { id: buildId, userId, actorId, versionNumber: '0.0', - buildNumber: '0.0.1', - tag: 'latest', + buildNumber, + tag, status: 'SUCCEEDED', startedAt: new Date().toISOString(), finishedAt: new Date().toISOString(), - imageId: 'fake-image:latest', + imageId: `fake-image:${tag}`, ...(inputSchema ? { inputSchema } : {}), }); - await updateActor(actorId, (current) => recordTaggedBuild(current, 'latest', buildId, '0.0.1')); + await updateActor(actorId, (current) => recordTaggedBuild(current, tag, buildId, buildNumber)); } async function storedInput(runId: string): Promise { @@ -111,24 +116,6 @@ describe('input validation and defaults (via real apify-client)', () => { expect(runs.items).toHaveLength(0); }); - it('rejects a missing required field that has no default', async () => { - const actor = await server.client.actors().create({ name: 'required-field-actor' }); - await seedBuild(actor.id, actor.userId, { - ...SAMPLE_SCHEMA, - properties: { - ...SAMPLE_SCHEMA.properties, - token: { title: 'Token', type: 'string', editor: 'textfield', description: 'A token.' }, - }, - required: ['token'], - }); - - await expect(server.client.actor(actor.id).start({ maxPages: 1 })).rejects.toMatchObject({ - statusCode: 400, - type: 'invalid-input', - message: 'Input is not valid: Field input.token is required', - }); - }); - it('rejects a non-object and a non-JSON body against a schema', async () => { const actor = await server.client.actors().create({ name: 'malformed-input-actor' }); await seedBuild(actor.id, actor.userId, SAMPLE_SCHEMA); @@ -177,24 +164,10 @@ describe('input validation and defaults (via real apify-client)', () => { it('validates against the schema of the build the run actually resolved, not the newest one', async () => { const actor = await server.client.actors().create({ name: 'two-tag-actor' }); - const { builds } = getRegistries(); - // `latest` demands maxPages >= 1; `beta`, built afterwards, has no schema at all. + // `latest` demands maxPages >= 1; `beta` has no schema at all. await seedBuild(actor.id, actor.userId, SAMPLE_SCHEMA); - const betaBuildId = 'betaBuildId12345b'; - await builds.set(betaBuildId, { - id: betaBuildId, - userId: actor.userId, - actorId: actor.id, - versionNumber: '0.1', - buildNumber: '0.1.1', - tag: 'beta', - status: 'SUCCEEDED', - startedAt: new Date().toISOString(), - finishedAt: new Date().toISOString(), - imageId: 'fake-image:beta', - }); - await updateActor(actor.id, (current) => recordTaggedBuild(current, 'beta', betaBuildId, '0.1.1')); + await seedBuild(actor.id, actor.userId, undefined, 'beta'); await expect(server.client.actor(actor.id).start({ maxPages: 0 })).rejects.toMatchObject({ statusCode: 400, diff --git a/test/unit/input-schema-location.test.ts b/test/unit/input-schema-location.test.ts index 07aa738..efd6868 100644 --- a/test/unit/input-schema-location.test.ts +++ b/test/unit/input-schema-location.test.ts @@ -1,6 +1,6 @@ import { describe, expect, it } from 'vitest'; -import { resolveInputSchemaLocation } from '../../src/services/input-schema-location.js'; +import { resolveInputSchemaLocation, type InputSchemaResolution } from '../../src/services/input-schema-location.js'; import type { SourceFile } from '../../src/storage/entities.js'; /** A minimal but genuinely valid input schema - every field the Apify meta-schema demands is present, @@ -25,102 +25,101 @@ function json(name: string, value: unknown): SourceFile { return text(name, JSON.stringify(value)); } +function expectResolved(resolution: InputSchemaResolution): Extract { + if (resolution.outcome !== 'resolved') throw new Error(`expected a resolved schema, got ${resolution.outcome}`); + return resolution; +} + +function expectFailure(resolution: InputSchemaResolution): Extract { + if (resolution.outcome !== 'failure') throw new Error(`expected a failure, got ${resolution.outcome}`); + return resolution; +} + describe('resolveInputSchemaLocation', () => { it('takes an inline schema object from the "input" field of .actor/actor.json', () => { - const resolution = resolveInputSchemaLocation([ - json('.actor/actor.json', { actorSpecification: 1, name: 'a', input: validSchema() }), - json('.actor/input_schema.json', validSchema({ title: 'The file, which must lose' })), - ]); - - expect(resolution.outcome).toBe('resolved'); - if (resolution.outcome !== 'resolved') return; + const resolution = expectResolved( + resolveInputSchemaLocation([ + json('.actor/actor.json', { actorSpecification: 1, name: 'a', input: validSchema() }), + json('.actor/input_schema.json', validSchema({ title: 'The file, which must lose' })), + ]), + ); expect(resolution.schema.title).toBe('Sample input'); expect(resolution.source).toContain('"input" field'); expect(resolution.logLines.join('')).toContain('Using the input schema from'); }); it('follows a path in the "input" field, resolved relative to .actor/', () => { - const resolution = resolveInputSchemaLocation([ - json('.actor/actor.json', { actorSpecification: 1, input: './schemas/custom.json' }), - json('.actor/schemas/custom.json', validSchema({ title: 'From the named file' })), - ]); - - expect(resolution.outcome).toBe('resolved'); - if (resolution.outcome !== 'resolved') return; + const resolution = expectResolved( + resolveInputSchemaLocation([ + json('.actor/actor.json', { actorSpecification: 1, input: './schemas/custom.json' }), + json('.actor/schemas/custom.json', validSchema({ title: 'From the named file' })), + ]), + ); expect(resolution.schema.title).toBe('From the named file'); expect(resolution.source).toContain('.actor/schemas/custom.json'); }); it('falls back to the default locations, with a warning, when the named file is not in the pushed source', () => { - const resolution = resolveInputSchemaLocation([ - json('.actor/actor.json', { actorSpecification: 1, input: './missing.json' }), - json('.actor/input_schema.json', validSchema({ title: 'The fallback' })), - ]); - - expect(resolution.outcome).toBe('resolved'); - if (resolution.outcome !== 'resolved') return; + const resolution = expectResolved( + resolveInputSchemaLocation([ + json('.actor/actor.json', { actorSpecification: 1, input: './missing.json' }), + json('.actor/input_schema.json', validSchema({ title: 'The fallback' })), + ]), + ); expect(resolution.schema.title).toBe('The fallback'); expect(resolution.logLines.join('')).toContain('is not in the pushed source'); }); it('treats an empty "input" field as "not found", falling through with a warning', () => { - const resolution = resolveInputSchemaLocation([ - json('.actor/actor.json', { actorSpecification: 1, input: '' }), - json('input_schema.json', validSchema({ title: 'Root fallback' })), - ]); - - expect(resolution.outcome).toBe('resolved'); - if (resolution.outcome !== 'resolved') return; + const resolution = expectResolved( + resolveInputSchemaLocation([ + json('.actor/actor.json', { actorSpecification: 1, input: '' }), + json('input_schema.json', validSchema({ title: 'Root fallback' })), + ]), + ); expect(resolution.schema.title).toBe('Root fallback'); expect(resolution.logLines.join('')).toContain('falling back to the default locations'); }); it('rejects an "input" path that escapes the Actor root, absolute or relative', () => { for (const field of ['/etc/passwd', '../../outside.json']) { - const resolution = resolveInputSchemaLocation([ - json('.actor/actor.json', { actorSpecification: 1, input: field }), - ]); - expect(resolution.outcome).toBe('failure'); - if (resolution.outcome !== 'failure') return; + const resolution = expectFailure( + resolveInputSchemaLocation([json('.actor/actor.json', { actorSpecification: 1, input: field })]), + ); expect(resolution.reason).toBe('escapes-actor-root'); expect(resolution.message).toContain('points outside the Actor root directory'); } }); it('rejects an "input" field that is neither a string nor an object', () => { - const resolution = resolveInputSchemaLocation([ - json('.actor/actor.json', { actorSpecification: 1, input: 42 }), - ]); - - expect(resolution.outcome).toBe('failure'); - if (resolution.outcome !== 'failure') return; + const resolution = expectFailure( + resolveInputSchemaLocation([json('.actor/actor.json', { actorSpecification: 1, input: 42 })]), + ); expect(resolution.reason).toBe('invalid-input-field'); expect(resolution.message).toBe('.actor/actor.json has invalid format: "input" must be a string or an object.'); }); it('prefers .actor/INPUT_SCHEMA.json over the Actor root one, matching case-insensitively', () => { - const resolution = resolveInputSchemaLocation([ - json('INPUT_SCHEMA.json', validSchema({ title: 'Root' })), - json('.actor/INPUT_SCHEMA.json', validSchema({ title: 'Actor dir' })), - ]); - - expect(resolution.outcome).toBe('resolved'); - if (resolution.outcome !== 'resolved') return; + const resolution = expectResolved( + resolveInputSchemaLocation([ + json('INPUT_SCHEMA.json', validSchema({ title: 'Root' })), + json('.actor/INPUT_SCHEMA.json', validSchema({ title: 'Actor dir' })), + ]), + ); expect(resolution.schema.title).toBe('Actor dir'); expect(resolution.source).toContain('.actor/INPUT_SCHEMA.json'); }); it('reads a BASE64-encoded schema file the same as a TEXT one', () => { - const resolution = resolveInputSchemaLocation([ - { - name: '.actor/input_schema.json', - format: 'BASE64', - content: Buffer.from(JSON.stringify(validSchema({ title: 'Encoded' })), 'utf8').toString('base64'), - }, - ]); - - expect(resolution.outcome).toBe('resolved'); - if (resolution.outcome !== 'resolved') return; + const resolution = expectResolved( + resolveInputSchemaLocation([ + { + name: '.actor/input_schema.json', + format: 'BASE64', + content: Buffer.from(JSON.stringify(validSchema({ title: 'Encoded' })), 'utf8').toString('base64'), + }, + ]), + ); expect(resolution.schema.title).toBe('Encoded'); }); @@ -130,60 +129,51 @@ describe('resolveInputSchemaLocation', () => { }); it('fails on a schema file that cannot be parsed', () => { - const resolution = resolveInputSchemaLocation([text('.actor/input_schema.json', '{ "title": ')]); - - expect(resolution.outcome).toBe('failure'); - if (resolution.outcome !== 'failure') return; + const resolution = expectFailure(resolveInputSchemaLocation([text('.actor/input_schema.json', '{ "title": ')])); expect(resolution.reason).toBe('unparseable-input-schema'); expect(resolution.message).toContain('Could not parse the input schema ".actor/input_schema.json"'); }); it("fails on a schema the platform's own meta-schema rejects, naming where it came from", () => { - const resolution = resolveInputSchemaLocation([ - json('.actor/input_schema.json', { - title: 'Missing a field description', - type: 'object', - schemaVersion: 1, - properties: { maxPages: { title: 'Max pages', type: 'integer' } }, - }), - ]); - - expect(resolution.outcome).toBe('failure'); - if (resolution.outcome !== 'failure') return; + const resolution = expectFailure( + resolveInputSchemaLocation([ + json('.actor/input_schema.json', { + title: 'Missing a field description', + type: 'object', + schemaVersion: 1, + properties: { maxPages: { title: 'Max pages', type: 'integer' } }, + }), + ]), + ); expect(resolution.reason).toBe('invalid-input-schema'); expect(resolution.message).toContain('.actor/input_schema.json'); expect(resolution.message).toContain('is not valid'); }); it('rejects a schema that is not an object at all', () => { - const resolution = resolveInputSchemaLocation([json('.actor/input_schema.json', ['not', 'a', 'schema'])]); - - expect(resolution.outcome).toBe('failure'); - if (resolution.outcome !== 'failure') return; + const resolution = expectFailure( + resolveInputSchemaLocation([json('.actor/input_schema.json', ['not', 'a', 'schema'])]), + ); expect(resolution.reason).toBe('invalid-input-schema'); }); - it('leaves an unparseable .actor/actor.json to the Dockerfile resolver, falling through to the default locations', () => { - // `services/dockerfile-location.ts` runs first on the same file and already fails that build with - // its own message; this must not add a second, differently-worded report of the same defect. - const resolution = resolveInputSchemaLocation([ - text('.actor/actor.json', '{ not json at all'), - json('.actor/input_schema.json', validSchema({ title: 'Still found' })), - ]); - - expect(resolution.outcome).toBe('resolved'); - if (resolution.outcome !== 'resolved') return; - expect(resolution.schema.title).toBe('Still found'); + it('fails on an unparseable .actor/actor.json rather than reporting no schema', () => { + const resolution = expectFailure( + resolveInputSchemaLocation([ + text('.actor/actor.json', '{ not json at all'), + json('.actor/input_schema.json', validSchema()), + ]), + ); + expect(resolution.reason).toBe('unparseable-actor-json'); }); it('parses .actor/actor.json as JSON5, like the Dockerfile lookup already does', () => { - const resolution = resolveInputSchemaLocation([ - text('.actor/actor.json', "{ actorSpecification: 1, input: './input_schema.json' /* a comment */ }"), - json('.actor/input_schema.json', validSchema({ title: 'Via JSON5' })), - ]); - - expect(resolution.outcome).toBe('resolved'); - if (resolution.outcome !== 'resolved') return; + const resolution = expectResolved( + resolveInputSchemaLocation([ + text('.actor/actor.json', "{ actorSpecification: 1, input: './input_schema.json' /* a comment */ }"), + json('.actor/input_schema.json', validSchema({ title: 'Via JSON5' })), + ]), + ); expect(resolution.schema.title).toBe('Via JSON5'); }); diff --git a/test/unit/input-schema.test.ts b/test/unit/input-schema.test.ts index a583a68..bbdf45d 100644 --- a/test/unit/input-schema.test.ts +++ b/test/unit/input-schema.test.ts @@ -1,12 +1,7 @@ -import { afterEach, describe, expect, it } from 'vitest'; +import { describe, expect, it } from 'vitest'; -import { - describeInputProcessingFailure, - describeInputSchemaDefect, - mergeDefaultsFromInputSchema, - processActorInput, - resetInputSchemaValidatorCacheForTests, -} from '../../src/services/input-schema.js'; +import { assignDefaults, processActorInput } from '../../src/services/input-schema.js'; +import { describeInputSchemaDefect } from '../../src/services/input-schema-location.js'; import type { InputSchema } from '../../src/storage/entities.js'; const SAMPLE_SCHEMA: InputSchema = { @@ -43,10 +38,6 @@ function effectiveInput(result: ReturnType): Record; } -afterEach(() => { - resetInputSchemaValidatorCacheForTests(); -}); - describe('processActorInput - defaults', () => { it('fills every missing field from the schema and leaves provided ones alone', () => { expect(effectiveInput(processActorInput(jsonInput({ maxPages: 7 }), SAMPLE_SCHEMA))).toEqual({ @@ -95,7 +86,7 @@ describe('processActorInput - defaults', () => { }); }); -describe('mergeDefaultsFromInputSchema', () => { +describe('assignDefaults', () => { const NESTED_SCHEMA: InputSchema = { title: 'Nested', type: 'object', @@ -122,28 +113,28 @@ describe('mergeDefaultsFromInputSchema', () => { }; it("builds a nested object out of its fields' defaults when the whole object is missing", () => { - expect(mergeDefaultsFromInputSchema({}, NESTED_SCHEMA)).toEqual({ + expect(assignDefaults({}, NESTED_SCHEMA)).toEqual({ config: { retries: 3, verbose: false }, }); }); it('leaves a provided nested object exactly as it came, adding no keys to it', () => { - expect(mergeDefaultsFromInputSchema({ config: { retries: 9 } }, NESTED_SCHEMA)).toEqual({ + expect(assignDefaults({ config: { retries: 9 } }, NESTED_SCHEMA)).toEqual({ config: { retries: 9 }, }); }); it('fills defaults into every item of a provided array, without overwriting what the item has', () => { - expect(mergeDefaultsFromInputSchema({ items: [{}, { keep: false }] }, NESTED_SCHEMA)).toEqual({ + expect(assignDefaults({ items: [{}, { keep: false }] }, NESTED_SCHEMA)).toEqual({ config: { retries: 3, verbose: false }, items: [{ keep: true }, { keep: false }], }); }); it('never hands out a reference into the schema, so a later edit of the input cannot corrupt it', () => { - const merged = mergeDefaultsFromInputSchema({}, NESTED_SCHEMA) as { config: Record }; + const merged = assignDefaults({}, NESTED_SCHEMA) as { config: Record }; merged.config.retries = 999; - expect(mergeDefaultsFromInputSchema({}, NESTED_SCHEMA)).toEqual({ config: { retries: 3, verbose: false } }); + expect(assignDefaults({}, NESTED_SCHEMA)).toEqual({ config: { retries: 3, verbose: false } }); }); it("applies the field's own default over the defaults of its nested fields", () => { @@ -165,23 +156,19 @@ describe('mergeDefaultsFromInputSchema', () => { }, }, }; - expect(mergeDefaultsFromInputSchema({}, schema)).toEqual({ config: { retries: 10, verbose: false } }); + expect(assignDefaults({}, schema)).toEqual({ config: { retries: 10, verbose: false } }); }); }); describe('processActorInput - validation', () => { - it("reports the platform's message for a value below a minimum", () => { - const result = processActorInput(jsonInput({ maxPages: 0 }), SAMPLE_SCHEMA); - expect(result).toEqual({ kind: 'invalid', message: 'Field input.maxPages must be >= 1' }); - expect(describeInputProcessingFailure(result as never)).toBe( - 'Input is not valid: Field input.maxPages must be >= 1', - ); - }); - - it("reports the platform's message for a value of the wrong type", () => { + it("reports the platform's own message for a rejected value", () => { + expect(processActorInput(jsonInput({ maxPages: 0 }), SAMPLE_SCHEMA)).toEqual({ + kind: 'invalid-input', + message: 'Input is not valid: Field input.maxPages must be >= 1', + }); expect(processActorInput(jsonInput({ label: 5 }), SAMPLE_SCHEMA)).toEqual({ - kind: 'invalid', - message: 'Field input.label must be string', + kind: 'invalid-input', + message: 'Input is not valid: Field input.label must be string', }); }); @@ -199,8 +186,8 @@ describe('processActorInput - validation', () => { required: ['startUrl', 'label'], }; expect(processActorInput(undefined, schema)).toEqual({ - kind: 'invalid', - message: 'Field input.label is required', + kind: 'invalid-input', + message: 'Input is not valid: Field input.label is required', }); }); @@ -222,8 +209,8 @@ describe('processActorInput - validation', () => { }, }; const result = processActorInput(jsonInput({ a: 1, startUrls: [{ url: 'not a url' }] }), schema); - expect(result.kind).toBe('invalid'); - if (result.kind !== 'invalid') return; + expect(result.kind).toBe('invalid-input'); + if (result.kind !== 'invalid-input') return; expect(result.message).toContain('Field input.a must be >= 10'); expect(result.message).toContain('input.startUrls'); expect(result.message).toContain(', '); @@ -244,31 +231,12 @@ describe('processActorInput - validation', () => { }, required: ['startUrls'], }; - expect(processActorInput(jsonInput({ startUrls: [] }), schema)).toMatchObject({ kind: 'invalid' }); + expect(processActorInput(jsonInput({ startUrls: [] }), schema)).toMatchObject({ kind: 'invalid-input' }); expect( effectiveInput(processActorInput(jsonInput({ startUrls: [{ url: 'https://crawlee.dev/' }] }), schema)), ).toEqual({ startUrls: [{ url: 'https://crawlee.dev/' }] }); }); - it('runs the checks AJV alone cannot express, such as requestListSources entries', () => { - const schema: InputSchema = { - title: 'Sources', - type: 'object', - schemaVersion: 1, - properties: { - startUrls: { - title: 'Start URLs', - type: 'array', - editor: 'requestListSources', - description: 'Where to start.', - }, - }, - }; - expect(processActorInput(jsonInput({ startUrls: [{ url: 'not a url' }] }), schema)).toMatchObject({ - kind: 'invalid', - }); - }); - it('accepts any apifyProxyGroups selection, since proxy groups are not emulated locally', () => { const schema: InputSchema = { title: 'Proxy', @@ -293,7 +261,7 @@ describe('processActorInput - validation', () => { jsonInput({ proxyConfiguration: { useApifyProxy: false, proxyUrls: ['nonsense'] } }), schema, ), - ).toMatchObject({ kind: 'invalid' }); + ).toMatchObject({ kind: 'invalid-input' }); }); it('compiles a schema carrying the $schema field templates ship with', () => { @@ -304,11 +272,9 @@ describe('processActorInput - validation', () => { describe('processActorInput - malformed requests', () => { it('rejects a body that is not sent as JSON', () => { - const result = processActorInput({ body: Buffer.from('{}', 'utf8'), contentType: 'text/plain' }, SAMPLE_SCHEMA); - expect(result).toEqual({ kind: 'not-json' }); - expect(describeInputProcessingFailure(result as never)).toBe( - 'Actor input must have content type "application/json".', - ); + expect( + processActorInput({ body: Buffer.from('{}', 'utf8'), contentType: 'text/plain' }, SAMPLE_SCHEMA), + ).toEqual({ kind: 'invalid-input', message: 'Actor input must have content type "application/json".' }); }); it('rejects a body that is not parseable JSON', () => { @@ -316,28 +282,28 @@ describe('processActorInput - malformed requests', () => { { body: Buffer.from('{oops', 'utf8'), contentType: 'application/json' }, SAMPLE_SCHEMA, ); - expect(result.kind).toBe('unparseable-json'); - expect(describeInputProcessingFailure(result as never)).toContain('Cannot parse input JSON body:'); + expect(result.kind).toBe('invalid-input'); + if (result.kind === 'ok') return; + expect(result.message).toContain('Cannot parse input JSON body:'); }); it('rejects a JSON body that is not an object, naming what it got instead', () => { - const array = processActorInput(jsonInput([1, 2]), SAMPLE_SCHEMA); - expect(array).toEqual({ kind: 'not-object', actualType: 'array' }); - expect(describeInputProcessingFailure(array as never)).toBe( - 'The input JSON must be object, got "array" instead.', - ); - + expect(processActorInput(jsonInput([1, 2]), SAMPLE_SCHEMA)).toEqual({ + kind: 'invalid-input', + message: 'The input JSON must be object, got "array" instead.', + }); expect(processActorInput(jsonInput('a string'), SAMPLE_SCHEMA)).toEqual({ - kind: 'not-object', - actualType: 'string', + kind: 'invalid-input', + message: 'The input JSON must be object, got "string" instead.', }); }); it('reports a schema that cannot be compiled at all, rather than throwing', () => { const broken: InputSchema = { title: 'Broken', type: 'object', properties: { a: { type: 'not-a-type' } } }; const result = processActorInput(jsonInput({}), broken); - expect(result.kind).toBe('invalid-schema'); - expect(describeInputProcessingFailure(result as never)).toContain('Input schema is not valid:'); + expect(result.kind).toBe('invalid-input-schema'); + if (result.kind === 'ok') return; + expect(result.message).toContain('Input schema is not valid:'); }); }); From 177df3f9df5a6e2b85b31931b91db199d917e382 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 23 Sep 2026 09:43:26 +0000 Subject: [PATCH 05/14] docs: Rewrite the input-schema requirements to the contribution conventions The sections described how the feature is implemented rather than what it does for a user: the schema lookup was explained in terms of what apify-cli does internally, and the API section talked about stored schemas being compiled. Both now state the behaviour and nothing else. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_011U9yUTUn78bqGDn5obYSXs --- requirements/actor-driver.md | 48 +++++++++++++++++------------------- requirements/api.md | 12 ++++----- requirements/test.md | 2 +- 3 files changed, 30 insertions(+), 32 deletions(-) diff --git a/requirements/actor-driver.md b/requirements/actor-driver.md index 084033b..a69b52d 100644 --- a/requirements/actor-driver.md +++ b/requirements/actor-driver.md @@ -36,31 +36,29 @@ # Input schema, validation and defaults -- **An Actor's input schema is read at build time**, and every run of that build has its input - validated against it with the schema's defaults applied. A build with no input schema takes every - input exactly as the caller sent it. -- The schema is looked for where `apify-cli` looks for it locally, stopping at the first hit: the - `input` field of `.actor/actor.json` (an inline schema, or a path relative to `.actor/`), then - `.actor/INPUT_SCHEMA.json`, then `INPUT_SCHEMA.json` at the Actor root. Matching is case-insensitive, - exact-case wins ties, and every outcome is stated in the build log. -- **A build fails** when the `input` field names a path outside the Actor root or a value that is - neither a path nor a schema, when the schema cannot be parsed, or when it is not a valid Apify input - schema - with the reason in the build log and the build's status message. A path naming no pushed - file only warns, and the default locations are tried instead. -- **A run with a rejected input never starts**: no run record, no container, and a `400` naming the - offending field (`api.md`). -- **Defaults are filled in** for every field the input leaves out, nested objects and array items - included, and the run's `INPUT` record holds the result - so a run started with no input at all - still runs on the schema's defaults. A required field that has a default is satisfied by it; a - required array must hold at least one item. -- A run is validated against the schema of the build it resolved, never another tag's. An Actor run - from a registered dev folder therefore uses its last build's schema: unlike a source edit, an edited - input schema needs an `apify push` to take effect. -- **Known differences from the platform**: Apify Proxy group availability is not checked, since no - proxy groups are emulated here - any `apifyProxyGroups` selection is accepted, while a `proxy` - field's shape, custom proxy URLs and country code still are. Encrypted secret input fields stay - unsupported (`unsupported.md`), so an `isSecret` field's value is validated as the plain value it - locally is. +- **An Actor's input schema is read when the Actor is built**, and every run of that build has its + input validated against it with the schema's defaults applied. A build with no input schema takes + every input exactly as the caller sent it. +- The schema is the `input` field of `.actor/actor.json` - an inline schema, or a path relative to + `.actor/` - else `.actor/INPUT_SCHEMA.json`, else `INPUT_SCHEMA.json` at the Actor root, whichever + comes first. Matching is case-insensitive, exact-case wins ties, and every outcome is stated in the + build log. +- **A build fails**, with the reason in its log and status message, when the `input` field is neither + a path nor a schema or points outside the Actor root, when the schema cannot be parsed, or when it + is not a valid Apify input schema. A path naming no pushed file only warns; the default locations + are tried instead. +- **A run whose input the schema rejects never starts**: no run is created, and the call is answered + `400` naming the offending field (`api.md`). +- **Every field the input leaves out is filled from the schema**, nested objects and array items + included, and the run's `INPUT` holds the result - so a run started with no input at all runs on + the schema's defaults. A required field that has a default is satisfied by it; a required array + must hold at least one item. +- A run is validated against the schema of the build it runs, never another tag's. An Actor running + from a registered dev folder therefore uses its last build's schema: unlike a source edit, an + edited input schema takes effect only after `apify push`. +- **Known differences from the platform**: Apify Proxy group availability is not checked, so any + `apifyProxyGroups` selection is accepted; a `proxy` field's shape, its custom proxy URLs and its + country code still are. Encrypted secret input fields stay unsupported (`unsupported.md`). # Bind mount volumes with Actor source code diff --git a/requirements/api.md b/requirements/api.md index 36cc43f..aa39b93 100644 --- a/requirements/api.md +++ b/requirements/api.md @@ -17,12 +17,12 @@ - `DELETE /v2/actor-builds/:buildId` and `DELETE /v2/actor-runs/:runId` on a **non-terminal** build/run are rejected, not aborted-then-deleted: `400` with error type `deleting-unfinished-build` (builds) or `cannot-remove-running-run` (runs), matching the Apify platform. -- `POST /v2/actors/:actorId/runs` validates the input against the resolved build's input schema, when - it has one (`actor-driver.md`), and rejects it with `400` and the platform's own error types and - messages, starting nothing: `invalid-input` for a body that is not `application/json`, is not - parseable JSON, is not a JSON object, or that the schema rejects (every validation error, - comma-separated); `invalid-input-schema` for a stored schema that cannot be compiled at all. A build - with no input schema accepts any body, unvalidated. +- `POST /v2/actors/:actorId/runs` validates the input against the input schema of the build it + resolved, when that build has one (`actor-driver.md`), and starts nothing when it does not pass: + `400` `invalid-input` for a body that is not `application/json`, is not parseable JSON, is not a + JSON object, or that the schema rejects, naming every offending field; `400` `invalid-input-schema` + when the Actor's own schema is not valid. Both messages match the Apify platform's. A build with no + input schema accepts any body, unvalidated. - Four endpoints are exceptions to the `{data}` envelope: - `GET /v2/logs/:buildOrRunId` (and its `actor-builds`/`actor-runs` aliases): the body is plain text, never `{data}`-wrapped, matching apify-client-js's `log().get()`. diff --git a/requirements/test.md b/requirements/test.md index ee4d19b..5af7c7d 100644 --- a/requirements/test.md +++ b/requirements/test.md @@ -36,7 +36,7 @@ Test case must verify full Actor development flow: - Push and build Actor in local actor runtime `apify push` - Run each sample Actor in the local actor runtime with `apify call --input '{"maxPages":N}'` for at least two different values of `N`, waiting for each run to finish - Assert via `apify datasets info ` that the default dataset's `itemCount` tracks `N` - the assertion is input-dependent, not just "some items exist" -- Cover the input schema through the CLI too (`actor-driver.md`'s "Input schema, validation and defaults"): a `apify call` with no `--input` at all must run on the schema's defaults, with the same input-dependent `itemCount` assertion, and a `apify call` whose input the schema rejects must fail, naming the offending field, with no run started +- Cover the input schema through the CLI too (`actor-driver.md`): `apify call` with no `--input` must run on the schema's defaults, asserted from the run's own `INPUT`, and `apify call` with an input the schema rejects must fail, naming the offending field ## Browser view From 1472ddb7ad80457a0952d1ae0ff6d594fe8a4ad4 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 23 Sep 2026 09:43:26 +0000 Subject: [PATCH 06/14] feat: Let the sample Actors take their defaults from the input schema Every sample repeated its input schema's defaults as an in-code fallback, so a missing field was silently papered over and the default lived in two places. The runtime (like the platform) fills those fields in before the run starts, so the fallbacks are gone and the schema is the only place a default is written. The two Playwright samples' start URLs had only a prefill, so they now carry a default too. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_011U9yUTUn78bqGDn5obYSXs --- sample_actor_crawler/src/main.py | 6 ++++-- sample_actor_nonstandard/app/main.py | 4 +++- sample_actor_playwright/.actor/input_schema.json | 3 ++- sample_actor_playwright/src/main.ts | 8 +++++--- sample_actor_playwright_py/.actor/input_schema.json | 3 ++- sample_actor_playwright_py/my_actor/main.py | 7 ++++--- sample_actor_py/src/main.py | 6 ++++-- sample_actor_ts/src/main.ts | 10 ++++++---- 8 files changed, 30 insertions(+), 17 deletions(-) diff --git a/sample_actor_crawler/src/main.py b/sample_actor_crawler/src/main.py index 157b62e..ddafa59 100644 --- a/sample_actor_crawler/src/main.py +++ b/sample_actor_crawler/src/main.py @@ -24,11 +24,13 @@ async def request_handler(context: ParselCrawlingContext) -> None: async def main() -> None: async with Actor: + # Both fields have a `default` in the input schema, so the runtime fills them in before the + # run starts (the Apify platform does the same) - the Actor needs no fallback of its own. actor_input = await Actor.get_input() or {} - start_url = actor_input.get("startUrl", "https://crawlee.dev") + start_url = actor_input["startUrl"] proxy_configuration = await Actor.create_proxy_configuration( - actor_proxy_input=actor_input.get("proxyConfiguration") + actor_proxy_input=actor_input["proxyConfiguration"] ) crawler = ParselCrawler( diff --git a/sample_actor_nonstandard/app/main.py b/sample_actor_nonstandard/app/main.py index 40409ac..ef1f3f5 100644 --- a/sample_actor_nonstandard/app/main.py +++ b/sample_actor_nonstandard/app/main.py @@ -46,7 +46,9 @@ def main() -> None: raw_input = api_request('GET', f'key-value-stores/{key_value_store_id}/records/INPUT') actor_input = json.loads(raw_input) if raw_input else {} - item_count = int(actor_input.get('itemCount', 2)) + # `itemCount` has a `default` in the input schema, so the runtime fills it in before the run + # starts - no fallback needed here. + item_count = int(actor_input['itemCount']) print(f'main.py: pushing {item_count} item(s) to dataset {dataset_id}.') # Both accepted body shapes of `POST /v2/datasets/:id/items`: one object, then an array. diff --git a/sample_actor_playwright/.actor/input_schema.json b/sample_actor_playwright/.actor/input_schema.json index fe27fb2..e93f71d 100644 --- a/sample_actor_playwright/.actor/input_schema.json +++ b/sample_actor_playwright/.actor/input_schema.json @@ -9,7 +9,8 @@ "type": "array", "description": "URLs to start with.", "editor": "requestListSources", - "prefill": [{ "url": "https://crawlee.dev/" }] + "prefill": [{ "url": "https://crawlee.dev/" }], + "default": [{ "url": "https://crawlee.dev/" }] }, "maxRequestsPerCrawl": { "title": "Max requests per crawl", diff --git a/sample_actor_playwright/src/main.ts b/sample_actor_playwright/src/main.ts index 331c943..4da395f 100644 --- a/sample_actor_playwright/src/main.ts +++ b/sample_actor_playwright/src/main.ts @@ -24,9 +24,11 @@ interface Input { // Initialize the Apify SDK await Actor.init(); -// Structure of input is defined in .actor/input_schema.json -const { startUrls = [{ url: 'https://crawlee.dev/' }], maxRequestsPerCrawl = 3 } = - (await Actor.getInput()) ?? ({} as Input); +// Structure of input is defined in .actor/input_schema.json. Every field has a `default` there, so +// the runtime fills it in before the run starts - the Actor needs no fallback of its own. +const input = await Actor.getInput(); +if (!input) throw new Error('No input: the input schema should have supplied its defaults.'); +const { startUrls, maxRequestsPerCrawl } = input; // Without a proxy password (a plain local run) the crawler connects directly instead of failing the access check. const proxyConfiguration = process.env.APIFY_PROXY_PASSWORD diff --git a/sample_actor_playwright_py/.actor/input_schema.json b/sample_actor_playwright_py/.actor/input_schema.json index 0f3040c..958f396 100644 --- a/sample_actor_playwright_py/.actor/input_schema.json +++ b/sample_actor_playwright_py/.actor/input_schema.json @@ -9,7 +9,8 @@ "type": "array", "description": "URLs to start with.", "editor": "requestListSources", - "prefill": [{ "url": "https://crawlee.dev/" }] + "prefill": [{ "url": "https://crawlee.dev/" }], + "default": [{ "url": "https://crawlee.dev/" }] }, "max_requests_per_crawl": { "title": "Max requests per crawl", diff --git a/sample_actor_playwright_py/my_actor/main.py b/sample_actor_playwright_py/my_actor/main.py index c3868a8..b08a199 100644 --- a/sample_actor_playwright_py/my_actor/main.py +++ b/sample_actor_playwright_py/my_actor/main.py @@ -23,10 +23,11 @@ async def main() -> None: """ # Enter the context of the Actor. async with Actor: - # Retrieve the Actor input, and use default values if not provided. + # Every field has a `default` in the input schema, so the runtime fills it in before the run + # starts (the Apify platform does the same) - the Actor needs no fallback of its own. actor_input = await Actor.get_input() or {} - start_urls = [url.get('url') for url in actor_input.get('start_urls', [{'url': 'https://crawlee.dev/'}])] - max_requests_per_crawl = int(actor_input.get('max_requests_per_crawl', 3)) + start_urls = [url.get('url') for url in actor_input['start_urls']] + max_requests_per_crawl = int(actor_input['max_requests_per_crawl']) # Exit if no start URLs are provided. if not start_urls: diff --git a/sample_actor_py/src/main.py b/sample_actor_py/src/main.py index 6d3e214..eb3dcab 100644 --- a/sample_actor_py/src/main.py +++ b/sample_actor_py/src/main.py @@ -36,9 +36,11 @@ async def log_resource_usage(event_data: EventSystemInfoData) -> None: cap = f'${pricing.max_total_charge_usd}' if pricing.max_total_charge_usd.is_finite() else 'none' Actor.log.info(f'Pay-per-event pricing in effect, max total charge: {cap}.') + # Both fields have a `default` in the input schema, so the runtime fills them in before the + # run starts (the Apify platform does the same) - the Actor needs no fallback of its own. actor_input = await Actor.get_input() or {} - start_url = actor_input.get('startUrl', 'https://crawlee.dev/') - max_pages = int(actor_input.get('maxPages', 2)) + start_url = actor_input['startUrl'] + max_pages = int(actor_input['maxPages']) Actor.log.info(f'Crawling up to {max_pages} page(s) starting from {start_url}.') # Crawling through the Actor's default request queue exercises the runtime's diff --git a/sample_actor_ts/src/main.ts b/sample_actor_ts/src/main.ts index 4dd153b..cf31962 100644 --- a/sample_actor_ts/src/main.ts +++ b/sample_actor_ts/src/main.ts @@ -7,8 +7,8 @@ import { CheerioCrawler } from '@crawlee/cheerio'; await Actor.init(); interface Input { - startUrl?: string; - maxPages?: number; + startUrl: string; + maxPages: number; } const { memoryMbytes } = Actor.getEnv(); @@ -34,9 +34,11 @@ if (isPayPerEvent) { log.info(`Pay-per-event pricing in effect, max total charge: ${cap}.`); } +// Both fields have a `default` in the input schema, so the runtime fills them in before the run +// starts (the Apify platform does the same) - the Actor needs no fallback of its own. const input = await Actor.getInput(); -const startUrl = input?.startUrl ?? 'https://crawlee.dev/'; -const maxPages = input?.maxPages ?? 2; +if (!input) throw new Error('No input: the input schema should have supplied its defaults.'); +const { startUrl, maxPages } = input; log.info(`Crawling up to ${maxPages} page(s) starting from ${startUrl}.`); From eba700eca5d2012cbfc9232a117ac50ab9a83cc2 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 23 Sep 2026 11:04:30 +0000 Subject: [PATCH 07/14] feat: Log the start pre-charge and warn on platform-incompatible memory A run's log now says how many `apify-actor-start` events it was pre-charged and what they cost, and warns when the requested memory is one the Apify platform would refuse. The memory is still applied as asked. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_011U9yUTUn78bqGDn5obYSXs --- requirements/actor-driver.md | 2 ++ requirements/unsupported.md | 3 ++- skills/actor-runtime/SKILL.md | 6 ++++-- src/resources.ts | 27 +++++++++++++++++++++++++ src/services/charging.ts | 28 ++++++++++++++++++++++++-- src/services/runs.ts | 14 +++++++++++-- test/integration/pay-per-event.test.ts | 21 +++++++++++++++++++ test/unit/resources.test.ts | 26 ++++++++++++++++++++++++ 8 files changed, 120 insertions(+), 7 deletions(-) create mode 100644 test/unit/resources.test.ts diff --git a/requirements/actor-driver.md b/requirements/actor-driver.md index a69b52d..3a7a971 100644 --- a/requirements/actor-driver.md +++ b/requirements/actor-driver.md @@ -241,6 +241,8 @@ start`, ...) is refused by name, naming both the `CMD` fix and how to clear debu per subscription tier is charged at the lowest paid tier; nothing is ever billed or paid out; and the rules tying a price change to payout details, notice periods and subscription tiers do not apply. - An Actor's own charging code therefore runs here unchanged, with no local-testing switch. +- A run's log states what it was pre-charged for starting, so the count is visible where the platform + leaves it to be discovered on the bill. - The pricing can also be set from the console (`console.md`). # Users diff --git a/requirements/unsupported.md b/requirements/unsupported.md index 05aa04d..1029678 100644 --- a/requirements/unsupported.md +++ b/requirements/unsupported.md @@ -100,7 +100,8 @@ real account, not in the runtime. ## Platform limits not enforced -- Memory steps and bounds (128 MB - 32 GB, powers of two) +- Memory steps and bounds (128 MB - 32 GB, powers of two); a run asking for anything else is warned + about in its log and started with it anyway - Record, item and input size limits - Concurrent run, rate and per-account quotas - Process, file-descriptor and shared-memory limits diff --git a/skills/actor-runtime/SKILL.md b/skills/actor-runtime/SKILL.md index a11f80f..fd88600 100644 --- a/skills/actor-runtime/SKILL.md +++ b/skills/actor-runtime/SKILL.md @@ -152,8 +152,10 @@ From then on every run of the Actor is a pay-per-event run: the SDKs read `prici `ACTOR_TEST_PAY_PER_EVENT` - the runtime already looks like the platform to the SDK, which refuses that variable together with `APIFY_IS_AT_HOME`. The synthetic events work too: `apify-actor-start` is charged at run start (once per GB of memory) and `apify-default-dataset-item` once per item pushed to the run's -default dataset, when the pricing defines them. Tiered event prices resolve to the `BRONZE` (Starter) -tier. The array is append-only, as on the platform: send the entries the Actor already has, unchanged, +default dataset, when the pricing defines them. The start pre-charge scales with the run's memory and +its count is written to the run log, so `-m 8192` charges eight start events, not one; a memory the +platform would refuse (anything but a power of two between 128 MB and 32 GB) is warned about in the log +and then used anyway. Tiered event prices resolve to the `BRONZE` (Starter) tier. The array is append-only, as on the platform: send the entries the Actor already has, unchanged, plus at most one new one starting after all of them - so making the Actor free again means appending a `{"pricingModel":"FREE"}` entry, not sending `[]`. The same form is on the Actor's console page. diff --git a/src/resources.ts b/src/resources.ts index fcf7b17..1b8d8e6 100644 --- a/src/resources.ts +++ b/src/resources.ts @@ -17,3 +17,30 @@ export function cpuQuotaFor(memoryMbytes: number): number { const rawQuota = dedicatedCpusFor(memoryMbytes) * CPU_PERIOD_US; return Math.max(MIN_CPU_QUOTA_US, Math.round(rawQuota)); } + +/** The platform's accepted memory grants: powers of two, from 128 MB to 32 GB. */ +const MIN_PLATFORM_MEMORY_MBYTES = 128; +const MAX_PLATFORM_MEMORY_MBYTES = 32_768; + +/** + * A warning for a memory grant the Apify platform would refuse outright, or `undefined` for one it + * accepts. This runtime applies whatever it is given (`unsupported.md` - memory steps and bounds are + * deliberately not enforced), so the run still starts; what the warning buys is that a developer finds + * out here rather than on the first platform run. Worth saying for pay-per-event Actors especially: the + * pre-charged `apify-actor-start` count is one per *whole* gigabyte, so a grant between two steps pays + * for the lower one. + */ +export function platformIncompatibleMemoryWarning(memoryMbytes: number): string | undefined { + const defects: string[] = []; + if (memoryMbytes < MIN_PLATFORM_MEMORY_MBYTES || memoryMbytes > MAX_PLATFORM_MEMORY_MBYTES) { + defects.push(`outside the ${MIN_PLATFORM_MEMORY_MBYTES} MB - ${MAX_PLATFORM_MEMORY_MBYTES} MB range`); + } + if (!Number.isInteger(Math.log2(memoryMbytes))) defects.push('not a power of two'); + if (defects.length === 0) return undefined; + + return ( + `Warning: the requested memory of ${memoryMbytes} MB is ${defects.join(' and ')}, which the Apify ` + + `platform refuses (it accepts powers of two from ${MIN_PLATFORM_MEMORY_MBYTES} MB to ` + + `${MAX_PLATFORM_MEMORY_MBYTES} MB). Running with it anyway.` + ); +} diff --git a/src/services/charging.ts b/src/services/charging.ts index 93462c2..f349d1b 100644 --- a/src/services/charging.ts +++ b/src/services/charging.ts @@ -11,9 +11,14 @@ import type { RunRecord } from '../storage/entities.js'; import { getRegistries } from '../storage/registries.js'; import { isTerminalJobStatus } from './job-status.js'; import { appendRuntimeLog } from './logs.js'; -import { APIFY_EVENTS_PREFIX, DEFAULT_DATASET_ITEM_EVENT_NAME, isPayPerEvent } from './pricing.js'; +import { + ACTOR_START_EVENT_NAME, + APIFY_EVENTS_PREFIX, + DEFAULT_DATASET_ITEM_EVENT_NAME, + isPayPerEvent, +} from './pricing.js'; import { abortRun } from './runs.js'; -import { eventUsageFor, sumEventUsageUsd } from './run-usage.js'; +import { eventUsageFor, roundUsd, sumEventUsageUsd } from './run-usage.js'; /** The platform's own window. */ const IDEMPOTENCY_TTL_MS = 3 * 60 * 1000; @@ -148,6 +153,25 @@ export async function enforceCostLimit(driver: Driver, run: RunRecord): Promise< if (stamped.status === 'RUNNING') await abortRun(driver, stamped, true, message); } +/** + * What the run was pre-charged for starting, or `undefined` when nothing was. The platform charges this + * silently, and its count follows the run's memory rather than anything the Actor does - so a developer + * who raises the memory sees the bill rise with no line in the log to explain it. + */ +export function actorStartChargeMessage(run: RunRecord): string | undefined { + if (!isPayPerEvent(run.pricingInfo)) return undefined; + const count = run.chargedEventCounts?.[ACTOR_START_EVENT_NAME]; + if (!count) return undefined; + + const { eventPriceUsd } = run.pricingInfo.pricingPerEvent.actorChargeEvents[ACTOR_START_EVENT_NAME] ?? { + eventPriceUsd: 0, + }; + return ( + `Pre-charged ${count} '${ACTOR_START_EVENT_NAME}' event(s), $${roundUsd(count * eventPriceUsd)} in total, ` + + `for ${run.options.memoryMbytes} MB of memory - one event per whole gigabyte, minimum one.` + ); +} + export function resetChargingForTests(): void { idempotencyRecords.clear(); defaultDatasetRuns.clear(); diff --git a/src/services/runs.ts b/src/services/runs.ts index 0f0a06a..08468dd 100644 --- a/src/services/runs.ts +++ b/src/services/runs.ts @@ -18,12 +18,16 @@ import { type DebugPlan, } from './debug-mode.js'; import { browserViewLogLine, describeBrowserViewerStartFailure } from './browser-view.js'; -import { dedicatedCpusFor } from '../resources.js'; +import { dedicatedCpusFor, platformIncompatibleMemoryWarning } from '../resources.js'; import { CONTAINER_EVENTS_WS_BASE_URL } from '../config.js'; import { formatRuntimeLogLines, type RuntimeLogLine } from '../runtime-log.js'; import { getRunTelemetry } from './events-channel.js'; import { initialChargedEventCounts, resolveRunPricingInfo } from './pricing.js'; -import { registerDefaultDatasetForCharging, unregisterDefaultDatasetForCharging } from './charging.js'; +import { + actorStartChargeMessage, + registerDefaultDatasetForCharging, + unregisterDefaultDatasetForCharging, +} from './charging.js'; const DEFAULT_MEMORY_MBYTES = 1024; const DEFAULT_TIMEOUT_SECS = 300; @@ -227,6 +231,12 @@ export async function startRun( await runs.set(record.id, record); registerDefaultDatasetForCharging(record); + // Both lines are about what the caller asked for, so they are written before the run does anything. + const memoryWarning = platformIncompatibleMemoryWarning(memoryMbytes); + if (memoryWarning) appendRuntimeLog(record.id, memoryWarning); + const startCharge = actorStartChargeMessage(record); + if (startCharge) appendRuntimeLog(record.id, startCharge); + void runInBackground(driver, actor, record, options).catch(async (error: unknown) => { // Every *expected* failure mode inside `runInBackground` is already caught internally and mapped // to a terminal status - this is only reached by a genuinely unexpected exception (e.g. a diff --git a/test/integration/pay-per-event.test.ts b/test/integration/pay-per-event.test.ts index d75135f..c4913d9 100644 --- a/test/integration/pay-per-event.test.ts +++ b/test/integration/pay-per-event.test.ts @@ -219,6 +219,11 @@ describe('pay-per-event: runs and charging', () => { 0.01, ); + // A grant the platform accepts draws no warning, only the pre-charge line. + const log = await server.client.log(runId).get(); + expect(log).not.toContain('Warning: the requested memory'); + expect(log).toContain("Pre-charged 2 'apify-actor-start' event(s), $0.01 in total, for 2048 MB of memory"); + const env = driver.startCalls[0]!.ctx.env; expect(env.ACTOR_MAX_TOTAL_CHARGE_USD).toBe('1.5'); // The SDKs must fetch the run object for pricing, never read stale env copies (`services/runs.ts`). @@ -228,6 +233,22 @@ describe('pay-per-event: runs and charging', () => { await finishRun(server, driver, runId); }); + it("warns about memory the platform would refuse, and says what the run's start pre-charge covers", async () => { + const actorId = await seedRunnableActor(server, 'odd-memory-actor', PPE_PRICING); + // 8096 MB is between two steps: the platform refuses it outright, this runtime runs it and + // pre-charges for the 7 whole gigabytes it covers. + const runId = await startLiveRun(server, driver, actorId, { memory: 8096 }); + + const log = await server.client.log(runId).get(); + expect(log).toContain('8096 MB is not a power of two'); + expect(log).toContain("Pre-charged 7 'apify-actor-start' event(s), $0.035 in total, for 8096 MB of memory"); + + const run = (await server.client.run(runId).get()) as unknown as Record; + expect((run.chargedEventCounts as Record)['apify-actor-start']).toBe(7); + + await finishRun(server, driver, runId); + }); + it('a run of an unpriced Actor has no pricingInfo, chargedEventCounts, or maxTotalChargeUsd, and no cap env var', async () => { const actorId = await seedRunnableActor(server, 'free-actor'); const runId = await startLiveRun(server, driver, actorId); diff --git a/test/unit/resources.test.ts b/test/unit/resources.test.ts new file mode 100644 index 0000000..ea644d0 --- /dev/null +++ b/test/unit/resources.test.ts @@ -0,0 +1,26 @@ +import { describe, expect, it } from 'vitest'; + +import { platformIncompatibleMemoryWarning } from '../../src/resources.js'; + +describe('platformIncompatibleMemoryWarning', () => { + it('says nothing about a grant the platform accepts', () => { + for (const memoryMbytes of [128, 256, 512, 1024, 2048, 4096, 8192, 32_768]) { + expect(platformIncompatibleMemoryWarning(memoryMbytes)).toBeUndefined(); + } + }); + + it('names the defect of a grant the platform refuses, without refusing it here', () => { + const betweenSteps = platformIncompatibleMemoryWarning(8096); + expect(betweenSteps).toContain('8096 MB is not a power of two'); + expect(betweenSteps).toContain('Running with it anyway.'); + + expect(platformIncompatibleMemoryWarning(64)).toContain('outside the 128 MB - 32768 MB range'); + expect(platformIncompatibleMemoryWarning(65_536)).toContain('outside the 128 MB - 32768 MB range'); + }); + + it('names both defects of a grant that is out of range and off the steps', () => { + expect(platformIncompatibleMemoryWarning(100)).toContain( + 'outside the 128 MB - 32768 MB range and not a power of two', + ); + }); +}); From 509cdf2b94553114d0c294848554bd41f131c408 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 23 Sep 2026 11:15:10 +0000 Subject: [PATCH 08/14] feat: Price the per-dataset-item synthetic event in the sample Actors Both simple samples now ship pricing for `apify-default-dataset-item` alongside `apify-actor-start`, so one run of a priced sample shows every way an event can be charged - two by the Actor, two by the runtime. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_011U9yUTUn78bqGDn5obYSXs --- README.md | 4 +++- sample_actor_py/pricing.json | 5 +++++ sample_actor_py/src/main.py | 4 +++- sample_actor_ts/pricing.json | 5 +++++ sample_actor_ts/src/main.ts | 4 +++- skills/actor-runtime/SKILL.md | 5 +++-- test/unit/pricing.test.ts | 32 ++++++++++++++++++++++++++++++++ 7 files changed, 54 insertions(+), 5 deletions(-) diff --git a/README.md b/README.md index 0c1b9a8..12e23b9 100644 --- a/README.md +++ b/README.md @@ -235,7 +235,9 @@ do. Like Python debug mode, this needs the runtime to run from its own built ima Both bundled samples charge two events when their Actor is priced - `page-scraped` once per page and `crawl-finished` once at the end - and ship the pricing that defines them in `pricing.json`, so a run -charges for real right after a push: +charges for real right after a push. That pricing also declares both synthetic events, which the Actor +never charges itself: `apify-actor-start` at run start (once per whole GB of the run's memory) and +`apify-default-dataset-item` per item pushed to the default dataset. ```bash cd sample_actor_ts # or sample_actor_py diff --git a/sample_actor_py/pricing.json b/sample_actor_py/pricing.json index 996121b..3d1b3db 100644 --- a/sample_actor_py/pricing.json +++ b/sample_actor_py/pricing.json @@ -19,6 +19,11 @@ "eventDescription": "Charged by the runtime at run start, once per GB of memory.", "eventPriceUsd": 0.005, "isOneTimeEvent": true + }, + "apify-default-dataset-item": { + "eventTitle": "Dataset item", + "eventDescription": "Charged by the runtime for every item pushed to the default dataset.", + "eventPriceUsd": 0.001 } } } diff --git a/sample_actor_py/src/main.py b/sample_actor_py/src/main.py index eb3dcab..6fe907f 100644 --- a/sample_actor_py/src/main.py +++ b/sample_actor_py/src/main.py @@ -30,7 +30,9 @@ async def log_resource_usage(event_data: EventSystemInfoData) -> None: # Under pay-per-event pricing (set on the Actor through the runtime's API or console) this Actor # charges two events: 'page-scraped' once per page and 'crawl-finished' once at the end. A free - # Actor skips both, so a plain push-and-call stays unchanged. + # Actor skips both, so a plain push-and-call stays unchanged. `pricing.json` prices two more, + # 'apify-actor-start' and 'apify-default-dataset-item': those are charged by the runtime itself, + # which is why no code here charges them. pricing = Actor.get_charging_manager().get_pricing_info() if pricing.is_pay_per_event: cap = f'${pricing.max_total_charge_usd}' if pricing.max_total_charge_usd.is_finite() else 'none' diff --git a/sample_actor_ts/pricing.json b/sample_actor_ts/pricing.json index 996121b..3d1b3db 100644 --- a/sample_actor_ts/pricing.json +++ b/sample_actor_ts/pricing.json @@ -19,6 +19,11 @@ "eventDescription": "Charged by the runtime at run start, once per GB of memory.", "eventPriceUsd": 0.005, "isOneTimeEvent": true + }, + "apify-default-dataset-item": { + "eventTitle": "Dataset item", + "eventDescription": "Charged by the runtime for every item pushed to the default dataset.", + "eventPriceUsd": 0.001 } } } diff --git a/sample_actor_ts/src/main.ts b/sample_actor_ts/src/main.ts index cf31962..26d7702 100644 --- a/sample_actor_ts/src/main.ts +++ b/sample_actor_ts/src/main.ts @@ -27,7 +27,9 @@ Actor.on('systemInfo', (info: { cpuCurrentUsage?: number; memCurrentBytes?: numb // Under pay-per-event pricing (set on the Actor through the runtime's API or console) this Actor charges // two events: 'page-scraped' once per page and 'crawl-finished' once at the end. A free Actor skips both, -// so a plain push-and-call stays unchanged. +// so a plain push-and-call stays unchanged. `pricing.json` prices two more, 'apify-actor-start' and +// 'apify-default-dataset-item': those are charged by the runtime itself, which is why no code here +// charges them. const { isPayPerEvent, maxTotalChargeUsd } = Actor.getChargingManager().getPricingInfo(); if (isPayPerEvent) { const cap = Number.isFinite(maxTotalChargeUsd) ? `$${maxTotalChargeUsd}` : 'none'; diff --git a/skills/actor-runtime/SKILL.md b/skills/actor-runtime/SKILL.md index fd88600..ff20c79 100644 --- a/skills/actor-runtime/SKILL.md +++ b/skills/actor-runtime/SKILL.md @@ -143,8 +143,9 @@ apify api PUT /v2/actors/ --body '{"pricingInfos":[{"pricingModel":"PAY ``` Both bundled samples already charge `page-scraped` per page and `crawl-finished` once at the end, and -carry the matching pricing in `pricing.json`: `apify api PUT /v2/actors/ --body "$(cat -sample_actor_ts/pricing.json)"` prices one in a single call. +carry the matching pricing in `pricing.json` - including both synthetic events, so one run shows all +four being charged: `apify api PUT /v2/actors/ --body "$(cat sample_actor_ts/pricing.json)"` +prices one in a single call. From then on every run of the Actor is a pay-per-event run: the SDKs read `pricingInfo` and `chargedEventCounts` off the run object exactly as on the platform, and `Actor.charge()` (or diff --git a/test/unit/pricing.test.ts b/test/unit/pricing.test.ts index e88b14a..c421b32 100644 --- a/test/unit/pricing.test.ts +++ b/test/unit/pricing.test.ts @@ -3,10 +3,14 @@ * effect at a date, per-run resolution (tiered prices collapsed to the BRONZE tier), and the initial * `chargedEventCounts` with the synthetic start event (`actor-driver.md`'s "Pay-per-event pricing"). */ +import { readFileSync } from 'node:fs'; +import { dirname, join } from 'node:path'; +import { fileURLToPath } from 'node:url'; import { describe, expect, it } from 'vitest'; import { ACTOR_START_EVENT_NAME, + DEFAULT_DATASET_ITEM_EVENT_NAME, effectivePricingInfo, initialChargedEventCounts, isPayPerEvent, @@ -15,6 +19,8 @@ import { validatePricingInfosUpdate, } from '../../src/services/pricing.js'; +const REPO_ROOT = join(dirname(fileURLToPath(import.meta.url)), '..', '..'); + const NOW = new Date('2026-09-21T10:00:00.000Z'); const ppe = (events: Record, extra: Record = {}) => ({ @@ -286,3 +292,29 @@ describe('validatePricingInfosUpdate (the append-only history)', () => { expect(update([ppe({ result: { eventTitle: 'Result', eventPriceUsd: 0.01 } })], []).kind).toBe('ok'); }); }); + +describe('the pricing the bundled samples ship', () => { + // The README tells a developer to PUT these files at the Actor verbatim, so what the runtime would + // answer to that is worth knowing here rather than from a 400 at the terminal. + it.each(['sample_actor_ts', 'sample_actor_py'])( + '%s/pricing.json is accepted and prices both synthetic events', + (sample) => { + const { pricingInfos } = JSON.parse(readFileSync(join(REPO_ROOT, sample, 'pricing.json'), 'utf8')) as { + pricingInfos: unknown; + }; + + const resolved = resolveRunPricingInfo(ok(pricingInfos), NOW); + if (!isPayPerEvent(resolved)) throw new Error('expected the sample to be priced per event'); + expect(Object.keys(resolved.pricingPerEvent.actorChargeEvents).sort()).toEqual([ + ACTOR_START_EVENT_NAME, + DEFAULT_DATASET_ITEM_EVENT_NAME, + 'crawl-finished', + 'page-scraped', + ]); + + // Neither synthetic event is ever charged by the sample's own code: the start event is + // pre-charged per whole GB, the item event by the dataset route on every push. + expect(initialChargedEventCounts(resolved, 2048)?.[ACTOR_START_EVENT_NAME]).toBe(2); + }, + ); +}); From 0929127056d0b927b8100982fe76f955c83935d3 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 23 Sep 2026 11:22:46 +0000 Subject: [PATCH 09/14] fix: Persist a run's sampled telemetry before its status turns terminal The CPU and memory figures are accumulated in memory, and were written to the run record only after the terminal status. A client that waited for the run to finish and then read the record could find them missing. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_011U9yUTUn78bqGDn5obYSXs --- src/services/runs.ts | 9 +++++++-- test/integration/pay-per-event.test.ts | 12 ++++++++++++ 2 files changed, 19 insertions(+), 2 deletions(-) diff --git a/src/services/runs.ts b/src/services/runs.ts index 08468dd..7528bcc 100644 --- a/src/services/runs.ts +++ b/src/services/runs.ts @@ -448,6 +448,9 @@ export async function runInBackground( // observes it turn terminal, and immediately does a non-stream `GET /v2/logs/:id` can never observe // the persisted log lagging behind the status it just saw. await flushLog(record.id); + // Before the status write for the same reason as the flush: the sampled figures live only in + // memory until this runs, and a client that sees the run turn terminal reads the record next. + await persistRunTelemetry(record.id); // Guarded: `container.wait()` resolving is not proof the run wasn't aborted - `container.stop()` // (from an in-flight `abortRun`) and the container exiting on its own race off the same // underlying Docker event with no ordering guarantee. If `abortRun` already moved the record to @@ -469,6 +472,7 @@ export async function runInBackground( // unusable mount) is what `apify call` streams, and the status message alone leaves it empty. appendRuntimeLog(record.id, `Cannot start run: ${statusMessage}`); await flushLog(record.id); + await persistRunTelemetry(record.id); await transitionJobStatus(runs, record.id, 'FAILED', { finishedAt: new Date().toISOString(), statusMessage, @@ -477,11 +481,12 @@ export async function runInBackground( // The container is gone, so an open graceful-abort window has nothing left to wait out - the path an // Actor that honours the `aborting` frame takes. The log is already flushed by both the success // path above and the catch below, so a client seeing this terminal status can still read all of it. + // The accumulators are in-memory, so a finished run keeps its figures only if they are written + // here - before the transition below, and again for the paths above that end the run elsewhere. + await persistRunTelemetry(record.id); if (cancelGracefulAbort(record.id)) { await transitionJobStatus(runs, record.id, 'ABORTED', { finishedAt: new Date().toISOString() }); } - // The accumulators are in-memory, so a finished run keeps its figures only if they are written here. - await persistRunTelemetry(record.id); unregisterDefaultDatasetForCharging(record); if (browserViewer) await driver.stopBrowserViewer(record.id); // A run that ends for real must not leave an armed migration-stop timer behind. diff --git a/test/integration/pay-per-event.test.ts b/test/integration/pay-per-event.test.ts index c4913d9..e9244a6 100644 --- a/test/integration/pay-per-event.test.ts +++ b/test/integration/pay-per-event.test.ts @@ -21,6 +21,7 @@ import { getRegistries } from '../../src/storage/registries.js'; import { generateId } from '../../src/storage/ids.js'; import { recordTaggedBuild, updateActor } from '../../src/services/actors.js'; import { subscribeEvents } from '../../src/services/events-channel.js'; +import { isTerminalJobStatus } from '../../src/services/job-status.js'; import type { ActorRecord, BuildRecord } from '../../src/storage/entities.js'; const PAGE_EVENT = 'page-scraped'; @@ -544,7 +545,18 @@ describe('run usage estimate', () => { expect(live.stats.computeUnits).toBeGreaterThan(0); expect(live.stats.computeUnits).toBeGreaterThanOrEqual(before.stats.computeUnits!); + // The sampled figures are written before the status turns terminal, the same guarantee the log + // flush has: whoever sees the run finish reads the record next, and must not find them missing. + const firstTerminalRead = (async () => { + for (;;) { + const current = await getRegistries().runs.get(runId); + if (current && isTerminalJobStatus(current.status)) return current; + await new Promise((resolve) => setImmediate(resolve)); + } + })(); await finishRun(server, driver, runId); + expect((await firstTerminalRead).stats?.memAvgBytes).toBe(200); + const finished = (await server.client.run(runId).get()) as unknown as { stats: Record }; expect(finished.stats.memAvgBytes).toBe(200); expect(finished.stats.cpuMaxUsage).toBe(30); From 8e4d3f8331460583b6660a4116d360ccb7794f47 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 23 Sep 2026 11:49:33 +0000 Subject: [PATCH 10/14] docs: Say how to price an Actor that already has a pricing The bundled `pricing.json` prices an Actor that has none yet; applying it to one that is already priced is refused, since the array is append-only. Both the README and the skill now give the append command. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_011U9yUTUn78bqGDn5obYSXs --- README.md | 13 +++++++++++++ skills/actor-runtime/SKILL.md | 5 ++++- 2 files changed, 17 insertions(+), 1 deletion(-) diff --git a/README.md b/README.md index 12e23b9..61ad86c 100644 --- a/README.md +++ b/README.md @@ -255,6 +255,19 @@ run's console page and the runs list show the same figures. platform: an update sends the Actor's existing entries unchanged plus at most one new one, starting after all of them. Making the Actor free again is therefore appending a `{"pricingModel": "FREE"}` entry. +The `PUT` above therefore prices an Actor that has no pricing yet. Once it has one, a second `PUT` of the +same file is refused (`pricingInfos[0] differs from the Actor's existing pricing info`) - the stored +entries carry the timestamps they were given, which the file does not. Append the file's entry to what +the Actor already has instead: + +```bash +apify api PUT /v2/actors/ --body "$(apify api GET /v2/actors/ | + jq --argjson new "$(jq '.pricingInfos[-1]' pricing.json)" '{pricingInfos: (.data.pricingInfos + [$new])}')" +``` + +The Actor's console page has the same thing as a form: the box holds the stored array, and adding an +entry below the existing ones does it without the shell. + Cap a run's spend the way a user does - the cap is a query parameter, with no `apify call` flag for it: ```bash diff --git a/skills/actor-runtime/SKILL.md b/skills/actor-runtime/SKILL.md index ff20c79..4ba84e5 100644 --- a/skills/actor-runtime/SKILL.md +++ b/skills/actor-runtime/SKILL.md @@ -158,7 +158,10 @@ its count is written to the run log, so `-m 8192` charges eight start events, no platform would refuse (anything but a power of two between 128 MB and 32 GB) is warned about in the log and then used anyway. Tiered event prices resolve to the `BRONZE` (Starter) tier. The array is append-only, as on the platform: send the entries the Actor already has, unchanged, plus at most one new one starting after all of them - so making the Actor free again means appending a -`{"pricingModel":"FREE"}` entry, not sending `[]`. The same form is on the Actor's console page. +`{"pricingModel":"FREE"}` entry, not sending `[]`. Re-sending a file that was already applied is refused +for the same reason (`pricingInfos[0] differs ...`): the stored entries carry timestamps the file does +not, so append its entry to what `GET /v2/actors/` returns rather than sending the file again. +The same form is on the Actor's console page. Cap a run's spend like a user would: `apify api POST '/v2/actors//runs?maxTotalChargeUsd=0.5'` (there is no `apify call` flag for it). When the charges reach the cap the run is aborted gracefully, From e5dd8746b42b5201eac26d82ccbd3938717b78ed Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 23 Sep 2026 11:54:04 +0000 Subject: [PATCH 11/14] feat: Name the field that breaks an append-only pricing update A rejected `pricingInfos` update now says which field of which entry differs from the stored one, and what was sent and stored there, instead of leaving the caller to diff two arrays by eye. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_011U9yUTUn78bqGDn5obYSXs --- src/services/pricing.ts | 27 +++++++++++++++++++++++++-- test/unit/pricing.test.ts | 19 +++++++++++++++++++ 2 files changed, 44 insertions(+), 2 deletions(-) diff --git a/src/services/pricing.ts b/src/services/pricing.ts index c032aeb..1339930 100644 --- a/src/services/pricing.ts +++ b/src/services/pricing.ts @@ -262,8 +262,8 @@ export function validatePricingInfosUpdate( for (const [index, entry] of current.entries()) { if (!isDeepStrictEqual(submitted[index], entry)) { return invalid( - `pricingInfos[${index}] differs from the Actor's existing pricing info - an update must start ` + - 'with the existing entries unchanged and may only append one more.', + `pricingInfos[${index}] differs from the Actor's existing pricing info ${describeDifference(submitted[index], entry)} - ` + + 'an update must start with the existing entries unchanged and may only append one more.', 'incorrect-pricing-modifier-prefix', ); } @@ -287,6 +287,29 @@ export function validatePricingInfosUpdate( return validated; } +/** + * The first field the submitted entry gets wrong, named as a path. Worth the code: the usual cause is an + * entry rebuilt from a file rather than from what the Actor already has, and the difference is then a + * timestamp nobody thinks to look at. + */ +function describeDifference(submitted: unknown, stored: unknown, path = ''): string { + if (isPlainObject(submitted) && isPlainObject(stored)) { + for (const key of new Set([...Object.keys(stored), ...Object.keys(submitted)])) { + const where = path === '' ? key : `${path}.${key}`; + if (!isDeepStrictEqual(submitted[key], stored[key])) + return describeDifference(submitted[key], stored[key], where); + } + } + const at = path === '' ? '' : `at "${path}" `; + return `${at}(sent ${describeValue(submitted)}, stored ${describeValue(stored)})`; +} + +function describeValue(value: unknown): string { + if (value === undefined) return 'nothing'; + const json = JSON.stringify(value) ?? String(value); + return json.length > 60 ? `${json.slice(0, 57)}...` : json; +} + /** The entry with the latest `startedAt` not after `date`, the platform's own rule. */ export function effectivePricingInfo( pricingInfos: readonly ActorPricingInfoRecord[] | undefined, diff --git a/test/unit/pricing.test.ts b/test/unit/pricing.test.ts index c421b32..03409e4 100644 --- a/test/unit/pricing.test.ts +++ b/test/unit/pricing.test.ts @@ -267,6 +267,25 @@ describe('validatePricingInfosUpdate (the append-only history)', () => { ); }); + it('names the field that differs, so the usual "rebuilt from a file" mistake is visible', () => { + expect(rejection([{ ...existing[0]!, startedAt: '2026-01-02T00:00:00.000Z' }]).message).toContain( + 'at "startedAt" (sent "2026-01-02T00:00:00.000Z", stored "2026-01-01T00:00:00.000Z")', + ); + // An entry sent without the timestamps the stored one carries: they are stamped with the time of + // the request, so `createdAt` is the first thing that cannot match. + expect(rejection([ppe({ result: { eventTitle: 'Result', eventPriceUsd: 0.01 } })]).message).toContain( + 'at "createdAt"', + ); + expect( + rejection([ + { + ...existing[0]!, + pricingPerEvent: { actorChargeEvents: { result: { eventTitle: 'Result', eventPriceUsd: 0.02 } } }, + }, + ]).message, + ).toContain('at "pricingPerEvent.actorChargeEvents.result.eventPriceUsd" (sent 0.02, stored 0.01)'); + }); + it('refuses a new entry that starts at or before one already stored', () => { expect(rejection([...existing, { pricingModel: 'FREE', startedAt: '2025-12-01T00:00:00.000Z' }]).type).toBe( 'cannot-add-pricing-info-that-alters-past', From 457076cd44b6fbc7ce76f70224e6370f70a697b3 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 23 Sep 2026 12:03:25 +0000 Subject: [PATCH 12/14] test: Compare rounded USD figures exactly instead of within a tolerance `usageUsd.ACTOR_COMPUTE_UNITS` is rounded to six decimals, so a product whose seventh decimal is a 5 moves by the full tolerance of `toBeCloseTo(..., 6)` and failed on the boundary. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_011U9yUTUn78bqGDn5obYSXs --- test/e2e/pay-per-event.test.ts | 6 +++++- test/integration/pay-per-event.test.ts | 3 ++- 2 files changed, 7 insertions(+), 2 deletions(-) diff --git a/test/e2e/pay-per-event.test.ts b/test/e2e/pay-per-event.test.ts index 6f02f39..5bdc3fa 100644 --- a/test/e2e/pay-per-event.test.ts +++ b/test/e2e/pay-per-event.test.ts @@ -131,7 +131,11 @@ describe('pay-per-event pricing and the run cost estimate via apify-cli (require expect(freeRun.pricingInfo).toBeUndefined(); expect(freeRun.chargedEventCounts).toBeUndefined(); expect(freeRun.stats.computeUnits).toBeGreaterThan(0); - expect(freeRun.usageUsd.ACTOR_COMPUTE_UNITS).toBeCloseTo(freeRun.usage.ACTOR_COMPUTE_UNITS * 0.2, 6); + // Exactly the runtime's own rounding, not a tolerance: a product whose 7th decimal is a 5 is + // moved by the full tolerance of `toBeCloseTo(..., 6)`, which then fails on the boundary. + expect(freeRun.usageUsd.ACTOR_COMPUTE_UNITS).toBe( + Number((freeRun.usage.ACTOR_COMPUTE_UNITS * 0.2).toFixed(6)), + ); expect(freeRun.usageTotalUsd).toBe(freeRun.usageUsd.ACTOR_COMPUTE_UNITS); // The container was sampled while it ran. expect(freeRun.stats.memAvgBytes).toBeGreaterThan(0); diff --git a/test/integration/pay-per-event.test.ts b/test/integration/pay-per-event.test.ts index e9244a6..c0e6257 100644 --- a/test/integration/pay-per-event.test.ts +++ b/test/integration/pay-per-event.test.ts @@ -511,7 +511,8 @@ describe('run usage estimate', () => { const usage = run.usage as Record; const usageUsd = run.usageUsd as Record; expect(usage.ACTOR_COMPUTE_UNITS).toBe(stats.computeUnits); - expect(usageUsd.ACTOR_COMPUTE_UNITS).toBeCloseTo(stats.computeUnits! * 0.2, 6); + // The runtime's own rounding rather than a tolerance - see the same assertion in the e2e test. + expect(usageUsd.ACTOR_COMPUTE_UNITS).toBe(Number((stats.computeUnits! * 0.2).toFixed(6))); expect(run.usageTotalUsd).toBe(usageUsd.ACTOR_COMPUTE_UNITS); expect(run).not.toHaveProperty('eventUsage'); From a25cf0b20793a4d3d566484ff5f91285db5667ce Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 23 Sep 2026 12:14:37 +0000 Subject: [PATCH 13/14] refactor: Drop the rules around the runtime's dev-folder block The block's own lines already carry the runtime marker, so the two `=` rules added width without adding information. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_011U9yUTUn78bqGDn5obYSXs --- src/services/runs.ts | 17 ++--------------- test/e2e/dev-folder-bind-mount.test.ts | 5 ++--- test/integration/dev-folder.test.ts | 7 +++---- test/unit/runtime-log.test.ts | 7 +++++-- 4 files changed, 12 insertions(+), 24 deletions(-) diff --git a/src/services/runs.ts b/src/services/runs.ts index 7528bcc..4236252 100644 --- a/src/services/runs.ts +++ b/src/services/runs.ts @@ -20,7 +20,7 @@ import { import { browserViewLogLine, describeBrowserViewerStartFailure } from './browser-view.js'; import { dedicatedCpusFor, platformIncompatibleMemoryWarning } from '../resources.js'; import { CONTAINER_EVENTS_WS_BASE_URL } from '../config.js'; -import { formatRuntimeLogLines, type RuntimeLogLine } from '../runtime-log.js'; +import { formatRuntimeLogLines } from '../runtime-log.js'; import { getRunTelemetry } from './events-channel.js'; import { initialChargedEventCounts, resolveRunPricingInfo } from './pricing.js'; import { @@ -367,7 +367,7 @@ export async function runInBackground( appendRuntimeLog(record.id, unknownWorkingDirectoryLine(actor.localDevFolder)); } const runtimeSection = devMount ? liveDevFolderWarningLines(devMount) : []; - if (runtimeSection.length > 0) appendLog(record.id, renderRuntimeLogSection(runtimeSection)); + if (runtimeSection.length > 0) appendLog(record.id, formatRuntimeLogLines(runtimeSection)); // The sidecar comes up before the Actor's container. Started before the pre-start abort re-check below, // so an abort landing during this (possibly slow) step is still caught by it. @@ -641,16 +641,3 @@ export async function reconcileOrphanedJobs(driver: Driver): Promise { ), ); } - -/** 80 columns, not a terminal's full width: every line already carries a timestamp and the runtime - * marker, so a wider rule only forces wrapping. */ -function renderRuntimeLogSection(lines: readonly RuntimeLogLine[]): string { - const title = ' Local Actor runtime '; - const width = 80; - const head = `${'='.repeat(4)}${title}${'='.repeat(width - 4 - title.length)}`; - return formatRuntimeLogLines([ - { text: head, emphasis: true }, - ...lines, - { text: '='.repeat(width), emphasis: true }, - ]); -} diff --git a/test/e2e/dev-folder-bind-mount.test.ts b/test/e2e/dev-folder-bind-mount.test.ts index 80f79a6..cd9f38a 100644 --- a/test/e2e/dev-folder-bind-mount.test.ts +++ b/test/e2e/dev-folder-bind-mount.test.ts @@ -224,7 +224,7 @@ describe('local dev-folder bind mount: edit-compile-call loop with no rebuild (r expect(optedOutLog).toContain(`Skipping the registered local dev folder ${actorDir}`); expect(optedOutLog).toContain(ORIGINAL_MARKER); expect(optedOutLog).not.toContain(EDITED_MARKER); - expect(optedOutLog).not.toContain('Local Actor runtime'); + expect(optedOutLog).not.toContain('Live dev folder'); // The registration survived. const call = JSON.parse( @@ -233,7 +233,6 @@ describe('local dev-folder bind mount: edit-compile-call loop with no rebuild (r expect(call.run.status).toBe('SUCCEEDED'); const callLog = apifyAllOutput(['runs', 'log', call.run.id], { cwd: REPO_ROOT, env }); expect(callLog).toContain(EDITED_MARKER); - expect(callLog).toContain('Local Actor runtime'); expect(callLog).toContain(`Live dev folder: ${actorDir}`); expect(callLog).toContain('Live dev folder mode'); expect(callLog).toContain('apify call --no-dev-folder'); @@ -342,7 +341,7 @@ describe('local dev-folder bind mount: edit-compile-call loop with no rebuild (r expect(call.run.status).toBe('SUCCEEDED'); const log = apifyAllOutput(['runs', 'log', call.run.id], { cwd: REPO_ROOT, env }); - expect(log).not.toContain('Local Actor runtime'); + expect(log).not.toContain('Live dev folder'); }, 5 * 60 * 1000, ); diff --git a/test/integration/dev-folder.test.ts b/test/integration/dev-folder.test.ts index 658cff9..e128a30 100644 --- a/test/integration/dev-folder.test.ts +++ b/test/integration/dev-folder.test.ts @@ -834,11 +834,10 @@ describe('run-start devMount derivation (actor fields -> RunContext.devMount, se imageWorkingDirectory: '/usr/src/app', }); const log = await server.client.log(run.id).get(); - expect(log).toContain('Local Actor runtime'); expect(log).toContain('Live dev folder: /abs/dev/src'); expect(log).toContain('Live dev folder mode'); expect(log).toContain('apify call --no-dev-folder'); - expect(log!.indexOf('Local Actor runtime')).toBeLessThan(log!.indexOf('done')); + expect(log!.indexOf('Live dev folder:')).toBeLessThan(log!.indexOf('done')); }); it("marks every runtime-authored line of a run's log with the runtime prefix and its blue, and leaves the Actor's own output untouched", async () => { @@ -877,7 +876,7 @@ describe('run-start devMount derivation (actor fields -> RunContext.devMount, se const run = await server.client.actor(actor.id).start({}, { waitForFinish: 5 }); expect(run.status).toBe('SUCCEEDED'); expect(capturing.getCapturedDevMount()).toBeUndefined(); - expect(await server.client.log(run.id).get()).not.toContain('Local Actor runtime'); + expect(await server.client.log(run.id).get()).not.toContain('Live dev folder'); }); it('an Actor whose registration was set and then cleared also gets devMount: undefined, not the stale pair', async () => { @@ -988,7 +987,7 @@ describe('per-run opt-out: POST /v2/actors/:actorId/runs?devFolder=false (servic expect(capturing.getCapturedDevMount()).toBeUndefined(); const log = await server.client.log(run.id).get(); expect(log).toContain('Skipping the registered local dev folder /abs/dev/src for this run'); - expect(log).not.toContain('Local Actor runtime'); + expect(log).not.toContain('Live dev folder'); }); it('devFolder=false leaves the registration itself untouched - the next run without the opt-out mounts again', async () => { diff --git a/test/unit/runtime-log.test.ts b/test/unit/runtime-log.test.ts index 4775172..90c2171 100644 --- a/test/unit/runtime-log.test.ts +++ b/test/unit/runtime-log.test.ts @@ -76,10 +76,13 @@ describe('formatRuntimeLogLines', () => { it('renders a block whose emphasis differs line by line as one chunk', () => { expect( formatRuntimeLogLines([ - { text: '==== Local Actor runtime ====', emphasis: true }, + { text: '!! Running in `Live dev folder mode`', emphasis: true }, { text: 'Live dev folder: /src' }, ]), - ).toBe(`${BOLD_MARKER} ${BOLD}==== Local Actor runtime ====${RESET}\n` + `${MARKER} Live dev folder: /src\n`); + ).toBe( + `${BOLD_MARKER} ${BOLD}!! Running in \`Live dev folder mode\`${RESET}\n` + + `${MARKER} Live dev folder: /src\n`, + ); }); }); From 630e13c99e7192ff3cb8e425012df8e36192a71e Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 23 Sep 2026 12:15:48 +0000 Subject: [PATCH 14/14] docs: State the input-schema requirements as the platform's, plus differences Seven bullets restating platform behaviour become one that names it, one listing what this runtime does differently, and the local dev-folder caveat. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_011U9yUTUn78bqGDn5obYSXs --- requirements/actor-driver.md | 32 ++++++++++---------------------- 1 file changed, 10 insertions(+), 22 deletions(-) diff --git a/requirements/actor-driver.md b/requirements/actor-driver.md index 3a7a971..a688894 100644 --- a/requirements/actor-driver.md +++ b/requirements/actor-driver.md @@ -36,29 +36,17 @@ # Input schema, validation and defaults -- **An Actor's input schema is read when the Actor is built**, and every run of that build has its - input validated against it with the schema's defaults applied. A build with no input schema takes +- **Input schemas work as on the Apify platform**: the schema is read from the pushed source when the + Actor is built - the `input` field of `.actor/actor.json`, else `.actor/INPUT_SCHEMA.json`, else + `INPUT_SCHEMA.json` - a build whose schema cannot be read or is not a valid input schema fails with + the reason in its log, and every run of a build is validated against that build's schema with its + defaults applied, a rejected input starting nothing (`api.md`). A build with no input schema takes every input exactly as the caller sent it. -- The schema is the `input` field of `.actor/actor.json` - an inline schema, or a path relative to - `.actor/` - else `.actor/INPUT_SCHEMA.json`, else `INPUT_SCHEMA.json` at the Actor root, whichever - comes first. Matching is case-insensitive, exact-case wins ties, and every outcome is stated in the - build log. -- **A build fails**, with the reason in its log and status message, when the `input` field is neither - a path nor a schema or points outside the Actor root, when the schema cannot be parsed, or when it - is not a valid Apify input schema. A path naming no pushed file only warns; the default locations - are tried instead. -- **A run whose input the schema rejects never starts**: no run is created, and the call is answered - `400` naming the offending field (`api.md`). -- **Every field the input leaves out is filled from the schema**, nested objects and array items - included, and the run's `INPUT` holds the result - so a run started with no input at all runs on - the schema's defaults. A required field that has a default is satisfied by it; a required array - must hold at least one item. -- A run is validated against the schema of the build it runs, never another tag's. An Actor running - from a registered dev folder therefore uses its last build's schema: unlike a source edit, an - edited input schema takes effect only after `apify push`. -- **Known differences from the platform**: Apify Proxy group availability is not checked, so any - `apifyProxyGroups` selection is accepted; a `proxy` field's shape, its custom proxy URLs and its - country code still are. Encrypted secret input fields stay unsupported (`unsupported.md`). +- **Differences**: Apify Proxy group availability is not checked, so any `apifyProxyGroups` selection + is accepted, while the rest of a `proxy` field is still validated; encrypted secret input fields + stay unsupported (`unsupported.md`). +- An Actor running from a registered dev folder uses its last build's schema: unlike a source edit, + an edited input schema takes effect only after `apify push`. # Bind mount volumes with Actor source code