diff --git a/README.md b/README.md index e13e14b..61ad86c 100644 --- a/README.md +++ b/README.md @@ -55,6 +55,23 @@ In a build or run log, everything the runtime itself has to say - dev-folder not attach line, the browser-view URL, migration markers, a run that could not be started - opens with a blue `[actor-runtime]` prefix. Your Actor's own output is passed through byte for byte. +## Input schema: defaults and validation + +An Actor that declares an input schema (the `input` field of `.actor/actor.json`, `.actor/INPUT_SCHEMA.json`, +or `INPUT_SCHEMA.json` at its root) gets the platform's behaviour locally: the schema's defaults are filled into +every run's input, and an input the schema rejects fails the call with the API's own message instead of starting a +container. + +```bash +apify call # runs on the schema's defaults +apify call --input '{"maxPages":0}' # 400 Input is not valid: Field input.maxPages must be >= 1 +``` + +The schema is read at build time, so editing it locally needs an `apify push` even under a registered dev folder, +and a schema the Apify meta-schema rejects fails the build with the defect in its log. Proxy group availability is +not checked locally, and encrypted secret input fields stay unsupported - see +`requirements/actor-driver.md`'s "Input schema, validation and defaults". + ## Running with Podman instead of Docker The runtime talks to the container engine only through its Docker-compatible API socket, and Podman @@ -218,7 +235,9 @@ do. Like Python debug mode, this needs the runtime to run from its own built ima Both bundled samples charge two events when their Actor is priced - `page-scraped` once per page and `crawl-finished` once at the end - and ship the pricing that defines them in `pricing.json`, so a run -charges for real right after a push: +charges for real right after a push. That pricing also declares both synthetic events, which the Actor +never charges itself: `apify-actor-start` at run start (once per whole GB of the run's memory) and +`apify-default-dataset-item` per item pushed to the default dataset. ```bash cd sample_actor_ts # or sample_actor_py @@ -236,6 +255,19 @@ run's console page and the runs list show the same figures. platform: an update sends the Actor's existing entries unchanged plus at most one new one, starting after all of them. Making the Actor free again is therefore appending a `{"pricingModel": "FREE"}` entry. +The `PUT` above therefore prices an Actor that has no pricing yet. Once it has one, a second `PUT` of the +same file is refused (`pricingInfos[0] differs from the Actor's existing pricing info`) - the stored +entries carry the timestamps they were given, which the file does not. Append the file's entry to what +the Actor already has instead: + +```bash +apify api PUT /v2/actors/ --body "$(apify api GET /v2/actors/ | + jq --argjson new "$(jq '.pricingInfos[-1]' pricing.json)" '{pricingInfos: (.data.pricingInfos + [$new])}')" +``` + +The Actor's console page has the same thing as a form: the box holds the stored array, and adding an +entry below the existing ones does it without the shell. + Cap a run's spend the way a user does - the cap is a query parameter, with no `apify call` flag for it: ```bash diff --git a/package.json b/package.json index cb3fbc6..0e6c2a5 100644 --- a/package.json +++ b/package.json @@ -26,9 +26,11 @@ "test:watch": "vitest" }, "dependencies": { + "@apify/input_schema": "^3.29.2", "@crawlee/core": "4.0.0-beta.145", "@crawlee/fs-storage": "4.0.0-beta.145", "@novnc/novnc": "1.7.0", + "ajv": "^8.20.0", "dockerode": "^4.0.5", "express": "^5.1.0", "json5": "^2.2.3", diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 8127ff0..5c528b0 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -11,6 +11,9 @@ importers: .: dependencies: + '@apify/input_schema': + specifier: ^3.29.2 + version: 3.29.2(ajv@8.20.0) '@crawlee/core': specifier: 4.0.0-beta.145 version: 4.0.0-beta.145 @@ -20,6 +23,9 @@ importers: '@novnc/novnc': specifier: 1.7.0 version: 1.7.0 + ajv: + specifier: ^8.20.0 + version: 8.20.0 dockerode: specifier: ^4.0.5 version: 4.0.12 @@ -90,12 +96,29 @@ packages: '@apify/consts@2.57.2': resolution: {integrity: sha512-iCAltDfZqEZazrMFoGIp3EEkvd7GTlgs9y74QZ+JTz+zH3GqBvOD2DfW4WYsqI1rfKlLyo5z3S7wipWpkSNNJQ==} + '@apify/consts@2.58.1': + resolution: {integrity: sha512-pKnoPincN2eqiSANmX2u19FihJATX3f1ucMhxss7t2epUmB/UhSjGM7G4immjC7Q/CXrNmbETBbOGmR0/EH4/Q==} + '@apify/datastructures@2.0.6': resolution: {integrity: sha512-hI/Q5cCjXXI/g4OPwX+/xuk+JsOtXRhKNXl2wnq/nKmdrji3nYCc9vGsAEc1iXVbs16xQMZy5uRhoY3XTABovA==} + '@apify/input_schema@3.29.2': + resolution: {integrity: sha512-hKW/7w1B7nu3v1gXeWAXhXV/qCQnZ1d/NdXUSVB4Ryb162Lp5Tn+lOjDbYpb7/+Qc5Y5JvOx/fLLg6/qdQ6WzQ==} + peerDependencies: + ajv: ^8.0.0 + + '@apify/input_secrets@1.2.60': + resolution: {integrity: sha512-KfrnFFTy7pozNxgX6MBdeCbVIhxM/t6t8gDroRbADJyYmaGtkqsxk9t4/xw9ri+ZObiWdWQrlU10xF+NUqDQBg==} + + '@apify/json_schemas@0.17.2': + resolution: {integrity: sha512-nN1QiBX+FJJyOG2OTjE6X6mG1EPYz5Je5TA4JmQ8JyWmWilQ/6hBDB6mUyg1JmvS3QDd+swtZS0G3dJ7sP+6uQ==} + '@apify/log@2.5.50': resolution: {integrity: sha512-obSTa6SUP4Ywy9EqcCMoHZXL0kH4lGTgajqlqTVkkdchLKAGNv+hsNLRSXz15N5aHJXwykqdWxidWJrRNawuxA==} + '@apify/log@2.5.54': + resolution: {integrity: sha512-EZRHrKCWnet0+6WaTx14Mpjxrn/PRudmRk6/N1dCTz+jtHN+Pbdvw5FFtIrCUngPhm4+bdI+kKVYUBmvTwrjvg==} + '@apify/ps-tree@1.2.0': resolution: {integrity: sha512-VHIswI7rD/R4bToeIDuJ9WJXt+qr5SdhfoZ9RzdjmCs9mgy7l0P4RugQEUCcU+WB4sfImbd4CKwzXcn0uYx1yw==} engines: {node: '>= 0.10'} @@ -104,6 +127,9 @@ packages: '@apify/timeout@0.4.10': resolution: {integrity: sha512-RlsqjT47MX5W5DC8yG1sBq38qZCYRGia6gCmmqiio1Cg6TWjIRlrUfFCY3lVuxn9MOPCOcHqxsEqsVvFX5990g==} + '@apify/utilities@2.35.11': + resolution: {integrity: sha512-od9Axa/+OSZB5JO8aljAnN7lS8vDb5KqUEAkUiZZh95K5xIrC0GzN45DNcK8oItnJ3QT/oqtHdaa22TipinDvQ==} + '@apify/utilities@2.35.7': resolution: {integrity: sha512-UUJkBCCWpOdTPpGenGtkNnnUbwmgANwxJDZoDPfa27+V23V8UfurE/lGA0LCGDH6DWYYJkQoEAtehxYJHR0U9g==} @@ -919,6 +945,10 @@ packages: peerDependencies: acorn: ^6.0.0 || ^7.0.0 || ^8.0.0 + acorn-loose@8.5.2: + resolution: {integrity: sha512-PPvV6g8UGMGgjrMu+n/f9E/tCSkNQ2Y97eFvuVdJfG11+xdIeDcLyNdC8SHcrHbRqkfwLASdplyR6B6sKM1U4A==} + engines: {node: '>=0.4.0'} + acorn@8.18.0: resolution: {integrity: sha512-lGq+9yr1/GuAWaVYIHRjvvySG5/4VfKIvC8EWxStPdcDh/Ka7FG3twP6v4d5BkravUilhIAsG4Qj83t02LWUPQ==} engines: {node: '>=0.4.0'} @@ -935,6 +965,9 @@ packages: ajv@6.15.0: resolution: {integrity: sha512-fgFx7Hfoq60ytK2c7DhnF8jIvzYgOMxfugjLOSMHjLIPgenqa7S7oaagATUq99mV6IYvN2tRmC0wnTYX6iPbMw==} + ajv@8.20.0: + resolution: {integrity: sha512-Thbli+OlOj+iMPYFBVBfJ3OmCAnaSyNn4M1vz9T6Gka5Jt9ba/HIR56joy65tY6kx/FCF5VXNB819Y7/GUrBGA==} + ansi-colors@4.1.3: resolution: {integrity: sha512-/6w/C21Pm1A7aZitlI5Ni/2J6FFQN8i1Cvz3kHABAAbw93v/NlvKdVOqz7CCWz/3iv/JplRSEEZ83XION15ovw==} engines: {node: '>=6'} @@ -1139,6 +1172,9 @@ packages: resolution: {integrity: sha512-yki5XnKuf750l50uGTllt6kKILY4nQ1eNIQatoXEByZ5dWgnKqbnqmTrBE5B4N7lrMJKQ2ytWMiTO2o0v6Ew/w==} engines: {node: '>= 0.6'} + countries-list@3.4.1: + resolution: {integrity: sha512-sKLGAZD5EAdzQwQ19vOp/9u33MTjmZ0CkdXPYtp/An6cyIlJ/E8TwGf2FcPe1AnzHO+h7nbTfoHyXIi58Hpa+Q==} + cpu-features@0.0.10: resolution: {integrity: sha512-9IkYqtX3YHPCzoVg1Py+o9057a3i0fp7S530UWokCSaFVTc7CwXPRiOjRjBQQ18ZCNafx78YfnG+HALxtVmOGA==} engines: {node: '>=10.0.0'} @@ -1385,6 +1421,9 @@ packages: fast-levenshtein@2.0.6: resolution: {integrity: sha512-DCXu6Ifhqcks7TZKY3Hxp3y6qphY5SJZmrWMDrKcERSOXWQdMhU9Ig/PYrzyw/ul9jOIyh0N4M0tbC5hodg8dw==} + fast-uri@3.1.8: + resolution: {integrity: sha512-GZMtZUTNRpOVIECoXwLNZS5xUGE+mVNbTB8h/7Rwh2TFWcBQiPzTgyZi05BF9UMZKkLJv8XBRJTlU7zg8+ZfMg==} + fdir@6.5.0: resolution: {integrity: sha512-tIbYtZbucOs0BRGqPJkshJUYdL+SDH7dVM8gjy+ERp3WAUjLEFJE+02kanyHtwjWOnwrKYBiwAmM0p4kLJAnXg==} engines: {node: '>=12.0.0'} @@ -1585,6 +1624,9 @@ packages: json-schema-traverse@0.4.1: resolution: {integrity: sha512-xbbCH5dCYU5T8LcEhhuh7HJ88HXuW3qsI3Y0zOZFKfZEHcpWiHU/Jxzk629Brsab/mMiHQti9wMP+845RPe3Vg==} + json-schema-traverse@1.0.0: + resolution: {integrity: sha512-NM8/P9n3XjXhIZn1lLhkFaACTOURQXjWhV4BA/RnOv8xvgqtqpAX9IO4mRQxSx1Rlo4tqzeqb0sOlruaOy3dug==} + json-stable-stringify-without-jsonify@1.0.1: resolution: {integrity: sha512-Bdboy+l7tA3OGW6FjyFHWkP5LuByj1Tk33Ljyq0axyzdk9//JSi2u3fP1QSmd1KNwq6VOKYGlAu87CisVir6Pw==} @@ -1832,6 +1874,10 @@ packages: resolution: {integrity: sha512-fGxEI7+wsG9xrvdjsrlmL22OMTTiHRwAMroiEeMgq8gzoLC/PQr7RsRDSTLUg/bZAZtF+TVIkHc6/4RIKrui+Q==} engines: {node: '>=0.10.0'} + require-from-string@2.0.2: + resolution: {integrity: sha512-Xf0nWe6RseziFMu+Ap9biiUbmplq6S9/p+7w7YXP/JBHhrUDDUhwa+vANyubuqfZWTveU//DYVGsDG7RKL/vEw==} + engines: {node: '>=0.10.0'} + resolve-from@4.0.0: resolution: {integrity: sha512-pb/MYmXstAkysRFx8piNI1tGFNQIFA3vkE3Gq4EuA1dF6gHp/+vgZqsCGJapvy8N3Q+4o7FwvquPJcnZ7RYy4g==} engines: {node: '>=4'} @@ -2245,19 +2291,51 @@ snapshots: '@apify/consts@2.57.2': {} + '@apify/consts@2.58.1': {} + '@apify/datastructures@2.0.6': {} + '@apify/input_schema@3.29.2(ajv@8.20.0)': + dependencies: + '@apify/consts': 2.58.1 + '@apify/input_secrets': 1.2.60 + '@apify/json_schemas': 0.17.2 + acorn-loose: 8.5.2 + ajv: 8.20.0 + countries-list: 3.4.1 + + '@apify/input_secrets@1.2.60': + dependencies: + '@apify/log': 2.5.54 + '@apify/utilities': 2.35.11 + ow: 0.28.2 + + '@apify/json_schemas@0.17.2': + dependencies: + '@apify/consts': 2.58.1 + ajv: 8.20.0 + '@apify/log@2.5.50': dependencies: '@apify/consts': 2.57.2 ansi-colors: 4.1.3 + '@apify/log@2.5.54': + dependencies: + '@apify/consts': 2.58.1 + ansi-colors: 4.1.3 + '@apify/ps-tree@1.2.0': dependencies: event-stream: 3.3.4 '@apify/timeout@0.4.10': {} + '@apify/utilities@2.35.11': + dependencies: + '@apify/consts': 2.58.1 + '@apify/log': 2.5.54 + '@apify/utilities@2.35.7': dependencies: '@apify/consts': 2.57.2 @@ -2937,6 +3015,10 @@ snapshots: dependencies: acorn: 8.18.0 + acorn-loose@8.5.2: + dependencies: + acorn: 8.18.0 + acorn@8.18.0: {} agent-base@6.0.2: @@ -2954,6 +3036,13 @@ snapshots: json-schema-traverse: 0.4.1 uri-js: 4.4.1 + ajv@8.20.0: + dependencies: + fast-deep-equal: 3.1.3 + fast-uri: 3.1.8 + json-schema-traverse: 1.0.0 + require-from-string: 2.0.2 + ansi-colors@4.1.3: {} ansi-regex@5.0.1: {} @@ -3174,6 +3263,8 @@ snapshots: cookie@0.7.2: {} + countries-list@3.4.1: {} + cpu-features@0.0.10: dependencies: buildcheck: 0.0.7 @@ -3515,6 +3606,8 @@ snapshots: fast-levenshtein@2.0.6: {} + fast-uri@3.1.8: {} + fdir@6.5.0(picomatch@4.0.5): optionalDependencies: picomatch: 4.0.5 @@ -3711,6 +3804,8 @@ snapshots: json-schema-traverse@0.4.1: {} + json-schema-traverse@1.0.0: {} + json-stable-stringify-without-jsonify@1.0.1: {} json5@2.2.3: {} @@ -3958,6 +4053,8 @@ snapshots: require-directory@2.1.1: {} + require-from-string@2.0.2: {} + resolve-from@4.0.0: {} retry@0.13.1: {} diff --git a/requirements/actor-driver.md b/requirements/actor-driver.md index 3b1d78e..a688894 100644 --- a/requirements/actor-driver.md +++ b/requirements/actor-driver.md @@ -34,6 +34,20 @@ 4. the platform's bundled default Dockerfile, for that build only - the pushed source itself is unchanged. - Matching is case-insensitive, exact-case wins ties, and every outcome is stated in the build log. +# Input schema, validation and defaults + +- **Input schemas work as on the Apify platform**: the schema is read from the pushed source when the + Actor is built - the `input` field of `.actor/actor.json`, else `.actor/INPUT_SCHEMA.json`, else + `INPUT_SCHEMA.json` - a build whose schema cannot be read or is not a valid input schema fails with + the reason in its log, and every run of a build is validated against that build's schema with its + defaults applied, a rejected input starting nothing (`api.md`). A build with no input schema takes + every input exactly as the caller sent it. +- **Differences**: Apify Proxy group availability is not checked, so any `apifyProxyGroups` selection + is accepted, while the rest of a `proxy` field is still validated; encrypted secret input fields + stay unsupported (`unsupported.md`). +- An Actor running from a registered dev folder uses its last build's schema: unlike a source edit, + an edited input schema takes effect only after `apify push`. + # Bind mount volumes with Actor source code - To let an Actor be re-run with source changes and no rebuild, the Actor's registered local dev @@ -215,6 +229,8 @@ start`, ...) is refused by name, naming both the `CMD` fix and how to clear debu per subscription tier is charged at the lowest paid tier; nothing is ever billed or paid out; and the rules tying a price change to payout details, notice periods and subscription tiers do not apply. - An Actor's own charging code therefore runs here unchanged, with no local-testing switch. +- A run's log states what it was pre-charged for starting, so the count is visible where the platform + leaves it to be discovered on the bill. - The pricing can also be set from the console (`console.md`). # Users diff --git a/requirements/api.md b/requirements/api.md index 2622ea0..aa39b93 100644 --- a/requirements/api.md +++ b/requirements/api.md @@ -17,6 +17,12 @@ - `DELETE /v2/actor-builds/:buildId` and `DELETE /v2/actor-runs/:runId` on a **non-terminal** build/run are rejected, not aborted-then-deleted: `400` with error type `deleting-unfinished-build` (builds) or `cannot-remove-running-run` (runs), matching the Apify platform. +- `POST /v2/actors/:actorId/runs` validates the input against the input schema of the build it + resolved, when that build has one (`actor-driver.md`), and starts nothing when it does not pass: + `400` `invalid-input` for a body that is not `application/json`, is not parseable JSON, is not a + JSON object, or that the schema rejects, naming every offending field; `400` `invalid-input-schema` + when the Actor's own schema is not valid. Both messages match the Apify platform's. A build with no + input schema accepts any body, unvalidated. - Four endpoints are exceptions to the `{data}` envelope: - `GET /v2/logs/:buildOrRunId` (and its `actor-builds`/`actor-runs` aliases): the body is plain text, never `{data}`-wrapped, matching apify-client-js's `log().get()`. @@ -305,9 +311,9 @@ This runtime emulates that observable experience on demand: - `fallbackNotFoundEnabled` covers a request that reaches a route this runtime does serve, but whose specific record id doesn't exist locally (`record-not-found`, see "Response envelopes" above). - - Every other error type - `invalid-request`, `user-not-authenticated`, - `cannot-remove-running-run`, `deleting-unfinished-build`, any `dev-folder-*` type, - `internal-error` - is never relayed, regardless of either toggle's state. + - Every other error type - `invalid-request`, `invalid-input`, `invalid-input-schema`, + `user-not-authenticated`, `cannot-remove-running-run`, `deleting-unfinished-build`, any + `dev-folder-*` type, `internal-error` - is never relayed, regardless of either toggle's state. - **All HTTP methods are eligible for both toggles, writes included**: a `POST`/`PUT`/`DELETE` that would otherwise 404/501 locally is relayed exactly like a `GET` when its toggle is on - and, if the platform accepts it, becomes a real write against the caller's real account. This is a deliberate diff --git a/requirements/test.md b/requirements/test.md index 495fadb..5af7c7d 100644 --- a/requirements/test.md +++ b/requirements/test.md @@ -36,6 +36,7 @@ Test case must verify full Actor development flow: - Push and build Actor in local actor runtime `apify push` - Run each sample Actor in the local actor runtime with `apify call --input '{"maxPages":N}'` for at least two different values of `N`, waiting for each run to finish - Assert via `apify datasets info ` that the default dataset's `itemCount` tracks `N` - the assertion is input-dependent, not just "some items exist" +- Cover the input schema through the CLI too (`actor-driver.md`): `apify call` with no `--input` must run on the schema's defaults, asserted from the run's own `INPUT`, and `apify call` with an input the schema rejects must fail, naming the offending field ## Browser view diff --git a/requirements/unsupported.md b/requirements/unsupported.md index 6202390..1029678 100644 --- a/requirements/unsupported.md +++ b/requirements/unsupported.md @@ -24,7 +24,6 @@ real account, not in the runtime. - Ad-hoc webhooks on run start - Metered usage other than compute units: storage operations, data transfer and proxy - Billing: a run's charges and costs are reported, never invoiced or paid out -- Input validation and defaults from the input schema - Encrypted secret input fields - Actor-level default run options - Dynamic and bounded memory from `.actor/actor.json` @@ -101,7 +100,8 @@ real account, not in the runtime. ## Platform limits not enforced -- Memory steps and bounds (128 MB - 32 GB, powers of two) +- Memory steps and bounds (128 MB - 32 GB, powers of two); a run asking for anything else is warned + about in its log and started with it anyway - Record, item and input size limits - Concurrent run, rate and per-account quotas - Process, file-descriptor and shared-memory limits diff --git a/sample_actor_crawler/src/main.py b/sample_actor_crawler/src/main.py index 157b62e..ddafa59 100644 --- a/sample_actor_crawler/src/main.py +++ b/sample_actor_crawler/src/main.py @@ -24,11 +24,13 @@ async def request_handler(context: ParselCrawlingContext) -> None: async def main() -> None: async with Actor: + # Both fields have a `default` in the input schema, so the runtime fills them in before the + # run starts (the Apify platform does the same) - the Actor needs no fallback of its own. actor_input = await Actor.get_input() or {} - start_url = actor_input.get("startUrl", "https://crawlee.dev") + start_url = actor_input["startUrl"] proxy_configuration = await Actor.create_proxy_configuration( - actor_proxy_input=actor_input.get("proxyConfiguration") + actor_proxy_input=actor_input["proxyConfiguration"] ) crawler = ParselCrawler( diff --git a/sample_actor_nonstandard/app/main.py b/sample_actor_nonstandard/app/main.py index 40409ac..ef1f3f5 100644 --- a/sample_actor_nonstandard/app/main.py +++ b/sample_actor_nonstandard/app/main.py @@ -46,7 +46,9 @@ def main() -> None: raw_input = api_request('GET', f'key-value-stores/{key_value_store_id}/records/INPUT') actor_input = json.loads(raw_input) if raw_input else {} - item_count = int(actor_input.get('itemCount', 2)) + # `itemCount` has a `default` in the input schema, so the runtime fills it in before the run + # starts - no fallback needed here. + item_count = int(actor_input['itemCount']) print(f'main.py: pushing {item_count} item(s) to dataset {dataset_id}.') # Both accepted body shapes of `POST /v2/datasets/:id/items`: one object, then an array. diff --git a/sample_actor_playwright/.actor/input_schema.json b/sample_actor_playwright/.actor/input_schema.json index fe27fb2..e93f71d 100644 --- a/sample_actor_playwright/.actor/input_schema.json +++ b/sample_actor_playwright/.actor/input_schema.json @@ -9,7 +9,8 @@ "type": "array", "description": "URLs to start with.", "editor": "requestListSources", - "prefill": [{ "url": "https://crawlee.dev/" }] + "prefill": [{ "url": "https://crawlee.dev/" }], + "default": [{ "url": "https://crawlee.dev/" }] }, "maxRequestsPerCrawl": { "title": "Max requests per crawl", diff --git a/sample_actor_playwright/src/main.ts b/sample_actor_playwright/src/main.ts index 331c943..4da395f 100644 --- a/sample_actor_playwright/src/main.ts +++ b/sample_actor_playwright/src/main.ts @@ -24,9 +24,11 @@ interface Input { // Initialize the Apify SDK await Actor.init(); -// Structure of input is defined in .actor/input_schema.json -const { startUrls = [{ url: 'https://crawlee.dev/' }], maxRequestsPerCrawl = 3 } = - (await Actor.getInput()) ?? ({} as Input); +// Structure of input is defined in .actor/input_schema.json. Every field has a `default` there, so +// the runtime fills it in before the run starts - the Actor needs no fallback of its own. +const input = await Actor.getInput(); +if (!input) throw new Error('No input: the input schema should have supplied its defaults.'); +const { startUrls, maxRequestsPerCrawl } = input; // Without a proxy password (a plain local run) the crawler connects directly instead of failing the access check. const proxyConfiguration = process.env.APIFY_PROXY_PASSWORD diff --git a/sample_actor_playwright_py/.actor/input_schema.json b/sample_actor_playwright_py/.actor/input_schema.json index 0f3040c..958f396 100644 --- a/sample_actor_playwright_py/.actor/input_schema.json +++ b/sample_actor_playwright_py/.actor/input_schema.json @@ -9,7 +9,8 @@ "type": "array", "description": "URLs to start with.", "editor": "requestListSources", - "prefill": [{ "url": "https://crawlee.dev/" }] + "prefill": [{ "url": "https://crawlee.dev/" }], + "default": [{ "url": "https://crawlee.dev/" }] }, "max_requests_per_crawl": { "title": "Max requests per crawl", diff --git a/sample_actor_playwright_py/my_actor/main.py b/sample_actor_playwright_py/my_actor/main.py index c3868a8..b08a199 100644 --- a/sample_actor_playwright_py/my_actor/main.py +++ b/sample_actor_playwright_py/my_actor/main.py @@ -23,10 +23,11 @@ async def main() -> None: """ # Enter the context of the Actor. async with Actor: - # Retrieve the Actor input, and use default values if not provided. + # Every field has a `default` in the input schema, so the runtime fills it in before the run + # starts (the Apify platform does the same) - the Actor needs no fallback of its own. actor_input = await Actor.get_input() or {} - start_urls = [url.get('url') for url in actor_input.get('start_urls', [{'url': 'https://crawlee.dev/'}])] - max_requests_per_crawl = int(actor_input.get('max_requests_per_crawl', 3)) + start_urls = [url.get('url') for url in actor_input['start_urls']] + max_requests_per_crawl = int(actor_input['max_requests_per_crawl']) # Exit if no start URLs are provided. if not start_urls: diff --git a/sample_actor_py/pricing.json b/sample_actor_py/pricing.json index 996121b..3d1b3db 100644 --- a/sample_actor_py/pricing.json +++ b/sample_actor_py/pricing.json @@ -19,6 +19,11 @@ "eventDescription": "Charged by the runtime at run start, once per GB of memory.", "eventPriceUsd": 0.005, "isOneTimeEvent": true + }, + "apify-default-dataset-item": { + "eventTitle": "Dataset item", + "eventDescription": "Charged by the runtime for every item pushed to the default dataset.", + "eventPriceUsd": 0.001 } } } diff --git a/sample_actor_py/src/main.py b/sample_actor_py/src/main.py index 6d3e214..6fe907f 100644 --- a/sample_actor_py/src/main.py +++ b/sample_actor_py/src/main.py @@ -30,15 +30,19 @@ async def log_resource_usage(event_data: EventSystemInfoData) -> None: # Under pay-per-event pricing (set on the Actor through the runtime's API or console) this Actor # charges two events: 'page-scraped' once per page and 'crawl-finished' once at the end. A free - # Actor skips both, so a plain push-and-call stays unchanged. + # Actor skips both, so a plain push-and-call stays unchanged. `pricing.json` prices two more, + # 'apify-actor-start' and 'apify-default-dataset-item': those are charged by the runtime itself, + # which is why no code here charges them. pricing = Actor.get_charging_manager().get_pricing_info() if pricing.is_pay_per_event: cap = f'${pricing.max_total_charge_usd}' if pricing.max_total_charge_usd.is_finite() else 'none' Actor.log.info(f'Pay-per-event pricing in effect, max total charge: {cap}.') + # Both fields have a `default` in the input schema, so the runtime fills them in before the + # run starts (the Apify platform does the same) - the Actor needs no fallback of its own. actor_input = await Actor.get_input() or {} - start_url = actor_input.get('startUrl', 'https://crawlee.dev/') - max_pages = int(actor_input.get('maxPages', 2)) + start_url = actor_input['startUrl'] + max_pages = int(actor_input['maxPages']) Actor.log.info(f'Crawling up to {max_pages} page(s) starting from {start_url}.') # Crawling through the Actor's default request queue exercises the runtime's diff --git a/sample_actor_ts/pricing.json b/sample_actor_ts/pricing.json index 996121b..3d1b3db 100644 --- a/sample_actor_ts/pricing.json +++ b/sample_actor_ts/pricing.json @@ -19,6 +19,11 @@ "eventDescription": "Charged by the runtime at run start, once per GB of memory.", "eventPriceUsd": 0.005, "isOneTimeEvent": true + }, + "apify-default-dataset-item": { + "eventTitle": "Dataset item", + "eventDescription": "Charged by the runtime for every item pushed to the default dataset.", + "eventPriceUsd": 0.001 } } } diff --git a/sample_actor_ts/src/main.ts b/sample_actor_ts/src/main.ts index 4dd153b..26d7702 100644 --- a/sample_actor_ts/src/main.ts +++ b/sample_actor_ts/src/main.ts @@ -7,8 +7,8 @@ import { CheerioCrawler } from '@crawlee/cheerio'; await Actor.init(); interface Input { - startUrl?: string; - maxPages?: number; + startUrl: string; + maxPages: number; } const { memoryMbytes } = Actor.getEnv(); @@ -27,16 +27,20 @@ Actor.on('systemInfo', (info: { cpuCurrentUsage?: number; memCurrentBytes?: numb // Under pay-per-event pricing (set on the Actor through the runtime's API or console) this Actor charges // two events: 'page-scraped' once per page and 'crawl-finished' once at the end. A free Actor skips both, -// so a plain push-and-call stays unchanged. +// so a plain push-and-call stays unchanged. `pricing.json` prices two more, 'apify-actor-start' and +// 'apify-default-dataset-item': those are charged by the runtime itself, which is why no code here +// charges them. const { isPayPerEvent, maxTotalChargeUsd } = Actor.getChargingManager().getPricingInfo(); if (isPayPerEvent) { const cap = Number.isFinite(maxTotalChargeUsd) ? `$${maxTotalChargeUsd}` : 'none'; log.info(`Pay-per-event pricing in effect, max total charge: ${cap}.`); } +// Both fields have a `default` in the input schema, so the runtime fills them in before the run +// starts (the Apify platform does the same) - the Actor needs no fallback of its own. const input = await Actor.getInput(); -const startUrl = input?.startUrl ?? 'https://crawlee.dev/'; -const maxPages = input?.maxPages ?? 2; +if (!input) throw new Error('No input: the input schema should have supplied its defaults.'); +const { startUrl, maxPages } = input; log.info(`Crawling up to ${maxPages} page(s) starting from ${startUrl}.`); diff --git a/skills/actor-runtime/SKILL.md b/skills/actor-runtime/SKILL.md index 58966cc..4ba84e5 100644 --- a/skills/actor-runtime/SKILL.md +++ b/skills/actor-runtime/SKILL.md @@ -53,6 +53,28 @@ both `apify/actor-node-playwright*` and `apify/actor-python-playwright*` are - a those is retried for `linux/amd64`, the architecture the Apify platform builds and runs on, with the reason in the build log. The engine emulates it (Rosetta on Apple Silicon), so it works, just slower. +## Input schema: defaults and validation + +If the Actor declares an input schema - the `input` field of `.actor/actor.json`, or +`.actor/INPUT_SCHEMA.json`, or `INPUT_SCHEMA.json` at its root - the runtime uses it the way the +platform does: + +```sh +apify call # no input: the run gets the schema's defaults +apify call --input '{"maxPages":0}' # rejected before anything starts, if the schema forbids it +``` + +- Every field left out of `--input` is filled from the schema's `default`, and that filled-in input is + what the run reads back as `INPUT`. +- An input the schema rejects fails the call with the same message the real API gives (`Input is not +valid: Field input.maxPages must be >= 1`); no run is created and no container starts. +- The schema comes from the **build**, so a locally edited `INPUT_SCHEMA.json` needs an `apify push` to + take effect - unlike a source edit under a registered dev folder. +- A schema the Apify input-schema meta-schema rejects fails the **build**, with the defect in the build + log, instead of being silently ignored. +- Two local differences: Apify Proxy groups are not checked (any `apifyProxyGroups` selection is + accepted), and encrypted secret input fields are not supported. + ## Iterate without rebuilding (dev folder) After that first `apify push`, the runtime registers the pushed directory as the Actor's **dev @@ -121,8 +143,9 @@ apify api PUT /v2/actors/ --body '{"pricingInfos":[{"pricingModel":"PAY ``` Both bundled samples already charge `page-scraped` per page and `crawl-finished` once at the end, and -carry the matching pricing in `pricing.json`: `apify api PUT /v2/actors/ --body "$(cat -sample_actor_ts/pricing.json)"` prices one in a single call. +carry the matching pricing in `pricing.json` - including both synthetic events, so one run shows all +four being charged: `apify api PUT /v2/actors/ --body "$(cat sample_actor_ts/pricing.json)"` +prices one in a single call. From then on every run of the Actor is a pay-per-event run: the SDKs read `pricingInfo` and `chargedEventCounts` off the run object exactly as on the platform, and `Actor.charge()` (or @@ -130,10 +153,15 @@ From then on every run of the Actor is a pay-per-event run: the SDKs read `prici `ACTOR_TEST_PAY_PER_EVENT` - the runtime already looks like the platform to the SDK, which refuses that variable together with `APIFY_IS_AT_HOME`. The synthetic events work too: `apify-actor-start` is charged at run start (once per GB of memory) and `apify-default-dataset-item` once per item pushed to the run's -default dataset, when the pricing defines them. Tiered event prices resolve to the `BRONZE` (Starter) -tier. The array is append-only, as on the platform: send the entries the Actor already has, unchanged, +default dataset, when the pricing defines them. The start pre-charge scales with the run's memory and +its count is written to the run log, so `-m 8192` charges eight start events, not one; a memory the +platform would refuse (anything but a power of two between 128 MB and 32 GB) is warned about in the log +and then used anyway. Tiered event prices resolve to the `BRONZE` (Starter) tier. The array is append-only, as on the platform: send the entries the Actor already has, unchanged, plus at most one new one starting after all of them - so making the Actor free again means appending a -`{"pricingModel":"FREE"}` entry, not sending `[]`. The same form is on the Actor's console page. +`{"pricingModel":"FREE"}` entry, not sending `[]`. Re-sending a file that was already applied is refused +for the same reason (`pricingInfos[0] differs ...`): the stored entries carry timestamps the file does +not, so append its entry to what `GET /v2/actors/` returns rather than sending the file again. +The same form is on the Actor's console page. Cap a run's spend like a user would: `apify api POST '/v2/actors//runs?maxTotalChargeUsd=0.5'` (there is no `apify call` flag for it). When the charges reach the cap the run is aborted gracefully, diff --git a/src/api/errors.ts b/src/api/errors.ts index 00b7ab4..2d0f6dc 100644 --- a/src/api/errors.ts +++ b/src/api/errors.ts @@ -30,6 +30,17 @@ export function endpointNotFound(message: string): ApiError { return new ApiError(404, 'not-found', message); } +/** Matches the real platform's rejection of an input its Actor's input schema does not accept + * (`@apify-packages/errors`'s `actor.inputNotValid` and its neighbours), message included. */ +export function invalidInput(message: string): ApiError { + return new ApiError(400, 'invalid-input', message); +} + +/** Matches the real platform's `actor.invalidInputSchema`. */ +export function invalidInputSchema(message: string): ApiError { + return new ApiError(400, 'invalid-input-schema', message); +} + /** * Matches the real Apify platform exactly: `DELETE /v2/actor-runs/:runId` on a non-terminal run is * rejected rather than aborted-then-deleted (the public API answers 400 `cannot-remove-running-run`), diff --git a/src/api/routes/actors.ts b/src/api/routes/actors.ts index a2baad4..418e336 100644 --- a/src/api/routes/actors.ts +++ b/src/api/routes/actors.ts @@ -3,7 +3,14 @@ import type { Router } from 'express'; import { requireUser } from '../auth.js'; import { paginate, sendData, sendPaginated, sortByTimestamp } from '../envelope.js'; -import { ApiError, cannotSetPricingOnCreate, recordNotFound, invalidRequest } from '../errors.js'; +import { + ApiError, + cannotSetPricingOnCreate, + invalidInput, + invalidInputSchema, + invalidRequest, + recordNotFound, +} from '../errors.js'; import { h, jsonBody, paginationParams, queryBoolean, queryNumber, queryString, rawBody } from '../handler.js'; import { addOrReplaceVersion, @@ -30,6 +37,7 @@ import type { ApiServerDeps } from '../server.js'; import { CONTAINER_API_BASE_URL } from '../../config.js'; import { resolveProxyPassword } from '../../services/users.js'; import { validatePricingInfosUpdate } from '../../services/pricing.js'; +import { resolveBuildInput } from '../../services/input-schema.js'; /** * `undefined` when the body does not mention the field; a body that does but is invalid throws, with the @@ -275,8 +283,15 @@ export function mountActors(router: Router, deps: ApiServerDeps): void { const build = lookup.build; const body = rawBody(req); - const input = - body.length > 0 ? { body, contentType: req.header('content-type') ?? 'application/json' } : undefined; + const processed = resolveBuildInput( + build, + body.length > 0 ? { body, contentType: req.header('content-type') ?? 'application/json' } : undefined, + ); + if (processed.kind !== 'ok') { + throw processed.kind === 'invalid-input-schema' + ? invalidInputSchema(processed.message) + : invalidInput(processed.message); + } // `resolveProxyPassword(requireUser(req))` is the *run owner's* proxy password, not just "the // caller's": `actor` was resolved via `resolveActorParam(req)` above, so @@ -284,7 +299,7 @@ export function mountActors(router: Router, deps: ApiServerDeps): void { // their own Actor - which makes the two the same user record (`actor-driver.md`'s "one // harvested-per-account password used specifically for each user"). const run = await startRun(deps.driver, actor, build, { - input, + input: processed.input, memoryMbytes: queryNumber(req, 'memory'), timeoutSecs: queryNumber(req, 'timeout'), maxTotalChargeUsd, diff --git a/src/resources.ts b/src/resources.ts index fcf7b17..1b8d8e6 100644 --- a/src/resources.ts +++ b/src/resources.ts @@ -17,3 +17,30 @@ export function cpuQuotaFor(memoryMbytes: number): number { const rawQuota = dedicatedCpusFor(memoryMbytes) * CPU_PERIOD_US; return Math.max(MIN_CPU_QUOTA_US, Math.round(rawQuota)); } + +/** The platform's accepted memory grants: powers of two, from 128 MB to 32 GB. */ +const MIN_PLATFORM_MEMORY_MBYTES = 128; +const MAX_PLATFORM_MEMORY_MBYTES = 32_768; + +/** + * A warning for a memory grant the Apify platform would refuse outright, or `undefined` for one it + * accepts. This runtime applies whatever it is given (`unsupported.md` - memory steps and bounds are + * deliberately not enforced), so the run still starts; what the warning buys is that a developer finds + * out here rather than on the first platform run. Worth saying for pay-per-event Actors especially: the + * pre-charged `apify-actor-start` count is one per *whole* gigabyte, so a grant between two steps pays + * for the lower one. + */ +export function platformIncompatibleMemoryWarning(memoryMbytes: number): string | undefined { + const defects: string[] = []; + if (memoryMbytes < MIN_PLATFORM_MEMORY_MBYTES || memoryMbytes > MAX_PLATFORM_MEMORY_MBYTES) { + defects.push(`outside the ${MIN_PLATFORM_MEMORY_MBYTES} MB - ${MAX_PLATFORM_MEMORY_MBYTES} MB range`); + } + if (!Number.isInteger(Math.log2(memoryMbytes))) defects.push('not a power of two'); + if (defects.length === 0) return undefined; + + return ( + `Warning: the requested memory of ${memoryMbytes} MB is ${defects.join(' and ')}, which the Apify ` + + `platform refuses (it accepts powers of two from ${MIN_PLATFORM_MEMORY_MBYTES} MB to ` + + `${MAX_PLATFORM_MEMORY_MBYTES} MB). Running with it anyway.` + ); +} diff --git a/src/services/actor-source-files.ts b/src/services/actor-source-files.ts new file mode 100644 index 0000000..a6d97dd --- /dev/null +++ b/src/services/actor-source-files.ts @@ -0,0 +1,92 @@ +/** + * What both build-time resolvers (`dockerfile-location.ts`, `input-schema-location.ts`) need from a + * version's pushed `sourceFiles`: the parsed `.actor/actor.json`, and the path fields inside it that + * name another pushed file. + * + * Shared rather than written per field so the Actor-root containment check has one implementation: + * a traversal hole patched in one copy would otherwise stay open in the other. + */ +import * as path from 'node:path'; +import JSON5 from 'json5'; + +import { normalizeEntryName } from '../driver/tar-entry-name.js'; +import type { SourceFile } from '../storage/entities.js'; + +export const ACTOR_DIR = '.actor'; +export const ACTOR_JSON_NAME = `${ACTOR_DIR}/actor.json`; + +export function sourceFileToText(file: SourceFile): string { + return file.format === 'BASE64' ? Buffer.from(file.content, 'base64').toString('utf8') : file.content; +} + +export interface IndexedFile { + normalizedName: string; + lowerName: string; + file: SourceFile; +} + +export function indexSourceFiles(sourceFiles: SourceFile[]): IndexedFile[] { + return sourceFiles.map((file) => { + const normalizedName = normalizeEntryName(file.name); + return { normalizedName, lowerName: normalizedName.toLowerCase(), file }; + }); +} + +/** Exact-case match wins; otherwise the first match in `sourceFiles` order. */ +export function findCaseInsensitive(indexed: IndexedFile[], candidate: string): IndexedFile | undefined { + const lowerCandidate = candidate.toLowerCase(); + let firstMatch: IndexedFile | undefined; + for (const file of indexed) { + if (file.lowerName !== lowerCandidate) continue; + if (file.normalizedName === candidate) return file; // exact case always wins immediately + firstMatch ??= file; + } + return firstMatch; +} + +/** `.actor/actor.json`'s own path is not case-folded, unlike the files its fields name. */ +export function findExact(sourceFiles: SourceFile[], normalizedTarget: string): SourceFile | undefined { + return sourceFiles.find((file) => normalizeEntryName(file.name) === normalizedTarget); +} + +/** `absent` is not an error: an Actor need not push `.actor/actor.json` at all. */ +export type ActorJsonParse = + { outcome: 'parsed'; specification: unknown } | { outcome: 'absent' } | { outcome: 'unparseable'; message: string }; + +export function parseActorJson(sourceFiles: SourceFile[]): ActorJsonParse { + const file = findExact(sourceFiles, ACTOR_JSON_NAME); + if (!file) return { outcome: 'absent' }; + try { + return { outcome: 'parsed', specification: JSON5.parse(sourceFileToText(file)) as unknown }; + } catch (error) { + return { outcome: 'unparseable', message: `Could not parse .actor/actor.json: ${(error as Error).message}` }; + } +} + +/** Where a `.actor/actor.json` path field points. `not-found` carries the path to name in the caller's + * warning - the empty field and a path naming no pushed file are the same outcome, both falling through + * to the default locations rather than failing. */ +export type ActorJsonPathField = + | { outcome: 'match'; file: IndexedFile } + | { outcome: 'not-found'; shownPath: string } + | { outcome: 'escapes-actor-root' }; + +/** `field` is resolved relative to `.actor/`, and may not leave the Actor root. */ +export function resolveActorJsonPathField(indexed: IndexedFile[], field: string): ActorJsonPathField { + if (field === '') return { outcome: 'not-found', shownPath: '' }; + if (field.startsWith('/')) return { outcome: 'escapes-actor-root' }; + + const joined = normalizeEntryName(path.posix.join(ACTOR_DIR, field)); + if (joined === '..' || joined.startsWith('../')) return { outcome: 'escapes-actor-root' }; + + const match = findCaseInsensitive(indexed, joined); + return match ? { outcome: 'match', file: match } : { outcome: 'not-found', shownPath: joined }; +} + +export function fallbackWarningLine(shownPath: string, fieldName: string): string { + return `Warning: "${shownPath}" (from the "${fieldName}" field in .actor/actor.json) is not in the pushed source; falling back to the default locations.\n`; +} + +export function escapesActorRootMessage(rawField: string, subject: string): string { + return `${subject} path "${rawField}" in .actor/actor.json points outside the Actor root directory.`; +} diff --git a/src/services/builds.ts b/src/services/builds.ts index 98b6089..9832597 100644 --- a/src/services/builds.ts +++ b/src/services/builds.ts @@ -6,8 +6,10 @@ import { recordTaggedBuild, updateActor } from './actors.js'; import type { Driver } from '../driver/types.js'; import { DriverTimedOutError } from '../driver/types.js'; import { normalizeEntryName } from '../driver/tar-entry-name.js'; +import { sourceFileToText } from './actor-source-files.js'; import { qualifyDockerfileImageReferences } from './dockerfile-image-refs.js'; import { resolveDockerfileLocation } from './dockerfile-location.js'; +import { resolveInputSchemaLocation } from './input-schema-location.js'; import { appendLog, appendRuntimeLog, flushLog, markLogTerminal } from './logs.js'; import { isTerminalJobStatus, transitionJobStatus } from './job-status.js'; @@ -89,8 +91,7 @@ function qualifyDockerfileImages( ): SourceFile[] { return sourceFiles.map((file) => { if (normalizeEntryName(file.name) !== dockerfilePath) return file; - const text = file.format === 'BASE64' ? Buffer.from(file.content, 'base64').toString('utf8') : file.content; - const { dockerfile, qualified } = qualifyDockerfileImageReferences(text); + const { dockerfile, qualified } = qualifyDockerfileImageReferences(sourceFileToText(file)); if (qualified.length === 0) return file; for (const { from, to } of qualified) { log(`Using "${to}" for FROM "${from}" - a short image name means Docker Hub, as on the platform.\n`); @@ -146,6 +147,19 @@ export async function startBuild( return record; } +/** Fails a build before any image exists, mirroring `services/runs.ts`'s `failBeforeContainer`: the + * reason reaches both the build log and the status message, which default to the same text. */ +async function failBuild(buildId: string, logMessage: string, statusMessage = logMessage): Promise { + const { builds } = getRegistries(); + appendRuntimeLog(buildId, logMessage); + await flushLog(buildId); + markLogTerminal(buildId); + await transitionJobStatus(builds, buildId, 'FAILED', { + finishedAt: new Date().toISOString(), + statusMessage, + }); +} + /** * Exported only for direct testing of the guarded transitions/pre-start abort window (see * `test/integration/job-lifecycle.test.ts`) - not part of the service's public surface for callers @@ -180,13 +194,7 @@ export async function runBuildInBackground( } if (!driver.available) { - appendRuntimeLog(record.id, `Docker is not available: ${driver.unavailableReason}`); - await flushLog(record.id); - markLogTerminal(record.id); - await transitionJobStatus(builds, record.id, 'FAILED', { - finishedAt: new Date().toISOString(), - statusMessage: driver.unavailableReason, - }); + await failBuild(record.id, `Docker is not available: ${driver.unavailableReason}`, driver.unavailableReason); return; } @@ -204,16 +212,21 @@ export async function runBuildInBackground( const dockerfileResolution = resolveDockerfileLocation(version.sourceFiles); if (dockerfileResolution.outcome === 'failure') { - appendRuntimeLog(record.id, dockerfileResolution.message); - await flushLog(record.id); - markLogTerminal(record.id); - await transitionJobStatus(builds, record.id, 'FAILED', { - finishedAt: new Date().toISOString(), - statusMessage: dockerfileResolution.message, - }); + await failBuild(record.id, dockerfileResolution.message); return; } for (const line of dockerfileResolution.logLines) appendRuntimeLog(record.id, line); + + // Before the image is built, not after: an Actor whose declared input contract cannot be read fails + // the build, rather than producing an image whose every run would skip validation. + const inputSchemaResolution = resolveInputSchemaLocation(version.sourceFiles); + if (inputSchemaResolution.outcome === 'failure') { + await failBuild(record.id, inputSchemaResolution.message); + return; + } + for (const line of inputSchemaResolution.logLines) appendRuntimeLog(record.id, line); + const inputSchema = inputSchemaResolution.outcome === 'resolved' ? inputSchemaResolution.schema : undefined; + const sourceFiles: SourceFile[] = qualifyDockerfileImages( dockerfileResolution.outcome === 'default' ? [...version.sourceFiles, dockerfileResolution.extraSourceFile] @@ -269,6 +282,7 @@ export async function runBuildInBackground( ...(outcome.imageWorkingDirectory !== undefined ? { imageWorkingDirectory: outcome.imageWorkingDirectory } : {}), + ...(inputSchema !== undefined ? { inputSchema } : {}), }); if (succeeded?.status !== 'SUCCEEDED') { await updateActor(actor.id, (current) => { diff --git a/src/services/charging.ts b/src/services/charging.ts index 93462c2..f349d1b 100644 --- a/src/services/charging.ts +++ b/src/services/charging.ts @@ -11,9 +11,14 @@ import type { RunRecord } from '../storage/entities.js'; import { getRegistries } from '../storage/registries.js'; import { isTerminalJobStatus } from './job-status.js'; import { appendRuntimeLog } from './logs.js'; -import { APIFY_EVENTS_PREFIX, DEFAULT_DATASET_ITEM_EVENT_NAME, isPayPerEvent } from './pricing.js'; +import { + ACTOR_START_EVENT_NAME, + APIFY_EVENTS_PREFIX, + DEFAULT_DATASET_ITEM_EVENT_NAME, + isPayPerEvent, +} from './pricing.js'; import { abortRun } from './runs.js'; -import { eventUsageFor, sumEventUsageUsd } from './run-usage.js'; +import { eventUsageFor, roundUsd, sumEventUsageUsd } from './run-usage.js'; /** The platform's own window. */ const IDEMPOTENCY_TTL_MS = 3 * 60 * 1000; @@ -148,6 +153,25 @@ export async function enforceCostLimit(driver: Driver, run: RunRecord): Promise< if (stamped.status === 'RUNNING') await abortRun(driver, stamped, true, message); } +/** + * What the run was pre-charged for starting, or `undefined` when nothing was. The platform charges this + * silently, and its count follows the run's memory rather than anything the Actor does - so a developer + * who raises the memory sees the bill rise with no line in the log to explain it. + */ +export function actorStartChargeMessage(run: RunRecord): string | undefined { + if (!isPayPerEvent(run.pricingInfo)) return undefined; + const count = run.chargedEventCounts?.[ACTOR_START_EVENT_NAME]; + if (!count) return undefined; + + const { eventPriceUsd } = run.pricingInfo.pricingPerEvent.actorChargeEvents[ACTOR_START_EVENT_NAME] ?? { + eventPriceUsd: 0, + }; + return ( + `Pre-charged ${count} '${ACTOR_START_EVENT_NAME}' event(s), $${roundUsd(count * eventPriceUsd)} in total, ` + + `for ${run.options.memoryMbytes} MB of memory - one event per whole gigabyte, minimum one.` + ); +} + export function resetChargingForTests(): void { idempotencyRecords.clear(); defaultDatasetRuns.clear(); diff --git a/src/services/dockerfile-location.ts b/src/services/dockerfile-location.ts index da7ea76..0bd530f 100644 --- a/src/services/dockerfile-location.ts +++ b/src/services/dockerfile-location.ts @@ -5,16 +5,19 @@ * case-insensitive; the returned path is always the matched file's own name, never the candidate's * casing - Docker's tar lookup is case-sensitive. An exact-case match wins over a case-differing one. */ -import * as path from 'node:path'; -import JSON5 from 'json5'; - import { normalizeEntryName } from '../driver/tar-entry-name.js'; import type { SourceFile } from '../storage/entities.js'; +import { + ACTOR_DIR, + escapesActorRootMessage, + fallbackWarningLine, + findCaseInsensitive, + indexSourceFiles, + parseActorJson, + resolveActorJsonPathField, +} from './actor-source-files.js'; import { DEFAULT_DOCKERFILE_CONTENT, DEFAULT_DOCKERFILE_NAME } from './default-dockerfile.js'; -const ACTOR_DIR = '.actor'; -const ACTOR_JSON_NAME = `${ACTOR_DIR}/actor.json`; - /** Why Dockerfile resolution failed. */ export type DockerfileResolutionFailureReason = 'escapes-actor-root' | 'invalid-dockerfile-field' | 'unparseable-actor-json'; @@ -26,44 +29,11 @@ export type DockerfileResolution = | { outcome: 'default'; dockerfilePath: string; logLines: string[]; extraSourceFile: SourceFile } | { outcome: 'failure'; reason: DockerfileResolutionFailureReason; message: string }; -function sourceFileToText(file: SourceFile): string { - return file.format === 'BASE64' ? Buffer.from(file.content, 'base64').toString('utf8') : file.content; -} - -interface IndexedFile { - normalizedName: string; - lowerName: string; -} - -function indexSourceFiles(sourceFiles: SourceFile[]): IndexedFile[] { - return sourceFiles.map((file) => { - const normalizedName = normalizeEntryName(file.name); - return { normalizedName, lowerName: normalizedName.toLowerCase() }; - }); -} - -/** Exact-case match wins; otherwise the first match in `sourceFiles` order. */ -function findCaseInsensitive(indexed: IndexedFile[], candidate: string): IndexedFile | undefined { - const lowerCandidate = candidate.toLowerCase(); - let firstMatch: IndexedFile | undefined; - for (const file of indexed) { - if (file.lowerName !== lowerCandidate) continue; - if (file.normalizedName === candidate) return file; // exact case always wins immediately - firstMatch ??= file; - } - return firstMatch; -} - -/** `.actor/actor.json`'s own path is not case-folded, unlike the Dockerfile candidates. */ -function findExact(sourceFiles: SourceFile[], normalizedTarget: string): SourceFile | undefined { - return sourceFiles.find((file) => normalizeEntryName(file.name) === normalizedTarget); -} - function escapesActorRootFailure(rawField: string): DockerfileResolution { return { outcome: 'failure', reason: 'escapes-actor-root', - message: `Dockerfile path "${rawField}" in .actor/actor.json points outside the Actor root directory.`, + message: escapesActorRootMessage(rawField, 'Dockerfile'), }; } @@ -71,22 +41,14 @@ export function resolveDockerfileLocation(sourceFiles: SourceFile[]): Dockerfile const indexed = indexSourceFiles(sourceFiles); const logLines: string[] = []; - const actorJsonFile = findExact(sourceFiles, ACTOR_JSON_NAME); - let actorSpecification: unknown; - if (actorJsonFile) { - try { - actorSpecification = JSON5.parse(sourceFileToText(actorJsonFile)); - } catch (error) { - return { - outcome: 'failure', - reason: 'unparseable-actor-json', - message: `Could not parse .actor/actor.json: ${(error as Error).message}`, - }; - } + const actorJson = parseActorJson(sourceFiles); + if (actorJson.outcome === 'unparseable') { + return { outcome: 'failure', reason: 'unparseable-actor-json', message: actorJson.message }; } + const specification = actorJson.outcome === 'parsed' ? actorJson.specification : undefined; - if (actorSpecification !== null && typeof actorSpecification === 'object' && 'dockerfile' in actorSpecification) { - const field: unknown = actorSpecification.dockerfile; + if (specification !== null && typeof specification === 'object' && 'dockerfile' in specification) { + const field: unknown = specification.dockerfile; if (typeof field !== 'string') { return { outcome: 'failure', @@ -95,34 +57,19 @@ export function resolveDockerfileLocation(sourceFiles: SourceFile[]): Dockerfile }; } - if (field === '') { - logLines.push( - 'Warning: "" (from the "dockerfile" field in .actor/actor.json) is not in the pushed source; falling back to the default locations.\n', - ); - } else if (field.startsWith('/')) { - return escapesActorRootFailure(field); - } else { - const joined = normalizeEntryName(path.posix.join(ACTOR_DIR, field)); - if (joined === '..' || joined.startsWith('../')) { - return escapesActorRootFailure(field); - } - - const match = findCaseInsensitive(indexed, joined); - if (match) { - return { - outcome: 'resolved', - dockerfilePath: match.normalizedName, - logLines: [ - ...logLines, - `Using Dockerfile "${match.normalizedName}" (from the "dockerfile" field in .actor/actor.json).\n`, - ], - }; - } - - logLines.push( - `Warning: "${joined}" (from the "dockerfile" field in .actor/actor.json) is not in the pushed source; falling back to the default locations.\n`, - ); + const resolved = resolveActorJsonPathField(indexed, field); + if (resolved.outcome === 'escapes-actor-root') return escapesActorRootFailure(field); + if (resolved.outcome === 'match') { + return { + outcome: 'resolved', + dockerfilePath: resolved.file.normalizedName, + logLines: [ + ...logLines, + `Using Dockerfile "${resolved.file.normalizedName}" (from the "dockerfile" field in .actor/actor.json).\n`, + ], + }; } + logLines.push(fallbackWarningLine(resolved.shownPath, 'dockerfile')); } const actorDirCandidate = normalizeEntryName(`${ACTOR_DIR}/${DEFAULT_DOCKERFILE_NAME}`); diff --git a/src/services/input-schema-location.ts b/src/services/input-schema-location.ts new file mode 100644 index 0000000..e976db2 --- /dev/null +++ b/src/services/input-schema-location.ts @@ -0,0 +1,150 @@ +/** + * Which input schema a build carries, resolved from the version's pushed `sourceFiles` + * (`actor-driver.md`'s "Input schema, validation and defaults"). + * + * The candidate order is `apify-cli`'s own `readInputSchema`, so a developer's local + * `apify validate-schema` and this build agree on which file is the Actor's schema. Case-insensitive + * matching is why the CLI's four candidates collapse to the two here. + */ +import JSON5 from 'json5'; +import ajv2019Package from 'ajv/dist/2019.js'; +import { validateInputSchema } from '@apify/input_schema'; + +import type { InputSchema, SourceFile } from '../storage/entities.js'; +import { + ACTOR_DIR, + escapesActorRootMessage, + fallbackWarningLine, + findCaseInsensitive, + indexSourceFiles, + parseActorJson, + resolveActorJsonPathField, + sourceFileToText, + type IndexedFile, +} from './actor-source-files.js'; + +// AJV ships as CommonJS, so under this package's ESM resolution its class arrives as the module's +// `default`. `2019` is the build with draft-2019-09 support, which the Apify input-schema meta-schema +// requires and the plain build lacks. +const Ajv2019 = ajv2019Package.default; + +/** Checked case-insensitively, so the `INPUT_SCHEMA.json` spelling matches these too. */ +const DEFAULT_SCHEMA_CANDIDATES = [`${ACTOR_DIR}/input_schema.json`, 'input_schema.json'] as const; + +/** Why resolution failed. Each one fails the build: an Actor whose declared input contract cannot be + * read must not be built with that contract silently dropped. */ +export type InputSchemaResolutionFailureReason = + | 'escapes-actor-root' + | 'invalid-input-field' + | 'unparseable-actor-json' + | 'unparseable-input-schema' + | 'invalid-input-schema'; + +/** `none` means the Actor declares no input schema, which is not an error. */ +export type InputSchemaResolution = + | { outcome: 'resolved'; schema: InputSchema; source: string; logLines: string[] } + | { outcome: 'none'; logLines: string[] } + | { outcome: 'failure'; reason: InputSchemaResolutionFailureReason; message: string }; + +/** + * `null` for a valid input schema, the defect otherwise. Runs at build time, so a broken schema is + * reported where the developer is already looking rather than silently at every later run. + */ +export function describeInputSchemaDefect(schema: unknown): string | null { + if (schema === null || typeof schema !== 'object' || Array.isArray(schema)) { + return 'Input schema must be an object.'; + } + try { + // A fresh instance per call: AJV never evicts its internal compiled-schema map, and this runs + // once per build, not per request. + const ajv = new Ajv2019({ strict: false, unicodeRegExp: false }); + // `validateInputSchema` normalizes in place, so it must not get the stored schema itself. + validateInputSchema(ajv, structuredClone(schema) as Record); + return null; + } catch (error) { + return (error as Error).message; + } +} + +/** One place, so an inline `input` object and a schema file are held to the same standard. */ +function acceptSchema(schema: unknown, source: string, logLines: string[]): InputSchemaResolution { + const defect = describeInputSchemaDefect(schema); + if (defect) { + return { + outcome: 'failure', + reason: 'invalid-input-schema', + message: `Input schema from ${source} is not valid: ${defect}`, + }; + } + return { + outcome: 'resolved', + schema: schema as InputSchema, + source, + logLines: [...logLines, `Using the input schema from ${source}.\n`], + }; +} + +function acceptSchemaFile(match: IndexedFile, source: string, logLines: string[]): InputSchemaResolution { + let parsed: unknown; + try { + parsed = JSON5.parse(sourceFileToText(match.file)); + } catch (error) { + return { + outcome: 'failure', + reason: 'unparseable-input-schema', + message: `Could not parse the input schema "${match.normalizedName}": ${(error as Error).message}`, + }; + } + return acceptSchema(parsed, source, logLines); +} + +export function resolveInputSchemaLocation(sourceFiles: SourceFile[]): InputSchemaResolution { + const indexed = indexSourceFiles(sourceFiles); + const logLines: string[] = []; + + const actorJson = parseActorJson(sourceFiles); + if (actorJson.outcome === 'unparseable') { + return { outcome: 'failure', reason: 'unparseable-actor-json', message: actorJson.message }; + } + const specification = actorJson.outcome === 'parsed' ? actorJson.specification : undefined; + + if (specification !== null && typeof specification === 'object' && 'input' in specification) { + const field: unknown = specification.input; + + // An inline schema, the other shape the Actor specification allows for this field. + if (field !== null && typeof field === 'object' && !Array.isArray(field)) { + return acceptSchema(field, 'the "input" field in .actor/actor.json', logLines); + } + + if (typeof field !== 'string') { + return { + outcome: 'failure', + reason: 'invalid-input-field', + message: '.actor/actor.json has invalid format: "input" must be a string or an object.', + }; + } + + const resolved = resolveActorJsonPathField(indexed, field); + if (resolved.outcome === 'escapes-actor-root') { + return { + outcome: 'failure', + reason: 'escapes-actor-root', + message: escapesActorRootMessage(field, 'Input schema'), + }; + } + if (resolved.outcome === 'match') { + const source = `"${resolved.file.normalizedName}" (the "input" field in .actor/actor.json)`; + return acceptSchemaFile(resolved.file, source, logLines); + } + // Falls through instead of failing - the tolerance `apify-cli` shows locally, and the one the + // Dockerfile field already has here. + logLines.push(fallbackWarningLine(resolved.shownPath, 'input')); + } + + for (const candidate of DEFAULT_SCHEMA_CANDIDATES) { + const match = findCaseInsensitive(indexed, candidate); + if (match) return acceptSchemaFile(match, `"${match.normalizedName}"`, logLines); + } + + return { outcome: 'none', logLines }; +} diff --git a/src/services/input-schema.ts b/src/services/input-schema.ts new file mode 100644 index 0000000..46e4c93 --- /dev/null +++ b/src/services/input-schema.ts @@ -0,0 +1,244 @@ +/** + * Input validation and defaults at run start; `services/input-schema-location.ts` is the build-time + * half that finds the schema in the pushed source. + * + * Validation goes through the platform's own `@apify/input_schema`, driven by the same AJV + * configuration the platform's API uses, so a developer reads locally the very message the real API + * would answer with rather than an approximation of it. + */ +import ajvPackage from 'ajv'; +import { validateInputUsingValidator } from '@apify/input_schema'; + +import type { BuildRecord, InputSchema } from '../storage/entities.js'; + +// AJV ships as CommonJS, so under this package's ESM resolution its class arrives as the module's +// `default`. +const Ajv = ajvPackage.default; +type InputValidator = ReturnType['compile']>; + +/** Stored verbatim as the run's `INPUT` record. */ +export interface ActorInput { + body: Buffer; + contentType: string; +} + +/** `kind` is the API error type the route answers with; `message` is the real Apify API's own wording + * for the same defect (`@apify-packages/errors`'s `actor.inputNotJson` and its neighbours). */ +export type InputProcessingResult = + | { kind: 'ok'; input: ActorInput } + | { kind: 'invalid-input'; message: string } + | { kind: 'invalid-input-schema'; message: string }; + +/** + * The input a run against `build` should actually start with: the caller's bytes untouched when the + * build declares no input schema, and the validated input with that schema's defaults applied when it + * does. + */ +export function resolveBuildInput(build: BuildRecord, input: ActorInput | undefined): InputProcessingResult { + if (!build.inputSchema) return { kind: 'ok', input: input as ActorInput }; + return processActorInput(input, build.inputSchema); +} + +/** + * Mirrors the platform's `processInputUsingSchema`: with a schema present, an absent input is an empty + * object the defaults are applied to, not "no input". + */ +export function processActorInput(input: ActorInput | undefined, schema: InputSchema): InputProcessingResult { + let parsedInput: Record = {}; + if (input) { + if (!isJsonContentType(input.contentType)) { + return { kind: 'invalid-input', message: 'Actor input must have content type "application/json".' }; + } + let body: unknown; + try { + body = JSON.parse(input.body.toString('utf8')); + } catch (error) { + return { kind: 'invalid-input', message: `Cannot parse input JSON body: ${(error as Error).message}` }; + } + // A literal `null` body is the platform's "no input" too, not a type error. + if (body !== null) { + if (!isPlainObject(body)) { + const actualType = Array.isArray(body) ? 'array' : typeof body; + return { + kind: 'invalid-input', + message: `The input JSON must be object, got "${actualType}" instead.`, + }; + } + parsedInput = body; + } + } + + let validator; + try { + validator = compileInputSchemaValidator(schema); + } catch (error) { + // Reachable only for a schema recorded before builds validated them, or one that meta-validates + // yet still will not compile. + return { kind: 'invalid-input-schema', message: `Input schema is not valid: ${(error as Error).message}` }; + } + + // `parsedInput` was parsed here and is referenced nowhere else, so the merge may own it. + const withDefaults = assignDefaults(parsedInput, schema); + + // No `proxy` options: this runtime emulates no proxy groups (`unsupported.md`), so any + // `apifyProxyGroups` selection is accepted, while the rest of a proxy field is still checked. + const validationErrors = validateInputUsingValidator(validator, schema, withDefaults, {}); + if (validationErrors.length > 0) { + // Joined, as the platform answers an API-origin run. + const message = validationErrors.map(({ message: text }) => text).join(', '); + return { kind: 'invalid-input', message: `Input is not valid: ${message}` }; + } + + return { + kind: 'ok', + input: { body: Buffer.from(JSON.stringify(withDefaults), 'utf8'), contentType: DEFAULT_INPUT_CONTENT_TYPE }, + }; +} + +const DEFAULT_INPUT_CONTENT_TYPE = 'application/json'; + +/** Parameters such as `; charset=utf-8` are ignored. */ +function isJsonContentType(contentType: string): boolean { + return contentType.split(';')[0]?.trim().toLowerCase() === DEFAULT_INPUT_CONTENT_TYPE; +} + +function isPlainObject(value: unknown): value is Record { + return typeof value === 'object' && value !== null && !Array.isArray(value); +} + +/** + * A port of the platform's `mergeDefaultsFromInputSchema`. `values` is filled in place and returned: + * every caller parsed it itself, and cloning it doubles the peak heap of a large input. + */ +export function assignDefaults(values: Record, schema: InputSchema): Record { + const properties = isPlainObject(schema.properties) ? schema.properties : {}; + const defaults: Record = {}; + for (const [key, fieldSchema] of Object.entries(properties)) { + defaults[key] = extractDefaults(fieldSchema); + } + + // `extractDefaults` returns references into the stored schema, so the values assigned into the + // input must be copies - otherwise a later mutation of either would reach the other. + assignRecursively(values, structuredClone(defaults), schema); + return values; +} + +/** A field's own `default` wins, key by key, over the defaults of its nested fields. */ +function extractDefaults(fieldSchema: unknown): unknown { + if (!isPlainObject(fieldSchema)) return undefined; + + const rootValue = fieldSchema.default; + const properties = isPlainObject(fieldSchema.properties) ? fieldSchema.properties : undefined; + if (properties) { + const nested: Record = {}; + for (const [key, subSchema] of Object.entries(properties)) { + nested[key] = extractDefaults(subSchema); + } + if (isPlainObject(rootValue)) return deepMerge(nested, rootValue); + if (Object.values(nested).some((value) => value !== undefined)) return nested; + } + return rootValue; +} + +/** `fillDefinedValues` is set for array items alone, where the platform fills an item's fields even + * though the item itself is present. */ +function assignRecursively( + target: Record, + defaults: Record, + schema: InputSchema | undefined, + fillDefinedValues = false, +): void { + const schemaProperties = isPlainObject(schema?.properties) ? schema.properties : {}; + const keys = new Set([...Object.keys(defaults), ...Object.keys(schemaProperties)]); + + for (const key of keys) { + const defaultValue = defaults[key]; + const currentValue = target[key]; + const fieldSchema = schemaProperties[key]; + + if (currentValue === undefined) { + if (defaultValue !== undefined) target[key] = defaultValue; + continue; + } + + if (isPlainObject(currentValue)) { + // A present nested object keeps exactly the keys it came with, so outside an array item + // there is nothing to fill and no defaults to extract. + const nestedDefaults = fillDefinedValues + ? isPlainObject(defaultValue) + ? defaultValue + : ((extractDefaults(fieldSchema) as Record | undefined) ?? {}) + : {}; + assignRecursively(currentValue, nestedDefaults, fieldSchema as InputSchema | undefined); + continue; + } + + if (Array.isArray(currentValue)) { + const itemSchema = isPlainObject(fieldSchema) ? fieldSchema.items : undefined; + const itemDefaults = extractDefaults(itemSchema); + if (!isPlainObject(itemDefaults)) continue; + for (const item of currentValue) { + if (!isPlainObject(item)) continue; + assignRecursively(item, itemDefaults, itemSchema as InputSchema | undefined, true); + } + } + } +} + +/** The one `lodash.merge` behaviour `extractDefaults` needs, without the dependency. */ +function deepMerge(base: Record, overrides: Record): Record { + const merged: Record = { ...base }; + for (const [key, value] of Object.entries(overrides)) { + const existing = merged[key]; + merged[key] = isPlainObject(existing) && isPlainObject(value) ? deepMerge(existing, value) : value; + } + return merged; +} + +/** + * Compiled validators, keyed by the schema's serialization: a run start would otherwise spend ~6ms + * recompiling a schema it has already seen. Bounded and evicted first-inserted-first, so a long-lived + * runtime does not keep an entry per schema it has ever seen. + */ +const validatorCache = new Map(); +const VALIDATOR_CACHE_MAX_ENTRIES = 100; + +function compileInputSchemaValidator(schema: InputSchema): InputValidator { + const cacheKey = JSON.stringify(schema); + const cached = validatorCache.get(cacheKey); + if (cached) return cached; + + const validator = new Ajv({ strict: false, unicodeRegExp: false }).compile(prepareSchemaForValidation(schema)); + + if (validatorCache.size >= VALIDATOR_CACHE_MAX_ENTRIES) { + const oldest = validatorCache.keys().next(); + if (!oldest.done) validatorCache.delete(oldest.value); + } + validatorCache.set(cacheKey, validator); + return validator; +} + +/** + * Ported from the platform's `getAjvValidator`. `$schema` is dropped because AJV would otherwise try + * to fetch the Apify meta-schema it names and fail to compile the schema at all. + */ +function prepareSchemaForValidation(schema: InputSchema): Record { + const copy = structuredClone(schema) as Record; + const required: string[] = []; + const originalRequired = Array.isArray(schema.required) ? schema.required : []; + const properties = isPlainObject(copy.properties) ? copy.properties : {}; + + for (const [key, fieldSchema] of Object.entries(properties)) { + if (!originalRequired.includes(key)) continue; + if (isPlainObject(fieldSchema) && fieldSchema.default !== undefined) continue; + required.push(key); + if (isPlainObject(fieldSchema) && fieldSchema.type === 'array') { + const minItems = typeof fieldSchema.minItems === 'number' ? fieldSchema.minItems : 0; + fieldSchema.minItems = Math.max(1, minItems); + } + } + + copy.required = required; + delete copy.$schema; + return copy; +} diff --git a/src/services/pricing.ts b/src/services/pricing.ts index c032aeb..1339930 100644 --- a/src/services/pricing.ts +++ b/src/services/pricing.ts @@ -262,8 +262,8 @@ export function validatePricingInfosUpdate( for (const [index, entry] of current.entries()) { if (!isDeepStrictEqual(submitted[index], entry)) { return invalid( - `pricingInfos[${index}] differs from the Actor's existing pricing info - an update must start ` + - 'with the existing entries unchanged and may only append one more.', + `pricingInfos[${index}] differs from the Actor's existing pricing info ${describeDifference(submitted[index], entry)} - ` + + 'an update must start with the existing entries unchanged and may only append one more.', 'incorrect-pricing-modifier-prefix', ); } @@ -287,6 +287,29 @@ export function validatePricingInfosUpdate( return validated; } +/** + * The first field the submitted entry gets wrong, named as a path. Worth the code: the usual cause is an + * entry rebuilt from a file rather than from what the Actor already has, and the difference is then a + * timestamp nobody thinks to look at. + */ +function describeDifference(submitted: unknown, stored: unknown, path = ''): string { + if (isPlainObject(submitted) && isPlainObject(stored)) { + for (const key of new Set([...Object.keys(stored), ...Object.keys(submitted)])) { + const where = path === '' ? key : `${path}.${key}`; + if (!isDeepStrictEqual(submitted[key], stored[key])) + return describeDifference(submitted[key], stored[key], where); + } + } + const at = path === '' ? '' : `at "${path}" `; + return `${at}(sent ${describeValue(submitted)}, stored ${describeValue(stored)})`; +} + +function describeValue(value: unknown): string { + if (value === undefined) return 'nothing'; + const json = JSON.stringify(value) ?? String(value); + return json.length > 60 ? `${json.slice(0, 57)}...` : json; +} + /** The entry with the latest `startedAt` not after `date`, the platform's own rule. */ export function effectivePricingInfo( pricingInfos: readonly ActorPricingInfoRecord[] | undefined, diff --git a/src/services/runs.ts b/src/services/runs.ts index 0f0a06a..4236252 100644 --- a/src/services/runs.ts +++ b/src/services/runs.ts @@ -18,12 +18,16 @@ import { type DebugPlan, } from './debug-mode.js'; import { browserViewLogLine, describeBrowserViewerStartFailure } from './browser-view.js'; -import { dedicatedCpusFor } from '../resources.js'; +import { dedicatedCpusFor, platformIncompatibleMemoryWarning } from '../resources.js'; import { CONTAINER_EVENTS_WS_BASE_URL } from '../config.js'; -import { formatRuntimeLogLines, type RuntimeLogLine } from '../runtime-log.js'; +import { formatRuntimeLogLines } from '../runtime-log.js'; import { getRunTelemetry } from './events-channel.js'; import { initialChargedEventCounts, resolveRunPricingInfo } from './pricing.js'; -import { registerDefaultDatasetForCharging, unregisterDefaultDatasetForCharging } from './charging.js'; +import { + actorStartChargeMessage, + registerDefaultDatasetForCharging, + unregisterDefaultDatasetForCharging, +} from './charging.js'; const DEFAULT_MEMORY_MBYTES = 1024; const DEFAULT_TIMEOUT_SECS = 300; @@ -227,6 +231,12 @@ export async function startRun( await runs.set(record.id, record); registerDefaultDatasetForCharging(record); + // Both lines are about what the caller asked for, so they are written before the run does anything. + const memoryWarning = platformIncompatibleMemoryWarning(memoryMbytes); + if (memoryWarning) appendRuntimeLog(record.id, memoryWarning); + const startCharge = actorStartChargeMessage(record); + if (startCharge) appendRuntimeLog(record.id, startCharge); + void runInBackground(driver, actor, record, options).catch(async (error: unknown) => { // Every *expected* failure mode inside `runInBackground` is already caught internally and mapped // to a terminal status - this is only reached by a genuinely unexpected exception (e.g. a @@ -357,7 +367,7 @@ export async function runInBackground( appendRuntimeLog(record.id, unknownWorkingDirectoryLine(actor.localDevFolder)); } const runtimeSection = devMount ? liveDevFolderWarningLines(devMount) : []; - if (runtimeSection.length > 0) appendLog(record.id, renderRuntimeLogSection(runtimeSection)); + if (runtimeSection.length > 0) appendLog(record.id, formatRuntimeLogLines(runtimeSection)); // The sidecar comes up before the Actor's container. Started before the pre-start abort re-check below, // so an abort landing during this (possibly slow) step is still caught by it. @@ -438,6 +448,9 @@ export async function runInBackground( // observes it turn terminal, and immediately does a non-stream `GET /v2/logs/:id` can never observe // the persisted log lagging behind the status it just saw. await flushLog(record.id); + // Before the status write for the same reason as the flush: the sampled figures live only in + // memory until this runs, and a client that sees the run turn terminal reads the record next. + await persistRunTelemetry(record.id); // Guarded: `container.wait()` resolving is not proof the run wasn't aborted - `container.stop()` // (from an in-flight `abortRun`) and the container exiting on its own race off the same // underlying Docker event with no ordering guarantee. If `abortRun` already moved the record to @@ -459,6 +472,7 @@ export async function runInBackground( // unusable mount) is what `apify call` streams, and the status message alone leaves it empty. appendRuntimeLog(record.id, `Cannot start run: ${statusMessage}`); await flushLog(record.id); + await persistRunTelemetry(record.id); await transitionJobStatus(runs, record.id, 'FAILED', { finishedAt: new Date().toISOString(), statusMessage, @@ -467,11 +481,12 @@ export async function runInBackground( // The container is gone, so an open graceful-abort window has nothing left to wait out - the path an // Actor that honours the `aborting` frame takes. The log is already flushed by both the success // path above and the catch below, so a client seeing this terminal status can still read all of it. + // The accumulators are in-memory, so a finished run keeps its figures only if they are written + // here - before the transition below, and again for the paths above that end the run elsewhere. + await persistRunTelemetry(record.id); if (cancelGracefulAbort(record.id)) { await transitionJobStatus(runs, record.id, 'ABORTED', { finishedAt: new Date().toISOString() }); } - // The accumulators are in-memory, so a finished run keeps its figures only if they are written here. - await persistRunTelemetry(record.id); unregisterDefaultDatasetForCharging(record); if (browserViewer) await driver.stopBrowserViewer(record.id); // A run that ends for real must not leave an armed migration-stop timer behind. @@ -626,16 +641,3 @@ export async function reconcileOrphanedJobs(driver: Driver): Promise { ), ); } - -/** 80 columns, not a terminal's full width: every line already carries a timestamp and the runtime - * marker, so a wider rule only forces wrapping. */ -function renderRuntimeLogSection(lines: readonly RuntimeLogLine[]): string { - const title = ' Local Actor runtime '; - const width = 80; - const head = `${'='.repeat(4)}${title}${'='.repeat(width - 4 - title.length)}`; - return formatRuntimeLogLines([ - { text: head, emphasis: true }, - ...lines, - { text: '='.repeat(width), emphasis: true }, - ]); -} diff --git a/src/storage/entities.ts b/src/storage/entities.ts index 4602904..538fa0f 100644 --- a/src/storage/entities.ts +++ b/src/storage/entities.ts @@ -42,6 +42,17 @@ export interface SourceFile { content: string; } +/** + * An Actor's input schema, exactly as it was pushed. A loose record rather than a modelled type: it is + * the developer's own document, handed to the platform's validator verbatim, and the runtime itself + * reads only `properties`/`required` out of it. + */ +export interface InputSchema { + [key: string]: unknown; + properties?: Record; + required?: string[]; +} + export interface ActorVersionRecord { versionNumber: string; buildTag: string; @@ -167,6 +178,11 @@ export interface BuildRecord { * inspect failed or the working directory was empty/`/` (mounting over `/` would destroy the * container) - never present on a non-`SUCCEEDED` build. */ imageWorkingDirectory?: string; + /** The input schema this build's own source files declared, written with `SUCCEEDED` like + * `imageWorkingDirectory` above, and build-specific for the same reason: a run validates against the + * schema of the build it resolved, never another tag's more recently pushed one. Absent when the + * Actor declares none - such a run takes its input exactly as the caller sent it. */ + inputSchema?: InputSchema; exitCode?: number; statusMessage?: string; } diff --git a/test/e2e/actor-dev-loop.test.ts b/test/e2e/actor-dev-loop.test.ts index 2675e09..2df095e 100644 --- a/test/e2e/actor-dev-loop.test.ts +++ b/test/e2e/actor-dev-loop.test.ts @@ -21,6 +21,7 @@ import { apify, apifyAllOutput, apifyEnv, + apifyExpectingFailure, createIsolatedApifyHome, loginApifyCli, removeIsolatedApifyHome, @@ -133,6 +134,41 @@ describe('full Actor dev loop via apify-cli (requires Docker)', () => { expect(log).toMatch(/Processing/); }); + it("a call with no input at all runs on the input schema's defaults, and one the schema rejects never starts", () => { + const env = apifyEnv(isolatedApifyHome); + const actorDir = join(REPO_ROOT, 'sample_actor_ts'); + + const callOutput = apify(['call', '--json'], { cwd: actorDir, env }); + const call = JSON.parse(callOutput) as CallResult; + expect(call.run.status).toBe('SUCCEEDED'); + + // The run's own INPUT record, which only the runtime ever writes: every field of the schema, at + // its default. This is what distinguishes the schema's defaults from the Actor's own internal + // fallback for a missing field - the item count alone cannot tell the two apart. + const storedInput = apify(['key-value-stores', 'get-value', call.storage.defaultKeyValueStoreId, 'INPUT'], { + cwd: actorDir, + env, + }); + expect(JSON.parse(storedInput) as Record).toEqual({ + startUrl: 'https://crawlee.dev/', + maxPages: 2, + }); + + const infoOutput = apify(['datasets', 'info', call.storage.defaultDatasetId, '--json'], { + cwd: actorDir, + env, + }); + expect((JSON.parse(infoOutput) as DatasetInfoResult).itemCount).toBe(2); + + // `maxPages` has `minimum: 1` - the run is refused before any container starts, and the CLI + // surfaces the runtime's validation message. + const rejected = apifyExpectingFailure(['call', '--input', JSON.stringify({ maxPages: 0 }), '--json'], { + cwd: actorDir, + env, + }); + expect(rejected).toMatch(/maxPages/); + }); + it('apify api reads back the run and its default dataset (requirements/cli.md: `apify api`)', () => { const env = apifyEnv(isolatedApifyHome); const callOutput = apify(['call', '--input', JSON.stringify({ maxPages: 1 }), '--json'], { diff --git a/test/e2e/dev-folder-bind-mount.test.ts b/test/e2e/dev-folder-bind-mount.test.ts index 80f79a6..cd9f38a 100644 --- a/test/e2e/dev-folder-bind-mount.test.ts +++ b/test/e2e/dev-folder-bind-mount.test.ts @@ -224,7 +224,7 @@ describe('local dev-folder bind mount: edit-compile-call loop with no rebuild (r expect(optedOutLog).toContain(`Skipping the registered local dev folder ${actorDir}`); expect(optedOutLog).toContain(ORIGINAL_MARKER); expect(optedOutLog).not.toContain(EDITED_MARKER); - expect(optedOutLog).not.toContain('Local Actor runtime'); + expect(optedOutLog).not.toContain('Live dev folder'); // The registration survived. const call = JSON.parse( @@ -233,7 +233,6 @@ describe('local dev-folder bind mount: edit-compile-call loop with no rebuild (r expect(call.run.status).toBe('SUCCEEDED'); const callLog = apifyAllOutput(['runs', 'log', call.run.id], { cwd: REPO_ROOT, env }); expect(callLog).toContain(EDITED_MARKER); - expect(callLog).toContain('Local Actor runtime'); expect(callLog).toContain(`Live dev folder: ${actorDir}`); expect(callLog).toContain('Live dev folder mode'); expect(callLog).toContain('apify call --no-dev-folder'); @@ -342,7 +341,7 @@ describe('local dev-folder bind mount: edit-compile-call loop with no rebuild (r expect(call.run.status).toBe('SUCCEEDED'); const log = apifyAllOutput(['runs', 'log', call.run.id], { cwd: REPO_ROOT, env }); - expect(log).not.toContain('Local Actor runtime'); + expect(log).not.toContain('Live dev folder'); }, 5 * 60 * 1000, ); diff --git a/test/e2e/helpers/apify-cli.ts b/test/e2e/helpers/apify-cli.ts index 96e5109..ff2bbae 100644 --- a/test/e2e/helpers/apify-cli.ts +++ b/test/e2e/helpers/apify-cli.ts @@ -15,23 +15,41 @@ export function apify(args: string[], options: { cwd: string; env: NodeJS.Proces }); } +interface ApifyOptions { + cwd: string; + env: NodeJS.ProcessEnv; +} + +function spawnApify(args: string[], options: ApifyOptions): ReturnType> { + return spawnSync('npx', ['-y', '-p', 'apify-cli', 'apify', ...args], { + cwd: options.cwd, + env: options.env, + encoding: 'utf8', + }); +} + /** * Like `apify()`, but returns stdout AND stderr combined. The CLI writes human-readable output — * including the log content of `apify runs log` (see `outputJobLog`'s `process.stderr.write`) — to * stderr, keeping stdout for machine-readable payloads, so log-content assertions must read stderr. */ -export function apifyAllOutput(args: string[], options: { cwd: string; env: NodeJS.ProcessEnv }): string { - const result = spawnSync('npx', ['-y', '-p', 'apify-cli', 'apify', ...args], { - cwd: options.cwd, - env: options.env, - encoding: 'utf8', - }); +export function apifyAllOutput(args: string[], options: ApifyOptions): string { + const result = spawnApify(args, options); if (result.status !== 0) { throw new Error(`apify ${args.join(' ')} exited with ${result.status}:\n${result.stderr}`); } return `${result.stdout}\n${result.stderr}`; } +/** Like `apifyAllOutput()`, but requires a non-zero exit. */ +export function apifyExpectingFailure(args: string[], options: ApifyOptions): string { + const result = spawnApify(args, options); + if (result.status === 0) { + throw new Error(`apify ${args.join(' ')} was expected to fail, but succeeded:\n${result.stdout}`); + } + return `${result.stdout}\n${result.stderr}`; +} + /** Any non-empty value works - the runtime creates a user for this token ad-hoc on its first API * request (`cli.md`'s User bootstrap), the same token throughout this e2e run mapping back to that one * user on every subsequent request. */ diff --git a/test/e2e/pay-per-event.test.ts b/test/e2e/pay-per-event.test.ts index 6f02f39..5bdc3fa 100644 --- a/test/e2e/pay-per-event.test.ts +++ b/test/e2e/pay-per-event.test.ts @@ -131,7 +131,11 @@ describe('pay-per-event pricing and the run cost estimate via apify-cli (require expect(freeRun.pricingInfo).toBeUndefined(); expect(freeRun.chargedEventCounts).toBeUndefined(); expect(freeRun.stats.computeUnits).toBeGreaterThan(0); - expect(freeRun.usageUsd.ACTOR_COMPUTE_UNITS).toBeCloseTo(freeRun.usage.ACTOR_COMPUTE_UNITS * 0.2, 6); + // Exactly the runtime's own rounding, not a tolerance: a product whose 7th decimal is a 5 is + // moved by the full tolerance of `toBeCloseTo(..., 6)`, which then fails on the boundary. + expect(freeRun.usageUsd.ACTOR_COMPUTE_UNITS).toBe( + Number((freeRun.usage.ACTOR_COMPUTE_UNITS * 0.2).toFixed(6)), + ); expect(freeRun.usageTotalUsd).toBe(freeRun.usageUsd.ACTOR_COMPUTE_UNITS); // The container was sampled while it ran. expect(freeRun.stats.memAvgBytes).toBeGreaterThan(0); diff --git a/test/integration/dev-folder.test.ts b/test/integration/dev-folder.test.ts index 658cff9..e128a30 100644 --- a/test/integration/dev-folder.test.ts +++ b/test/integration/dev-folder.test.ts @@ -834,11 +834,10 @@ describe('run-start devMount derivation (actor fields -> RunContext.devMount, se imageWorkingDirectory: '/usr/src/app', }); const log = await server.client.log(run.id).get(); - expect(log).toContain('Local Actor runtime'); expect(log).toContain('Live dev folder: /abs/dev/src'); expect(log).toContain('Live dev folder mode'); expect(log).toContain('apify call --no-dev-folder'); - expect(log!.indexOf('Local Actor runtime')).toBeLessThan(log!.indexOf('done')); + expect(log!.indexOf('Live dev folder:')).toBeLessThan(log!.indexOf('done')); }); it("marks every runtime-authored line of a run's log with the runtime prefix and its blue, and leaves the Actor's own output untouched", async () => { @@ -877,7 +876,7 @@ describe('run-start devMount derivation (actor fields -> RunContext.devMount, se const run = await server.client.actor(actor.id).start({}, { waitForFinish: 5 }); expect(run.status).toBe('SUCCEEDED'); expect(capturing.getCapturedDevMount()).toBeUndefined(); - expect(await server.client.log(run.id).get()).not.toContain('Local Actor runtime'); + expect(await server.client.log(run.id).get()).not.toContain('Live dev folder'); }); it('an Actor whose registration was set and then cleared also gets devMount: undefined, not the stale pair', async () => { @@ -988,7 +987,7 @@ describe('per-run opt-out: POST /v2/actors/:actorId/runs?devFolder=false (servic expect(capturing.getCapturedDevMount()).toBeUndefined(); const log = await server.client.log(run.id).get(); expect(log).toContain('Skipping the registered local dev folder /abs/dev/src for this run'); - expect(log).not.toContain('Local Actor runtime'); + expect(log).not.toContain('Live dev folder'); }); it('devFolder=false leaves the registration itself untouched - the next run without the opt-out mounts again', async () => { diff --git a/test/integration/input-schema.test.ts b/test/integration/input-schema.test.ts new file mode 100644 index 0000000..fd0509c --- /dev/null +++ b/test/integration/input-schema.test.ts @@ -0,0 +1,246 @@ +/** + * Input validation and defaults at run start, over the real HTTP server and a real `apify-client` + * (`actor-driver.md`'s "Input schema, validation and defaults"). Builds are seeded directly, the way + * the rest of the run-start integration tests do, so no Docker daemon is needed: what matters here is + * what the API answers and what lands in the run's `INPUT` record, neither of which involves the driver. + */ +import { afterEach, beforeEach, describe, expect, it } from 'vitest'; + +import { fixedBuildOutcomeDriver, startTestServer, type TestServerHandle } from './helpers/test-server.js'; +import { getRegistries } from '../../src/storage/registries.js'; +import { generateId } from '../../src/storage/ids.js'; +import { recordTaggedBuild, updateActor } from '../../src/services/actors.js'; +import { runBuildInBackground } from '../../src/services/builds.js'; +import type { InputSchema, SourceFile } from '../../src/storage/entities.js'; + +const SAMPLE_SCHEMA: InputSchema = { + title: 'Sample actor input', + type: 'object', + schemaVersion: 1, + properties: { + startUrl: { + title: 'Start URL', + type: 'string', + editor: 'textfield', + description: 'The page the crawler starts from.', + default: 'https://crawlee.dev/', + }, + maxPages: { + title: 'Max pages', + type: 'integer', + description: 'The maximum number of pages to crawl.', + default: 2, + minimum: 1, + }, + }, + required: [], +}; + +describe('input validation and defaults (via real apify-client)', () => { + let server: TestServerHandle; + + beforeEach(async () => { + server = await startTestServer(); + }); + + afterEach(async () => { + await server.close(); + }); + + /** Seeds a tagged, successful build - with or without an input schema - for `actorId`. */ + async function seedBuild( + actorId: string, + userId: string, + inputSchema?: InputSchema, + tag = 'latest', + ): Promise { + const { builds } = getRegistries(); + const buildId = generateId(); + const buildNumber = '0.0.1'; + await builds.set(buildId, { + id: buildId, + userId, + actorId, + versionNumber: '0.0', + buildNumber, + tag, + status: 'SUCCEEDED', + startedAt: new Date().toISOString(), + finishedAt: new Date().toISOString(), + imageId: `fake-image:${tag}`, + ...(inputSchema ? { inputSchema } : {}), + }); + await updateActor(actorId, (current) => recordTaggedBuild(current, tag, buildId, buildNumber)); + } + + async function storedInput(runId: string): Promise { + const run = await server.client.run(runId).get(); + const record = await server.client.keyValueStore(run!.defaultKeyValueStoreId).getRecord('INPUT'); + return record?.value; + } + + it("fills the schema's defaults into the run's INPUT, keeping what the caller sent", async () => { + const actor = await server.client.actors().create({ name: 'defaults-actor' }); + await seedBuild(actor.id, actor.userId, SAMPLE_SCHEMA); + + const run = await server.client.actor(actor.id).start({ maxPages: 7 }); + expect(await storedInput(run.id)).toEqual({ maxPages: 7, startUrl: 'https://crawlee.dev/' }); + }); + + it('writes the defaults even for a run started with no input at all', async () => { + const actor = await server.client.actors().create({ name: 'no-input-actor' }); + await seedBuild(actor.id, actor.userId, SAMPLE_SCHEMA); + + // No body at all on the wire - the raw endpoint, since `apify-client` always sends the object + // it is given. + const response = await fetch(`${server.baseUrl}/v2/actors/${actor.id}/runs?token=${server.token}`, { + method: 'POST', + }); + expect(response.status).toBe(201); + const { data } = (await response.json()) as { data: { id: string } }; + expect(await storedInput(data.id)).toEqual({ startUrl: 'https://crawlee.dev/', maxPages: 2 }); + }); + + it("rejects an input the schema does not accept, with the platform's error type and message", async () => { + const actor = await server.client.actors().create({ name: 'invalid-input-actor' }); + await seedBuild(actor.id, actor.userId, SAMPLE_SCHEMA); + + await expect(server.client.actor(actor.id).start({ maxPages: 0 })).rejects.toMatchObject({ + statusCode: 400, + type: 'invalid-input', + message: 'Input is not valid: Field input.maxPages must be >= 1', + }); + + // Nothing was started: a rejected input never creates a run. + const runs = await server.client.actor(actor.id).runs().list(); + expect(runs.items).toHaveLength(0); + }); + + it('rejects a non-object and a non-JSON body against a schema', async () => { + const actor = await server.client.actors().create({ name: 'malformed-input-actor' }); + await seedBuild(actor.id, actor.userId, SAMPLE_SCHEMA); + const url = `${server.baseUrl}/v2/actors/${actor.id}/runs?token=${server.token}`; + + const array = await fetch(url, { + method: 'POST', + headers: { 'content-type': 'application/json' }, + body: '[1,2]', + }); + expect(array.status).toBe(400); + expect(await array.json()).toEqual({ + error: { type: 'invalid-input', message: 'The input JSON must be object, got "array" instead.' }, + }); + + const text = await fetch(url, { + method: 'POST', + headers: { 'content-type': 'text/plain' }, + body: '{"maxPages":1}', + }); + expect(text.status).toBe(400); + expect(await text.json()).toEqual({ + error: { type: 'invalid-input', message: 'Actor input must have content type "application/json".' }, + }); + + const broken = await fetch(url, { + method: 'POST', + headers: { 'content-type': 'application/json' }, + body: '{oops', + }); + expect(broken.status).toBe(400); + expect(((await broken.json()) as { error: { message: string } }).error.message).toContain( + 'Cannot parse input JSON body:', + ); + }); + + it('leaves the input untouched for a build that carries no input schema', async () => { + const actor = await server.client.actors().create({ name: 'schemaless-actor' }); + await seedBuild(actor.id, actor.userId); + + // Not valid against SAMPLE_SCHEMA at all - and accepted, byte for byte, because this build has + // no schema to validate against. + const run = await server.client.actor(actor.id).start({ maxPages: 0, whatever: 'anything' }); + expect(await storedInput(run.id)).toEqual({ maxPages: 0, whatever: 'anything' }); + }); + + it('validates against the schema of the build the run actually resolved, not the newest one', async () => { + const actor = await server.client.actors().create({ name: 'two-tag-actor' }); + + // `latest` demands maxPages >= 1; `beta` has no schema at all. + await seedBuild(actor.id, actor.userId, SAMPLE_SCHEMA); + await seedBuild(actor.id, actor.userId, undefined, 'beta'); + + await expect(server.client.actor(actor.id).start({ maxPages: 0 })).rejects.toMatchObject({ + statusCode: 400, + type: 'invalid-input', + }); + + const betaRun = await server.client.actor(actor.id).start({ maxPages: 0 }, { build: 'beta' }); + expect(await storedInput(betaRun.id)).toEqual({ maxPages: 0 }); + }); + + it('records the schema found in the pushed source on the build, and fails the build on a broken one', async () => { + const actor = await server.client.actors().create({ name: 'building-actor' }); + const { actors, builds } = getRegistries(); + + /** Runs a real build (driver stubbed out at the `docker build` itself) over `sourceFiles`, so the + * schema really is resolved from the pushed source the way a pushed Actor's would be. */ + const build = async (buildId: string, sourceFiles: SourceFile[]): Promise => { + await builds.set(buildId, { + id: buildId, + userId: actor.userId, + actorId: actor.id, + versionNumber: '0.0', + buildNumber: '0.0.1', + tag: 'latest', + status: 'READY', + startedAt: new Date().toISOString(), + }); + await runBuildInBackground( + fixedBuildOutcomeDriver({ imageId: 'built-image:latest' }), + (await actors.get(actor.id))!, + { versionNumber: '0.0', buildTag: 'latest', sourceType: 'SOURCE_FILES', sourceFiles }, + (await builds.get(buildId))!, + { tag: 'latest', useCache: true }, + ); + }; + + const goodBuildId = 'goodBuildId12345g'; + await build(goodBuildId, [ + { name: 'main.js', format: 'TEXT', content: 'console.log(1)' }, + { name: '.actor/input_schema.json', format: 'TEXT', content: JSON.stringify(SAMPLE_SCHEMA) }, + ]); + + expect((await builds.get(goodBuildId))?.status).toBe('SUCCEEDED'); + expect((await builds.get(goodBuildId))?.inputSchema).toEqual(SAMPLE_SCHEMA); + expect(await server.client.log(goodBuildId).get()).toContain('Using the input schema from'); + + // End to end from that build: a run against it is validated, and gets the defaults. + await expect(server.client.actor(actor.id).start({ maxPages: 0 })).rejects.toMatchObject({ + statusCode: 400, + type: 'invalid-input', + }); + const run = await server.client.actor(actor.id).start({ maxPages: 5 }); + expect(await storedInput(run.id)).toEqual({ maxPages: 5, startUrl: 'https://crawlee.dev/' }); + + // A build whose schema is invalid fails, with the reason in both its status message and its log - + // rather than producing an image whose every later run would silently skip validation. + const brokenBuildId = 'brokenBuildId123b'; + await build(brokenBuildId, [ + { + name: '.actor/input_schema.json', + format: 'TEXT', + content: JSON.stringify({ + title: 'Missing a description', + type: 'object', + schemaVersion: 1, + properties: { maxPages: { title: 'Max pages', type: 'integer' } }, + }), + }, + ]); + const broken = await server.client.build(brokenBuildId).get(); + expect(broken?.status).toBe('FAILED'); + expect(broken?.statusMessage).toContain('is not valid'); + expect(await server.client.log(brokenBuildId).get()).toContain('.actor/input_schema.json'); + expect((await builds.get(brokenBuildId))?.inputSchema).toBeUndefined(); + }); +}); diff --git a/test/integration/pay-per-event.test.ts b/test/integration/pay-per-event.test.ts index d75135f..c0e6257 100644 --- a/test/integration/pay-per-event.test.ts +++ b/test/integration/pay-per-event.test.ts @@ -21,6 +21,7 @@ import { getRegistries } from '../../src/storage/registries.js'; import { generateId } from '../../src/storage/ids.js'; import { recordTaggedBuild, updateActor } from '../../src/services/actors.js'; import { subscribeEvents } from '../../src/services/events-channel.js'; +import { isTerminalJobStatus } from '../../src/services/job-status.js'; import type { ActorRecord, BuildRecord } from '../../src/storage/entities.js'; const PAGE_EVENT = 'page-scraped'; @@ -219,6 +220,11 @@ describe('pay-per-event: runs and charging', () => { 0.01, ); + // A grant the platform accepts draws no warning, only the pre-charge line. + const log = await server.client.log(runId).get(); + expect(log).not.toContain('Warning: the requested memory'); + expect(log).toContain("Pre-charged 2 'apify-actor-start' event(s), $0.01 in total, for 2048 MB of memory"); + const env = driver.startCalls[0]!.ctx.env; expect(env.ACTOR_MAX_TOTAL_CHARGE_USD).toBe('1.5'); // The SDKs must fetch the run object for pricing, never read stale env copies (`services/runs.ts`). @@ -228,6 +234,22 @@ describe('pay-per-event: runs and charging', () => { await finishRun(server, driver, runId); }); + it("warns about memory the platform would refuse, and says what the run's start pre-charge covers", async () => { + const actorId = await seedRunnableActor(server, 'odd-memory-actor', PPE_PRICING); + // 8096 MB is between two steps: the platform refuses it outright, this runtime runs it and + // pre-charges for the 7 whole gigabytes it covers. + const runId = await startLiveRun(server, driver, actorId, { memory: 8096 }); + + const log = await server.client.log(runId).get(); + expect(log).toContain('8096 MB is not a power of two'); + expect(log).toContain("Pre-charged 7 'apify-actor-start' event(s), $0.035 in total, for 8096 MB of memory"); + + const run = (await server.client.run(runId).get()) as unknown as Record; + expect((run.chargedEventCounts as Record)['apify-actor-start']).toBe(7); + + await finishRun(server, driver, runId); + }); + it('a run of an unpriced Actor has no pricingInfo, chargedEventCounts, or maxTotalChargeUsd, and no cap env var', async () => { const actorId = await seedRunnableActor(server, 'free-actor'); const runId = await startLiveRun(server, driver, actorId); @@ -489,7 +511,8 @@ describe('run usage estimate', () => { const usage = run.usage as Record; const usageUsd = run.usageUsd as Record; expect(usage.ACTOR_COMPUTE_UNITS).toBe(stats.computeUnits); - expect(usageUsd.ACTOR_COMPUTE_UNITS).toBeCloseTo(stats.computeUnits! * 0.2, 6); + // The runtime's own rounding rather than a tolerance - see the same assertion in the e2e test. + expect(usageUsd.ACTOR_COMPUTE_UNITS).toBe(Number((stats.computeUnits! * 0.2).toFixed(6))); expect(run.usageTotalUsd).toBe(usageUsd.ACTOR_COMPUTE_UNITS); expect(run).not.toHaveProperty('eventUsage'); @@ -523,7 +546,18 @@ describe('run usage estimate', () => { expect(live.stats.computeUnits).toBeGreaterThan(0); expect(live.stats.computeUnits).toBeGreaterThanOrEqual(before.stats.computeUnits!); + // The sampled figures are written before the status turns terminal, the same guarantee the log + // flush has: whoever sees the run finish reads the record next, and must not find them missing. + const firstTerminalRead = (async () => { + for (;;) { + const current = await getRegistries().runs.get(runId); + if (current && isTerminalJobStatus(current.status)) return current; + await new Promise((resolve) => setImmediate(resolve)); + } + })(); await finishRun(server, driver, runId); + expect((await firstTerminalRead).stats?.memAvgBytes).toBe(200); + const finished = (await server.client.run(runId).get()) as unknown as { stats: Record }; expect(finished.stats.memAvgBytes).toBe(200); expect(finished.stats.cpuMaxUsage).toBe(30); diff --git a/test/unit/input-schema-location.test.ts b/test/unit/input-schema-location.test.ts new file mode 100644 index 0000000..efd6868 --- /dev/null +++ b/test/unit/input-schema-location.test.ts @@ -0,0 +1,190 @@ +import { describe, expect, it } from 'vitest'; + +import { resolveInputSchemaLocation, type InputSchemaResolution } from '../../src/services/input-schema-location.js'; +import type { SourceFile } from '../../src/storage/entities.js'; + +/** A minimal but genuinely valid input schema - every field the Apify meta-schema demands is present, + * so a test that is not about schema validity never trips over it. */ +function validSchema(extra: Record = {}): Record { + return { + title: 'Sample input', + type: 'object', + schemaVersion: 1, + properties: { + maxPages: { title: 'Max pages', type: 'integer', description: 'How many pages.', default: 2 }, + }, + ...extra, + }; +} + +function text(name: string, content: string): SourceFile { + return { name, format: 'TEXT', content }; +} + +function json(name: string, value: unknown): SourceFile { + return text(name, JSON.stringify(value)); +} + +function expectResolved(resolution: InputSchemaResolution): Extract { + if (resolution.outcome !== 'resolved') throw new Error(`expected a resolved schema, got ${resolution.outcome}`); + return resolution; +} + +function expectFailure(resolution: InputSchemaResolution): Extract { + if (resolution.outcome !== 'failure') throw new Error(`expected a failure, got ${resolution.outcome}`); + return resolution; +} + +describe('resolveInputSchemaLocation', () => { + it('takes an inline schema object from the "input" field of .actor/actor.json', () => { + const resolution = expectResolved( + resolveInputSchemaLocation([ + json('.actor/actor.json', { actorSpecification: 1, name: 'a', input: validSchema() }), + json('.actor/input_schema.json', validSchema({ title: 'The file, which must lose' })), + ]), + ); + expect(resolution.schema.title).toBe('Sample input'); + expect(resolution.source).toContain('"input" field'); + expect(resolution.logLines.join('')).toContain('Using the input schema from'); + }); + + it('follows a path in the "input" field, resolved relative to .actor/', () => { + const resolution = expectResolved( + resolveInputSchemaLocation([ + json('.actor/actor.json', { actorSpecification: 1, input: './schemas/custom.json' }), + json('.actor/schemas/custom.json', validSchema({ title: 'From the named file' })), + ]), + ); + expect(resolution.schema.title).toBe('From the named file'); + expect(resolution.source).toContain('.actor/schemas/custom.json'); + }); + + it('falls back to the default locations, with a warning, when the named file is not in the pushed source', () => { + const resolution = expectResolved( + resolveInputSchemaLocation([ + json('.actor/actor.json', { actorSpecification: 1, input: './missing.json' }), + json('.actor/input_schema.json', validSchema({ title: 'The fallback' })), + ]), + ); + expect(resolution.schema.title).toBe('The fallback'); + expect(resolution.logLines.join('')).toContain('is not in the pushed source'); + }); + + it('treats an empty "input" field as "not found", falling through with a warning', () => { + const resolution = expectResolved( + resolveInputSchemaLocation([ + json('.actor/actor.json', { actorSpecification: 1, input: '' }), + json('input_schema.json', validSchema({ title: 'Root fallback' })), + ]), + ); + expect(resolution.schema.title).toBe('Root fallback'); + expect(resolution.logLines.join('')).toContain('falling back to the default locations'); + }); + + it('rejects an "input" path that escapes the Actor root, absolute or relative', () => { + for (const field of ['/etc/passwd', '../../outside.json']) { + const resolution = expectFailure( + resolveInputSchemaLocation([json('.actor/actor.json', { actorSpecification: 1, input: field })]), + ); + expect(resolution.reason).toBe('escapes-actor-root'); + expect(resolution.message).toContain('points outside the Actor root directory'); + } + }); + + it('rejects an "input" field that is neither a string nor an object', () => { + const resolution = expectFailure( + resolveInputSchemaLocation([json('.actor/actor.json', { actorSpecification: 1, input: 42 })]), + ); + expect(resolution.reason).toBe('invalid-input-field'); + expect(resolution.message).toBe('.actor/actor.json has invalid format: "input" must be a string or an object.'); + }); + + it('prefers .actor/INPUT_SCHEMA.json over the Actor root one, matching case-insensitively', () => { + const resolution = expectResolved( + resolveInputSchemaLocation([ + json('INPUT_SCHEMA.json', validSchema({ title: 'Root' })), + json('.actor/INPUT_SCHEMA.json', validSchema({ title: 'Actor dir' })), + ]), + ); + expect(resolution.schema.title).toBe('Actor dir'); + expect(resolution.source).toContain('.actor/INPUT_SCHEMA.json'); + }); + + it('reads a BASE64-encoded schema file the same as a TEXT one', () => { + const resolution = expectResolved( + resolveInputSchemaLocation([ + { + name: '.actor/input_schema.json', + format: 'BASE64', + content: Buffer.from(JSON.stringify(validSchema({ title: 'Encoded' })), 'utf8').toString('base64'), + }, + ]), + ); + expect(resolution.schema.title).toBe('Encoded'); + }); + + it('reports no schema at all - not a failure - when the source declares none', () => { + const resolution = resolveInputSchemaLocation([text('main.js', 'console.log(1)')]); + expect(resolution.outcome).toBe('none'); + }); + + it('fails on a schema file that cannot be parsed', () => { + const resolution = expectFailure(resolveInputSchemaLocation([text('.actor/input_schema.json', '{ "title": ')])); + expect(resolution.reason).toBe('unparseable-input-schema'); + expect(resolution.message).toContain('Could not parse the input schema ".actor/input_schema.json"'); + }); + + it("fails on a schema the platform's own meta-schema rejects, naming where it came from", () => { + const resolution = expectFailure( + resolveInputSchemaLocation([ + json('.actor/input_schema.json', { + title: 'Missing a field description', + type: 'object', + schemaVersion: 1, + properties: { maxPages: { title: 'Max pages', type: 'integer' } }, + }), + ]), + ); + expect(resolution.reason).toBe('invalid-input-schema'); + expect(resolution.message).toContain('.actor/input_schema.json'); + expect(resolution.message).toContain('is not valid'); + }); + + it('rejects a schema that is not an object at all', () => { + const resolution = expectFailure( + resolveInputSchemaLocation([json('.actor/input_schema.json', ['not', 'a', 'schema'])]), + ); + expect(resolution.reason).toBe('invalid-input-schema'); + }); + + it('fails on an unparseable .actor/actor.json rather than reporting no schema', () => { + const resolution = expectFailure( + resolveInputSchemaLocation([ + text('.actor/actor.json', '{ not json at all'), + json('.actor/input_schema.json', validSchema()), + ]), + ); + expect(resolution.reason).toBe('unparseable-actor-json'); + }); + + it('parses .actor/actor.json as JSON5, like the Dockerfile lookup already does', () => { + const resolution = expectResolved( + resolveInputSchemaLocation([ + text('.actor/actor.json', "{ actorSpecification: 1, input: './input_schema.json' /* a comment */ }"), + json('.actor/input_schema.json', validSchema({ title: 'Via JSON5' })), + ]), + ); + expect(resolution.schema.title).toBe('Via JSON5'); + }); + + it('accepts the $schema field real Actor templates carry', () => { + const resolution = resolveInputSchemaLocation([ + json( + '.actor/input_schema.json', + validSchema({ $schema: 'https://apify.com/schemas/v1/input.ide.json', title: 'Templated' }), + ), + ]); + + expect(resolution.outcome).toBe('resolved'); + }); +}); diff --git a/test/unit/input-schema.test.ts b/test/unit/input-schema.test.ts new file mode 100644 index 0000000..bbdf45d --- /dev/null +++ b/test/unit/input-schema.test.ts @@ -0,0 +1,341 @@ +import { describe, expect, it } from 'vitest'; + +import { assignDefaults, processActorInput } from '../../src/services/input-schema.js'; +import { describeInputSchemaDefect } from '../../src/services/input-schema-location.js'; +import type { InputSchema } from '../../src/storage/entities.js'; + +const SAMPLE_SCHEMA: InputSchema = { + title: 'Sample actor input', + type: 'object', + schemaVersion: 1, + properties: { + startUrl: { + title: 'Start URL', + type: 'string', + editor: 'textfield', + description: 'The page the crawler starts from.', + default: 'https://crawlee.dev/', + }, + maxPages: { + title: 'Max pages', + type: 'integer', + description: 'The maximum number of pages to crawl.', + default: 2, + minimum: 1, + }, + label: { title: 'Label', type: 'string', editor: 'textfield', description: 'A free-form label.' }, + }, + required: ['startUrl'], +}; + +function jsonInput(value: unknown): { body: Buffer; contentType: string } { + return { body: Buffer.from(JSON.stringify(value), 'utf8'), contentType: 'application/json' }; +} + +function effectiveInput(result: ReturnType): Record { + if (result.kind !== 'ok') throw new Error(`expected an accepted input, got ${result.kind}`); + expect(result.input.contentType).toBe('application/json'); + return JSON.parse(result.input.body.toString('utf8')) as Record; +} + +describe('processActorInput - defaults', () => { + it('fills every missing field from the schema and leaves provided ones alone', () => { + expect(effectiveInput(processActorInput(jsonInput({ maxPages: 7 }), SAMPLE_SCHEMA))).toEqual({ + maxPages: 7, + startUrl: 'https://crawlee.dev/', + }); + }); + + it('turns "no input at all" into the schema\'s defaults, the way the platform does', () => { + expect(effectiveInput(processActorInput(undefined, SAMPLE_SCHEMA))).toEqual({ + startUrl: 'https://crawlee.dev/', + maxPages: 2, + }); + }); + + it('treats a literal null body as no input, not as a type error', () => { + const result = processActorInput( + { body: Buffer.from('null', 'utf8'), contentType: 'application/json' }, + SAMPLE_SCHEMA, + ); + expect(effectiveInput(result)).toEqual({ startUrl: 'https://crawlee.dev/', maxPages: 2 }); + }); + + it("keeps a provided value even when it equals the type's zero value", () => { + const schema: InputSchema = { + title: 'Flags', + type: 'object', + schemaVersion: 1, + properties: { + enabled: { title: 'Enabled', type: 'boolean', description: 'A flag.', default: true }, + count: { title: 'Count', type: 'integer', description: 'A count.', default: 5 }, + }, + }; + expect(effectiveInput(processActorInput(jsonInput({ enabled: false, count: 0 }), schema))).toEqual({ + enabled: false, + count: 0, + }); + }); + + it('accepts application/json with parameters, such as a charset', () => { + const result = processActorInput( + { body: Buffer.from('{"maxPages":3}', 'utf8'), contentType: 'application/json; charset=utf-8' }, + SAMPLE_SCHEMA, + ); + expect(effectiveInput(result)).toMatchObject({ maxPages: 3 }); + }); +}); + +describe('assignDefaults', () => { + const NESTED_SCHEMA: InputSchema = { + title: 'Nested', + type: 'object', + schemaVersion: 1, + properties: { + config: { + title: 'Config', + type: 'object', + editor: 'json', + description: 'A nested object.', + properties: { + retries: { title: 'Retries', type: 'integer', description: 'How many.', default: 3 }, + verbose: { title: 'Verbose', type: 'boolean', description: 'Chatty?', default: false }, + }, + }, + items: { + title: 'Items', + type: 'array', + editor: 'json', + description: 'A list of objects.', + items: { type: 'object', properties: { keep: { type: 'boolean', default: true } } }, + }, + }, + }; + + it("builds a nested object out of its fields' defaults when the whole object is missing", () => { + expect(assignDefaults({}, NESTED_SCHEMA)).toEqual({ + config: { retries: 3, verbose: false }, + }); + }); + + it('leaves a provided nested object exactly as it came, adding no keys to it', () => { + expect(assignDefaults({ config: { retries: 9 } }, NESTED_SCHEMA)).toEqual({ + config: { retries: 9 }, + }); + }); + + it('fills defaults into every item of a provided array, without overwriting what the item has', () => { + expect(assignDefaults({ items: [{}, { keep: false }] }, NESTED_SCHEMA)).toEqual({ + config: { retries: 3, verbose: false }, + items: [{ keep: true }, { keep: false }], + }); + }); + + it('never hands out a reference into the schema, so a later edit of the input cannot corrupt it', () => { + const merged = assignDefaults({}, NESTED_SCHEMA) as { config: Record }; + merged.config.retries = 999; + expect(assignDefaults({}, NESTED_SCHEMA)).toEqual({ config: { retries: 3, verbose: false } }); + }); + + it("applies the field's own default over the defaults of its nested fields", () => { + const schema: InputSchema = { + title: 'Merged', + type: 'object', + schemaVersion: 1, + properties: { + config: { + title: 'Config', + type: 'object', + editor: 'json', + description: 'A nested object with its own default.', + default: { retries: 10 }, + properties: { + retries: { title: 'Retries', type: 'integer', description: 'How many.', default: 3 }, + verbose: { title: 'Verbose', type: 'boolean', description: 'Chatty?', default: false }, + }, + }, + }, + }; + expect(assignDefaults({}, schema)).toEqual({ config: { retries: 10, verbose: false } }); + }); +}); + +describe('processActorInput - validation', () => { + it("reports the platform's own message for a rejected value", () => { + expect(processActorInput(jsonInput({ maxPages: 0 }), SAMPLE_SCHEMA)).toEqual({ + kind: 'invalid-input', + message: 'Input is not valid: Field input.maxPages must be >= 1', + }); + expect(processActorInput(jsonInput({ label: 5 }), SAMPLE_SCHEMA)).toEqual({ + kind: 'invalid-input', + message: 'Input is not valid: Field input.label must be string', + }); + }); + + it('treats a required field that has a default as satisfied by that default', () => { + // `startUrl` is required and has a default - the platform relaxes exactly this case, since there + // is always a value for such a field by the time validation runs. + expect(effectiveInput(processActorInput(jsonInput({}), SAMPLE_SCHEMA))).toMatchObject({ + startUrl: 'https://crawlee.dev/', + }); + }); + + it('still reports a required field that has no default, even for a run started with no input', () => { + const schema: InputSchema = { + ...SAMPLE_SCHEMA, + required: ['startUrl', 'label'], + }; + expect(processActorInput(undefined, schema)).toEqual({ + kind: 'invalid-input', + message: 'Input is not valid: Field input.label is required', + }); + }); + + it('joins every validation error, the way the API-origin platform response does', () => { + // Two errors of different provenance: the first from AJV (which, configured as the platform + // configures it, reports one at a time), the second from the schema-aware checks that run after it. + const schema: InputSchema = { + title: 'Two problems', + type: 'object', + schemaVersion: 1, + properties: { + a: { title: 'A', type: 'integer', description: 'A number.', minimum: 10 }, + startUrls: { + title: 'Start URLs', + type: 'array', + editor: 'requestListSources', + description: 'Where to start.', + }, + }, + }; + const result = processActorInput(jsonInput({ a: 1, startUrls: [{ url: 'not a url' }] }), schema); + expect(result.kind).toBe('invalid-input'); + if (result.kind !== 'invalid-input') return; + expect(result.message).toContain('Field input.a must be >= 10'); + expect(result.message).toContain('input.startUrls'); + expect(result.message).toContain(', '); + }); + + it('requires at least one item in a required array field', () => { + const schema: InputSchema = { + title: 'Sources', + type: 'object', + schemaVersion: 1, + properties: { + startUrls: { + title: 'Start URLs', + type: 'array', + editor: 'requestListSources', + description: 'Where to start.', + }, + }, + required: ['startUrls'], + }; + expect(processActorInput(jsonInput({ startUrls: [] }), schema)).toMatchObject({ kind: 'invalid-input' }); + expect( + effectiveInput(processActorInput(jsonInput({ startUrls: [{ url: 'https://crawlee.dev/' }] }), schema)), + ).toEqual({ startUrls: [{ url: 'https://crawlee.dev/' }] }); + }); + + it('accepts any apifyProxyGroups selection, since proxy groups are not emulated locally', () => { + const schema: InputSchema = { + title: 'Proxy', + type: 'object', + schemaVersion: 1, + properties: { + proxyConfiguration: { + title: 'Proxy configuration', + type: 'object', + editor: 'proxy', + description: 'Proxy settings.', + default: { useApifyProxy: true, apifyProxyGroups: ['RESIDENTIAL'] }, + }, + }, + }; + expect(effectiveInput(processActorInput(undefined, schema))).toEqual({ + proxyConfiguration: { useApifyProxy: true, apifyProxyGroups: ['RESIDENTIAL'] }, + }); + // Shape checks still apply: a custom proxy URL that is not one is still rejected. + expect( + processActorInput( + jsonInput({ proxyConfiguration: { useApifyProxy: false, proxyUrls: ['nonsense'] } }), + schema, + ), + ).toMatchObject({ kind: 'invalid-input' }); + }); + + it('compiles a schema carrying the $schema field templates ship with', () => { + const schema: InputSchema = { ...SAMPLE_SCHEMA, $schema: 'https://apify.com/schemas/v1/input.ide.json' }; + expect(effectiveInput(processActorInput(jsonInput({ maxPages: 4 }), schema))).toMatchObject({ maxPages: 4 }); + }); +}); + +describe('processActorInput - malformed requests', () => { + it('rejects a body that is not sent as JSON', () => { + expect( + processActorInput({ body: Buffer.from('{}', 'utf8'), contentType: 'text/plain' }, SAMPLE_SCHEMA), + ).toEqual({ kind: 'invalid-input', message: 'Actor input must have content type "application/json".' }); + }); + + it('rejects a body that is not parseable JSON', () => { + const result = processActorInput( + { body: Buffer.from('{oops', 'utf8'), contentType: 'application/json' }, + SAMPLE_SCHEMA, + ); + expect(result.kind).toBe('invalid-input'); + if (result.kind === 'ok') return; + expect(result.message).toContain('Cannot parse input JSON body:'); + }); + + it('rejects a JSON body that is not an object, naming what it got instead', () => { + expect(processActorInput(jsonInput([1, 2]), SAMPLE_SCHEMA)).toEqual({ + kind: 'invalid-input', + message: 'The input JSON must be object, got "array" instead.', + }); + expect(processActorInput(jsonInput('a string'), SAMPLE_SCHEMA)).toEqual({ + kind: 'invalid-input', + message: 'The input JSON must be object, got "string" instead.', + }); + }); + + it('reports a schema that cannot be compiled at all, rather than throwing', () => { + const broken: InputSchema = { title: 'Broken', type: 'object', properties: { a: { type: 'not-a-type' } } }; + const result = processActorInput(jsonInput({}), broken); + expect(result.kind).toBe('invalid-input-schema'); + if (result.kind === 'ok') return; + expect(result.message).toContain('Input schema is not valid:'); + }); +}); + +describe('describeInputSchemaDefect', () => { + it('accepts a valid schema', () => { + expect(describeInputSchemaDefect(SAMPLE_SCHEMA)).toBeNull(); + }); + + it('names the defect of an invalid one', () => { + const defect = describeInputSchemaDefect({ + title: 'No description', + type: 'object', + schemaVersion: 1, + properties: { a: { title: 'A', type: 'string', editor: 'textfield' } }, + }); + expect(defect).toContain('description'); + }); + + it('rejects a required field that no property defines', () => { + const defect = describeInputSchemaDefect({ + title: 'Ghost', + type: 'object', + schemaVersion: 1, + properties: {}, + required: ['ghost'], + }); + expect(defect).toContain('ghost'); + }); + + it('rejects anything that is not an object', () => { + expect(describeInputSchemaDefect(['a'])).toBe('Input schema must be an object.'); + expect(describeInputSchemaDefect('a schema')).toBe('Input schema must be an object.'); + expect(describeInputSchemaDefect(null)).toBe('Input schema must be an object.'); + }); +}); diff --git a/test/unit/pricing.test.ts b/test/unit/pricing.test.ts index e88b14a..03409e4 100644 --- a/test/unit/pricing.test.ts +++ b/test/unit/pricing.test.ts @@ -3,10 +3,14 @@ * effect at a date, per-run resolution (tiered prices collapsed to the BRONZE tier), and the initial * `chargedEventCounts` with the synthetic start event (`actor-driver.md`'s "Pay-per-event pricing"). */ +import { readFileSync } from 'node:fs'; +import { dirname, join } from 'node:path'; +import { fileURLToPath } from 'node:url'; import { describe, expect, it } from 'vitest'; import { ACTOR_START_EVENT_NAME, + DEFAULT_DATASET_ITEM_EVENT_NAME, effectivePricingInfo, initialChargedEventCounts, isPayPerEvent, @@ -15,6 +19,8 @@ import { validatePricingInfosUpdate, } from '../../src/services/pricing.js'; +const REPO_ROOT = join(dirname(fileURLToPath(import.meta.url)), '..', '..'); + const NOW = new Date('2026-09-21T10:00:00.000Z'); const ppe = (events: Record, extra: Record = {}) => ({ @@ -261,6 +267,25 @@ describe('validatePricingInfosUpdate (the append-only history)', () => { ); }); + it('names the field that differs, so the usual "rebuilt from a file" mistake is visible', () => { + expect(rejection([{ ...existing[0]!, startedAt: '2026-01-02T00:00:00.000Z' }]).message).toContain( + 'at "startedAt" (sent "2026-01-02T00:00:00.000Z", stored "2026-01-01T00:00:00.000Z")', + ); + // An entry sent without the timestamps the stored one carries: they are stamped with the time of + // the request, so `createdAt` is the first thing that cannot match. + expect(rejection([ppe({ result: { eventTitle: 'Result', eventPriceUsd: 0.01 } })]).message).toContain( + 'at "createdAt"', + ); + expect( + rejection([ + { + ...existing[0]!, + pricingPerEvent: { actorChargeEvents: { result: { eventTitle: 'Result', eventPriceUsd: 0.02 } } }, + }, + ]).message, + ).toContain('at "pricingPerEvent.actorChargeEvents.result.eventPriceUsd" (sent 0.02, stored 0.01)'); + }); + it('refuses a new entry that starts at or before one already stored', () => { expect(rejection([...existing, { pricingModel: 'FREE', startedAt: '2025-12-01T00:00:00.000Z' }]).type).toBe( 'cannot-add-pricing-info-that-alters-past', @@ -286,3 +311,29 @@ describe('validatePricingInfosUpdate (the append-only history)', () => { expect(update([ppe({ result: { eventTitle: 'Result', eventPriceUsd: 0.01 } })], []).kind).toBe('ok'); }); }); + +describe('the pricing the bundled samples ship', () => { + // The README tells a developer to PUT these files at the Actor verbatim, so what the runtime would + // answer to that is worth knowing here rather than from a 400 at the terminal. + it.each(['sample_actor_ts', 'sample_actor_py'])( + '%s/pricing.json is accepted and prices both synthetic events', + (sample) => { + const { pricingInfos } = JSON.parse(readFileSync(join(REPO_ROOT, sample, 'pricing.json'), 'utf8')) as { + pricingInfos: unknown; + }; + + const resolved = resolveRunPricingInfo(ok(pricingInfos), NOW); + if (!isPayPerEvent(resolved)) throw new Error('expected the sample to be priced per event'); + expect(Object.keys(resolved.pricingPerEvent.actorChargeEvents).sort()).toEqual([ + ACTOR_START_EVENT_NAME, + DEFAULT_DATASET_ITEM_EVENT_NAME, + 'crawl-finished', + 'page-scraped', + ]); + + // Neither synthetic event is ever charged by the sample's own code: the start event is + // pre-charged per whole GB, the item event by the dataset route on every push. + expect(initialChargedEventCounts(resolved, 2048)?.[ACTOR_START_EVENT_NAME]).toBe(2); + }, + ); +}); diff --git a/test/unit/resources.test.ts b/test/unit/resources.test.ts new file mode 100644 index 0000000..ea644d0 --- /dev/null +++ b/test/unit/resources.test.ts @@ -0,0 +1,26 @@ +import { describe, expect, it } from 'vitest'; + +import { platformIncompatibleMemoryWarning } from '../../src/resources.js'; + +describe('platformIncompatibleMemoryWarning', () => { + it('says nothing about a grant the platform accepts', () => { + for (const memoryMbytes of [128, 256, 512, 1024, 2048, 4096, 8192, 32_768]) { + expect(platformIncompatibleMemoryWarning(memoryMbytes)).toBeUndefined(); + } + }); + + it('names the defect of a grant the platform refuses, without refusing it here', () => { + const betweenSteps = platformIncompatibleMemoryWarning(8096); + expect(betweenSteps).toContain('8096 MB is not a power of two'); + expect(betweenSteps).toContain('Running with it anyway.'); + + expect(platformIncompatibleMemoryWarning(64)).toContain('outside the 128 MB - 32768 MB range'); + expect(platformIncompatibleMemoryWarning(65_536)).toContain('outside the 128 MB - 32768 MB range'); + }); + + it('names both defects of a grant that is out of range and off the steps', () => { + expect(platformIncompatibleMemoryWarning(100)).toContain( + 'outside the 128 MB - 32768 MB range and not a power of two', + ); + }); +}); diff --git a/test/unit/runtime-log.test.ts b/test/unit/runtime-log.test.ts index 4775172..90c2171 100644 --- a/test/unit/runtime-log.test.ts +++ b/test/unit/runtime-log.test.ts @@ -76,10 +76,13 @@ describe('formatRuntimeLogLines', () => { it('renders a block whose emphasis differs line by line as one chunk', () => { expect( formatRuntimeLogLines([ - { text: '==== Local Actor runtime ====', emphasis: true }, + { text: '!! Running in `Live dev folder mode`', emphasis: true }, { text: 'Live dev folder: /src' }, ]), - ).toBe(`${BOLD_MARKER} ${BOLD}==== Local Actor runtime ====${RESET}\n` + `${MARKER} Live dev folder: /src\n`); + ).toBe( + `${BOLD_MARKER} ${BOLD}!! Running in \`Live dev folder mode\`${RESET}\n` + + `${MARKER} Live dev folder: /src\n`, + ); }); });