diff --git a/.env.example b/.env.example
new file mode 100644
index 0000000..377ae3d
--- /dev/null
+++ b/.env.example
@@ -0,0 +1,85 @@
+# Copy to .env and adjust. Every value below is the built-in default, so an
+# unset variable behaves exactly like the line shown.
+#
+# cp .env.example .env
+#
+# .env is gitignored. Never commit real secrets.
+
+# --- Required in production --------------------------------------------------
+
+# Flask/CSRF signing key. The built-in development default is publicly known, so
+# the app REFUSES TO START with it when APP_ENV=production.
+# python -c "import secrets; print(secrets.token_hex(32))"
+SECRET_KEY=dev-secret-key-change-in-production
+
+# The single canonical production signal. Setting it to "production" enables the
+# SECRET_KEY fail-fast, the Secure session cookie, and the rate-limit topology
+# guard. No other spelling is accepted -- two names would let a deployment
+# satisfy one gate and silently miss another.
+# APP_ENV=production
+
+# --- Deployment topology (see README "Deployment topology and rate limiting") --
+
+# Worker count. Single source of truth: gunicorn reads it natively and every
+# documented start command passes --workers "$WEB_CONCURRENCY". Required under
+# APP_ENV=production.
+WEB_CONCURRENCY=1
+
+# Replica count. Invisible from inside the process, so it is declared here and
+# must mirror render.yaml's numInstances. Required under APP_ENV=production.
+APP_REPLICAS=1
+
+# Rate-limit counter storage. memory:// counters are process-local, so the
+# effective limit is multiplied by workers x replicas. Anything above 1x1
+# requires a shared backend (and requirements-redis.txt installed).
+RATELIMIT_STORAGE_URI=memory://
+
+# Trust X-Forwarded-* from exactly one proxy hop. Enable ONLY when the app sits
+# behind a proxy you control; otherwise clients can forge their rate-limit bucket.
+TRUST_PROXY=0
+
+# --- Limits ------------------------------------------------------------------
+
+MAX_UPLOAD_SIZE=10485760
+PREVIEW_ROW_LIMIT=25
+FLATTEN_MAX_DEPTH=10
+
+# XLSX-only export budget, in cells (rows x columns). Derived from the
+# measurement in docs/export-budget-v1.2.md. CSV/TSV are streamed and uncapped.
+# 0 disables the guard.
+MAX_EXPORT_CELLS=250000
+
+# --- API fetch ---------------------------------------------------------------
+
+API_FETCH_TIMEOUT=30
+API_FETCH_MAX_RESPONSE=10485760
+
+# Ports the API-fetch feature may connect to. Empty disables the check.
+API_ALLOWED_PORTS=80,443,8443
+
+# DNS admission control. API_DNS_TIMEOUT bounds how long a REQUEST waits, not how
+# long the lookup runs -- getaddrinfo exposes no timeout and cannot be cancelled.
+# API_DNS_MAX_WORKERS bounds concurrency, which is the actual starvation fix.
+API_DNS_TIMEOUT=3
+API_DNS_MAX_WORKERS=4
+API_DNS_ADMISSION_TIMEOUT=1
+
+# --- Rate limits -------------------------------------------------------------
+
+RATE_LIMIT_DEFAULT=120/minute
+RATE_LIMIT_PROCESS=30/minute
+RATE_LIMIT_EXPORT=60/minute
+
+# --- Misc --------------------------------------------------------------------
+
+# Cache lifetime for static assets. Safe because asset URLs carry ?v=APP_VERSION.
+STATIC_MAX_AGE=86400
+
+# Responses below this size are not worth compressing.
+GZIP_MIN_SIZE=1024
+
+# Set to 0 to omit the version from /health.
+HEALTH_REVEAL_VERSION=1
+
+# Flask debug mode. Never enable in production.
+FLASK_DEBUG=0
diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml
new file mode 100644
index 0000000..2480cd9
--- /dev/null
+++ b/.github/workflows/ci.yml
@@ -0,0 +1,88 @@
+name: CI
+
+# Every commit is covered exactly once: a branch commit by its pull_request run,
+# a main commit by its push run.
+#
+# `push: ["**"]` alongside `pull_request` ran BOTH on every push to a PR branch.
+# The concurrency group below collapsed that pair, but collapsing means one of
+# the two is cancelled -- and GitHub attaches a cancelled check run to the head
+# commit, where it is not a success. So the PR reported `mergeable_state:
+# unstable` on every push even with CI fully green, which is indistinguishable
+# at a glance from a real failure. Not generating the duplicate beats cancelling
+# it.
+#
+# The trade-off: a branch with no pull request open gets no CI. That is the
+# intended shape -- the checks exist to gate the merge, and opening the PR is
+# what starts them.
+on:
+ push:
+ branches: ["main"]
+ pull_request:
+
+# Nothing here touches the GitHub API beyond checkout, so the token needs no
+# write scope; an explicit block also stops it inheriting wider repo defaults.
+permissions:
+ contents: read
+
+# Supersede a branch's own earlier run when a new commit lands on it: pushing
+# three times in a minute should not leave three runs competing for runners to
+# report on commits nobody is waiting for any more.
+#
+# This cancels only runs for SUPERSEDED commits, never the current head's --
+# the `on:` block above is what guarantees one run per commit, so there is no
+# same-SHA pair left to collapse. A cancelled check run lands on the old commit,
+# which is why this no longer costs the head commit its green state.
+#
+# The branch name alone is not unique across repositories: two pull requests
+# opened from different forks on a branch both named `feature` would share a
+# group, and cancel-in-progress would have them kill each other's checks. Keying
+# on the source repo keeps forks apart. head_ref is the source branch on a
+# pull_request and empty on a push; ref_name is the branch without the
+# refs/heads/ prefix (github.ref would keep it).
+concurrency:
+ group: ${{ github.workflow }}-${{ github.event.pull_request.head.repo.full_name || github.repository }}-${{ github.head_ref || github.ref_name }}
+ cancel-in-progress: true
+
+jobs:
+ test:
+ runs-on: ubuntu-latest
+ steps:
+ - uses: actions/checkout@v4
+ with:
+ # Nothing after this step talks to GitHub, so leaving the token in
+ # .git/config only widens what a compromised dependency can reach.
+ persist-credentials: false
+
+ - uses: actions/setup-python@v5
+ with:
+ python-version: "3.11"
+
+ - name: Install dependencies
+ run: |
+ python -m pip install --upgrade pip
+ pip install -r requirements-dev.txt
+
+ - name: Lint
+ run: ruff check .
+
+ - name: Format check
+ run: ruff format --check .
+
+ - uses: actions/setup-node@v4
+ with:
+ node-version: "22"
+
+ - name: Tests
+ run: python -m pytest tests/ -v
+
+ - name: Client-side export assertions (F1)
+ run: node tests/js/test_export_sanitize.mjs
+
+ - name: Client-side render/cap assertions (P4/P5/P13)
+ run: node tests/js/test_render_caps.mjs
+
+ - name: Client-side feature assertions (Phase 4)
+ run: node tests/js/test_features.mjs
+
+ - name: Audit runtime dependencies
+ run: pip-audit -r requirements.txt
diff --git a/.gitignore b/.gitignore
index 6e098ba..8b96fd9 100644
--- a/.gitignore
+++ b/.gitignore
@@ -19,6 +19,9 @@ ENV/
.DS_Store
Thumbs.db
+# Node (test assertions only; no build step)
+node_modules/
+
# Project specific
*.log
.env
diff --git a/AGENTS.md b/AGENTS.md
index e7c4914..12287c8 100644
--- a/AGENTS.md
+++ b/AGENTS.md
@@ -18,15 +18,15 @@ Contract for AI coding agents (Claude Code, Cursor, GitHub Copilot Workspace, Ai
| Item | Value |
|-------------------|----------------------------------------------------------------|
| Language | Python 3.11+ (Render targets 3.14.5) |
-| Web framework | Flask 3.0.0 (app factory) |
+| Web framework | Flask 3.1.3 (app factory) |
| Security libs | Flask-WTF 1.2.1 (CSRF), Flask-Limiter 3.5.0 (rate limit) |
-| HTTP client | requests 2.31.0 |
-| Excel export | openpyxl 3.1.2 |
-| WSGI server | gunicorn 21.2.0 — invoked as `gunicorn "app:create_app()"` |
-| Test runner | pytest 7.4.4 |
+| HTTP client | requests 2.33.0 |
+| Excel export | openpyxl 3.1.5 |
+| WSGI server | gunicorn 22.0.0 — `gunicorn "app:create_app()" --workers "$WEB_CONCURRENCY" --timeout 60` |
+| Test runner | pytest 9.0.3 (`requirements-dev.txt`) + ruff, coverage, pip-audit |
| Frontend | HTML + external CSS/JS (no framework, no build) |
| Deployment | Render (free tier) via `render.yaml`; also docs for Docker/Nginx |
-| App version | `Config.APP_VERSION = '1.1.0'` (returned by `/health`) |
+| App version | `Config.APP_VERSION = '1.2.0'` (returned by `/health`) |
## Setup
@@ -37,7 +37,7 @@ python -m venv .venv
# macOS/Linux
source .venv/bin/activate
-pip install -r requirements.txt
+pip install -r requirements-dev.txt
python app.py # http://localhost:5000
# enable debug
@@ -56,43 +56,52 @@ python -m pytest tests/ -v
### Backend
-- **`app.py`** — `create_app(config_class=Config)` factory. Initializes CSRF + Limiter, registers `apply_security_headers` as an `after_request` middleware, registers the `routes.bp` Blueprint. Module-level `app = create_app()` exists for tooling that expects it, but production uses the factory directly.
-- **`config.py`** — `Config` class; every setting reads from `os.environ.get(...)` with a default. Includes `SECRET_KEY`, `MAX_CONTENT_LENGTH`, `PREVIEW_ROW_LIMIT`, `API_FETCH_TIMEOUT`, `API_FETCH_MAX_RESPONSE`, `FLATTEN_MAX_DEPTH`, `RATELIMIT_*`, `APP_VERSION`, `DEBUG`.
-- **`extensions.py`** — Bare `CSRFProtect()` and `Limiter(key_func=get_remote_address)` instances, bound by `app.py` via `init_app`. Importing this module never has side effects on the Flask app — that's the point.
+- **`app.py`** — `create_app(config_class=Config)` factory. Asserts the production SECRET_KEY, runs `check_rate_limit_topology`, optionally installs `ProxyFix` (`TRUST_PROXY=1`), initializes CSRF + Limiter, registers `apply_security_headers` and `compress_response` as `after_request` middleware plus the JSON 413/500/404 handlers, and registers the `routes.bp` Blueprint. Module-level `app = create_app()` exists for tooling that expects it, but production uses the factory directly.
+- **`config.py`** — `Config` class plus `is_production()`, `env_int()` and `env_int_set()`. Every setting reads from the environment with a default; integers report which variable was mistyped instead of raising a bare `ValueError`. Includes `SECRET_KEY`, `MAX_CONTENT_LENGTH`, `PREVIEW_ROW_LIMIT`, `API_FETCH_*`, `API_DNS_*`, `API_ALLOWED_PORTS`, `FLATTEN_MAX_DEPTH`, `MAX_EXPORT_CELLS`, `RATELIMIT_*`, `WEB_CONCURRENCY`, `APP_REPLICAS`, `TRUST_PROXY`, `SESSION_COOKIE_*`, `SEND_FILE_MAX_AGE_DEFAULT`, `GZIP_MIN_SIZE`, `HEALTH_REVEAL_VERSION`, `APP_VERSION`, `DEBUG`. See `.env.example`.
+- **`extensions.py`** — Bare `CSRFProtect()` and `Limiter(key_func=client_ip_key)` instances, bound by `app.py` via `init_app`. `client_ip_key` reads `request.remote_addr` and never the raw `X-Forwarded-For`, so the bucket is whatever ProxyFix decided rather than something a client can assert. Importing this module never has side effects on the Flask app — that's the point.
- **`security.py`**
- - `validate_url(url)` — returns `(is_valid, error_or_none)`. Rejects non-http(s) schemes, missing hostname, non-resolvable hostnames, and any resolved IP where `not ip.is_global or ip.is_multicast`.
- - `apply_security_headers(response)` — sets CSP (strict, `script-src 'self'`), `X-Frame-Options: DENY`, `X-Content-Type-Options: nosniff`, `Referrer-Policy: strict-origin-when-cross-origin`. CSP allows Google Fonts (style/font) and `data:` images.
+ - `validate_url(url)` — returns `(is_valid, error_or_none)`. Rejects non-http(s) schemes, missing hostname, ports outside `API_ALLOWED_PORTS`, non-resolvable hostnames, and any resolved IP where `not ip.is_global or ip.is_multicast`.
+ - `resolve_hostname(hostname)` / `get_resolver_pool()` — DNS on a shared fixed-size pool with admission control. Bounds the caller's wait and concurrency; **not** the lookup itself, and **not** teardown.
+ - `apply_security_headers(response)` — sets CSP (strict, `script-src 'self'`, plus `object-src 'none'`, `base-uri 'self'`, `frame-ancestors 'none'`, `form-action 'self'`, `upgrade-insecure-requests`), `X-Frame-Options: DENY`, `X-Content-Type-Options: nosniff`, `Referrer-Policy`, `Permissions-Policy`, COOP/CORP, HSTS on secure requests only, and `Cache-Control: no-store` on the data and health endpoints. CSP allows Google Fonts (style/font) and `data:` images.
- **`helpers.py`**
- `flatten_for_csv(data, parent_key='', sep='.', _depth=0, max_depth=10)` — depth-capped recursion; deep nesting is JSON-stringified instead of stack-overflowing.
- - `extract_table_data(json_data)` — heuristic: list-of-dicts → use directly; dict with array value → that array; nested dicts → recurse; otherwise single-row.
+ - `extract_table_data(json_data, _depth=0, max_depth=10)` — heuristic: list-of-dicts → use directly; dict with array value → that array; nested dicts → recurse (depth-capped); otherwise single-row.
+ - `flatten_rows(rows, max_depth=10)` — one pass returning `(flattened_rows, sorted_columns)`; replaces flatten-then-`get_all_columns`.
+ - `sanitize_cell(value)` / `serialize_cell_value(value)` / `is_formula_trigger(value)` — spreadsheet formula-injection defenses. **CSV/TSV/XLSX only.**
+ - `preview_truncate(row)` — capped *copy* of a preview row; never mutates the source.
+ - `format_size(num_bytes)` — byte count in the largest non-zero unit. Shared utilities live here, **never in `app.py`**: `create_app()` imports `bp` from `routes.py`, so importing up creates a cycle.
- `parse_jsonl(text)` — line-by-line JSON; raises `ValueError` with the offending line number.
- - `find_candidate_arrays(json_data, prefix='', candidates=None)` — discovers every array-of-objects with `{path, length, sample_keys}` metadata so the frontend can prompt the user.
- `extract_by_path(json_data, path)` — dot-notation navigation; `'(root)'` is the sentinel for top-level lists.
- `get_all_columns(data)` — sorted union of keys.
- **`routes.py`** — Blueprint `bp`. Routes:
- `GET /` → `templates/index.html`.
- - `GET /health` → `{"status": "ok", "version": APP_VERSION}`.
- - `POST /process` → rate-limited (default `RATE_LIMIT_PROCESS=30/min`). Reads `input_method` (`file`/`paste`/`api`), `data_format` (`json`/`jsonl`), optional `json_path`. Returns `{success, columns, preview, total_rows, csv_data, csv_columns}` **or** `{needs_selection: true, candidates: [...]}` when multiple arrays found and no `json_path` selected.
- - `POST /export-csv` → rate-limited (default `RATE_LIMIT_EXPORT=60/min`). Server-side CSV fallback.
- - `POST /export-xlsx` → rate-limited. Server-side Excel via openpyxl.
+ - `GET /health` → `{"status": "ok", "version": APP_VERSION}` (version omitted when `HEALTH_REVEAL_VERSION=0`).
+ - `GET /health/live` → process liveness; checks nothing, so a dependency outage cannot cause a restart loop.
+ - `GET /health/ready` → readiness; 200 or 503 with a `checks` map.
+ - `POST /process` → rate-limited (default `RATE_LIMIT_PROCESS=30/min`). Reads `input_method` (`file`/`paste`/`api`), `data_format` (`json`/`jsonl`), optional `json_path`. Returns `{success, columns, preview, preview_limit, total_rows, total_cells, max_export_cells, csv_data, csv_columns}` **or** `{needs_selection: true, raw_json: ...}` when no `json_path` was selected, so the frontend can render the JSON tree picker. The body is built by `_load_input()` and `_select_table_data()`; keep `process_json` itself thin.
+ - `POST /export-csv` → rate-limited (default `RATE_LIMIT_EXPORT=60/min`). Generator-streamed and **uncapped**.
+ - `POST /export-xlsx` → rate-limited. Server-side Excel via openpyxl, capped by `MAX_EXPORT_CELLS` (400 when exceeded, never truncated). Diskless: no OS temp files.
### Frontend
- **`templates/index.html`** — pure structure. CSRF meta tag at the top (``). References `style.css` and `app.js` via `url_for('static', ...)`. No inline JS/CSS (CSP would block it).
-- **`static/css/style.css`** — `:root` (dark, default) + `:root.light` overrides. All component styles, sort indicators, modal, export dropdown, theme toggle.
-- **`static/js/app.js`** — reads CSRF token from meta tag; attaches it to FormData and `X-CSRFToken`. Handles tab switching, drag-drop, JSON/JSONL toggle, auth method visibility, client-side sort, **client-side** CSV/TSV download, **server-side** Excel via `/export-xlsx`, theme persistence in `localStorage`, and the path-selection modal triggered by `needs_selection: true`.
+- **`static/css/style.css`** — `:root` (dark, default) + `:root.light` overrides. All component styles, sort indicators, modals, export dropdown, table toolbar (filter / load-more / column visibility), theme toggle.
+- **`static/js/app.js`** — reads CSRF token from meta tag; attaches it to FormData and `X-CSRFToken`. Handles tab switching, drag-drop, JSON/JSONL toggle, auth method visibility, client-side sort/filter/pagination/column visibility, **client-side** CSV, TSV, JSONL and Markdown downloads, **server-side** Excel via `/export-xlsx`, theme persistence in `localStorage`, the lazily-built JSON tree picker triggered by `needs_selection: true` (with `#path=` deep links), and the About modal. No `alert()`.
### Tests
- **`tests/conftest.py`** — provides `app` (with `TESTING=True`, `WTF_CSRF_ENABLED=False`) and `client` fixtures.
- **`tests/test_helpers.py`** — pure-function tests for the data-processing layer.
- **`tests/test_security.py`** — `validate_url` with mocked `socket.getaddrinfo`; verifies private-IP/loopback/link-local rejection.
-- **`tests/test_routes.py`** — integration tests for every route, including security headers, JSONL, path selection, Excel export.
+- **`tests/test_routes.py`** — integration tests for every route, including security headers, JSONL, path selection, exports, formula injection, log hygiene, config gates and the topology guard.
+- **`tests/js/*.mjs`** — Node assertions that load the real `static/js/app.js` in a stubbed DOM (`dom_stub.mjs`). No build step, no dependencies. Run them with `make test-js`; CI runs them too.
### Deploy / Config
-- **`render.yaml`** — Render Blueprint. `startCommand: gunicorn "app:create_app()" --bind 0.0.0.0:$PORT`. `SECRET_KEY` is `generateValue: true` so Render auto-fills it.
-- **`requirements.txt`** — exact-pinned. Don't loosen.
+- **`render.yaml`** — Render Blueprint. The start command derives `--workers` from `$WEB_CONCURRENCY` and sets `--timeout 60`; `APP_ENV=production`, `WEB_CONCURRENCY` and `APP_REPLICAS` are declared; `SECRET_KEY` is `generateValue: true`; `autoDeployTrigger: checksPass` gates deploys on CI.
+- **`requirements.txt`** — exact-pinned runtime deps. Don't loosen. `requirements-dev.txt` holds test tooling; `requirements-redis.txt` holds the optional Redis client for shared rate-limit storage.
+- **`.github/workflows/ci.yml`** — ruff check, ruff format --check, pytest, the Node assertions, and `pip-audit`.
+- **`Makefile` / `.env.example`** — developer entry points and every environment variable with its default.
## Conventions
@@ -103,7 +112,11 @@ python -m pytest tests/ -v
- Apply `@limiter.limit(lambda: current_app.config.get('RATE_LIMIT_...'))` on any new mutating route.
- Read config via `current_app.config['KEY']`, not by re-importing `Config` at request time.
- For new auth methods on the API tab: add a select option in `index.html`, a `data-auth="..."` fieldset, the JS visibility branch in `app.js`, and the conditional in `routes.py`.
-- For new exports: add the entry to the export dropdown, the handler in `app.js`, optionally a server route in `routes.py`. Pin any new dependency.
+- For new exports: add the entry to the export dropdown, the handler in `app.js`, optionally a server route in `routes.py`. Pin any new dependency. **Decide the sanitization policy explicitly.** Spreadsheet-compatible formats need one of two defenses, not both:
+ - **Quote-prefix via `sanitize_cell`** for formats with no type channel (CSV, TSV). The prefix is the only place a value can be marked as text.
+ - **Pin the cell type** for formats that have one. `/export-xlsx` sets `data_type='s'` on every cell, so openpyxl writes a string cell and Excel never evaluates it; prefixing on top would put a stray `'` in the data.
+
+ Non-spreadsheet formats (JSONL, Markdown) get neither — a prefix there corrupts values while protecting nothing.
- Add or update tests under `tests/` for any backend behavior change. Class-style grouping (`class TestXxx:`) is the existing pattern.
- Preserve the CSP. If new third-party CSS/fonts are needed, edit `apply_security_headers` deliberately.
@@ -115,21 +128,33 @@ python -m pytest tests/ -v
- Don't unbound the streamed download — `API_FETCH_MAX_RESPONSE` (10 MB default) is enforced inside the chunk loop.
- Don't lower `FLATTEN_MAX_DEPTH` recursion guard without checking known payload shapes.
- Don't log full exceptions to the response — the routes intentionally return generic messages and `logger.exception(...)` for server logs.
-- Don't use the dev `SECRET_KEY` in production. Render generates one via `render.yaml`; for other deployments set the env var.
+- Don't use the dev `SECRET_KEY` in production. Render generates one via `render.yaml`; for other deployments set the env var. `create_app` refuses to start on it when `APP_ENV=production`.
+- Don't infer production from `not DEBUG`, and don't add a second spelling of `APP_ENV` — `config.is_production()` is the only signal.
+- Don't log the URL, the exception, or any request field on the API-fetch failure path — a token can ride in the query string, fragment, userinfo *or* path.
+- Don't describe the DNS resolver teardown as bounded. `API_DNS_TIMEOUT` bounds the caller's wait; `getaddrinfo` cannot be cancelled.
+- Don't write payloads to disk. openpyxl's `write_only` mode and a rolled-over `SpooledTemporaryFile` both do.
+- Don't truncate an oversized export — refuse it. A partial spreadsheet is worse than none.
+- Don't hardcode a bare `--workers N` in a start command; derive it from `$WEB_CONCURRENCY` so the declared and running counts cannot drift.
+- Don't relax the CSP, and don't reintroduce `alert()` — the About dialog is an in-page modal.
## Verification Checklist (before reporting a task done)
-- [ ] `python -m pytest tests/ -v` passes.
-- [ ] `python app.py` starts cleanly; `GET /` renders; `GET /health` returns the current `APP_VERSION`.
+- [ ] `python -m pytest tests/ -v` passes (the passing command is the criterion, not a count).
+- [ ] `ruff check .` and `ruff format --check .` exit 0.
+- [ ] The Node assertions pass: `make test-js`.
+- [ ] `pip-audit -r requirements.txt` reports 0 vulnerabilities.
+- [ ] `python app.py` starts cleanly; `GET /` renders; `GET /health` returns the current `APP_VERSION`; `/health/live` and `/health/ready` respond.
- [ ] CSP headers still present (`curl -sI http://localhost:5000/ | findstr /i security` or browser DevTools).
- [ ] CSRF still required on POSTs (a POST without the token returns 400 from Flask-WTF).
- [ ] All three input methods (file / paste / API), both formats (JSON / JSONL), and all four auth methods still work end to end.
-- [ ] Multi-array JSON triggers the path-selector modal; selecting a path returns rows.
-- [ ] CSV (client-side), TSV (client-side), and XLSX (server-side) all download with full row counts, not just 25.
-- [ ] SSRF guard blocks `http://127.0.0.1`, `http://localhost`, `http://169.254.169.254`, and similar private IPs.
+- [ ] Multi-array JSON triggers the JSON tree picker; selecting a path returns rows; `#path=...` pre-selects one.
+- [ ] CSV, TSV, JSONL and Markdown (client-side) and XLSX (server-side) all download with full row counts, not just the preview.
+- [ ] A cell value of `=SUM(A1)` opens as inert text in CSV and XLSX, and survives verbatim in JSONL and Markdown.
+- [ ] SSRF guard blocks `http://127.0.0.1`, `http://localhost`, `http://169.254.169.254`, `http://2130706433`, `http://0x7f000001`, `http://[::ffff:7f00:1]`, and any port outside `API_ALLOWED_PORTS`.
- [ ] Rate limit kicks in at the configured threshold (manual: rapid-fire `/process`).
-- [ ] No new file writes, DB calls, payload logging, or inline JS/CSS were introduced.
-- [ ] Any new dependency is **exact-pinned** in `requirements.txt`.
+- [ ] No new file writes, DB calls, payload logging, inline JS/CSS, or CSP relaxations were introduced.
+- [ ] Any new dependency is **exact-pinned** in `requirements.txt` (or `requirements-dev.txt` / `requirements-redis.txt`).
+- [ ] `.env.example`, `README.md` and `CHANGELOG.md` cover any new environment variable.
## When in Doubt
diff --git a/CHANGELOG.md b/CHANGELOG.md
new file mode 100644
index 0000000..64ec424
--- /dev/null
+++ b/CHANGELOG.md
@@ -0,0 +1,140 @@
+# Changelog
+
+All notable changes to this project are documented here.
+
+The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and
+this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
+
+## [1.2.0] - 2026-08-21 — "Hardening & Performance"
+
+Implements `docs/roadmap-v1.2.md`, closing every finding in
+`docs/security-review-v1.2.md` (F1–F17) and `docs/performance-review-v1.2.md`
+(P1–P13).
+
+No response key changed name, type or meaning; `/process` only gained keys.
+
+### Security
+
+- **CSV/Excel formula injection (F1, Critical).** Values beginning with
+ `=`, `+`, `-`, `@`, tab, CR or LF were written verbatim into every export.
+ CSV/TSV now prefix them with a single quote; XLSX pins the cell's `data_type`
+ to a string, since openpyxl otherwise serializes a leading `=` as a formula.
+ Applies to all four export paths, headers included. JSONL and Markdown exports
+ are deliberately exempt — see *Added*.
+- **Dependency CVEs (F2, High).** Flask 3.1.3, requests 2.33.0, gunicorn 22.0.0
+ (CVE-2024-1135 request smuggling), openpyxl 3.1.5.
+ `pip-audit -r requirements.txt` goes from 7 vulnerabilities to 0.
+- **Rate limiting behind a proxy (F12, High).** Opt-in `TRUST_PROXY=1` installs
+ ProxyFix with exactly one trusted hop, so users stop sharing a single bucket.
+ Off by default; forwarded headers are ignored unless enabled.
+- **Credential leakage into logs (F3/F9).** The API-fetch failure log is now a
+ fixed string. `requests`' exception text carries the full URL, and the query
+ string, fragment, userinfo *and* path can each hold a token.
+- **Outbound header allowlist (F4).** The client supplies the header *name* for
+ API-key auth; names are stripped and lowercased before an explicit allowlist
+ check, so `Host`, `Proxy-Authorization` and `CoNnEcTiOn` no longer pass.
+- **Security headers (F5).** Added HSTS (secure requests only),
+ `Permissions-Policy`, `Cross-Origin-Opener-Policy` and
+ `Cross-Origin-Resource-Policy`; CSP gained `object-src 'none'`,
+ `base-uri 'self'`, `frame-ancestors 'none'`, `form-action 'self'` and
+ `upgrade-insecure-requests`.
+- **SSRF: bounded DNS and a port allowlist (F6).** Lookups run on a shared,
+ fixed-size pool with admission control, so a slow nameserver no longer holds a
+ request thread for the length of the lookup: the caller returns on
+ `API_DNS_TIMEOUT`, and once the pool is saturated further requests are refused
+ on `API_DNS_ADMISSION_TIMEOUT` instead of queueing. The lookup itself is not
+ bounded — `getaddrinfo` exposes no timeout and cannot be cancelled, so the
+ pool thread stays occupied until the platform resolver returns.
+ `reset_resolver_pool()` shuts the pool down with `wait=False` and returns
+ immediately, so it does not block on that thread; interpreter exit still can. `API_ALLOWED_PORTS` defaults to `80,443,8443`.
+- **SECRET_KEY fail-fast (F7).** `APP_ENV=production` with the publicly known
+ dev key (or an empty one) refuses to start. Integer settings now report which
+ variable was mistyped.
+- **Recursion-depth DoS (F8).** `extract_table_data` gained the depth cap
+ `flatten_for_csv` already had, and `RecursionError` anywhere in the pipeline
+ returns 400 `JSON nesting too deep` instead of 500.
+- **JSON error handlers (F10).** 413, 500 and 404 return JSON, so the client's
+ `response.json()` no longer throws on an HTML body.
+- **`Cache-Control: no-store` (F11)** on all data-bearing and health responses.
+- **Upload validation (F13).** Non-`.json`/`.jsonl` filenames and clearly wrong
+ content types are rejected server-side.
+- **Cookie hardening (F16).** `HttpOnly`, `SameSite=Lax`, and `Secure` tied to
+ `APP_ENV=production`.
+
+### Performance
+
+- **gzip compression (P1).** ~40 lines of middleware, no new dependency. Skips
+ streamed, bodyless, already-encoded and non-text responses.
+- **Diskless, memory-bounded exports (P3).** CSV/TSV are generator-streamed and
+ uncapped. XLSX keeps a normal-mode workbook (no OS temp files) plus
+ `MAX_EXPORT_CELLS`, measured rather than guessed — see
+ `docs/export-budget-v1.2.md`.
+- **Lazy tree picker (P4)** and **client render caps (P5)**: children build on
+ first toggle; nested objects stop at 20 keys, primitive arrays at 20 items,
+ strings at 500 characters.
+- **Non-mutating preview truncation (P2.2/P5).** The preview is a capped copy;
+ `csv_data` and every export keep full fidelity.
+- **Static asset caching (P6)** for a day, behind `?v=APP_VERSION` URLs.
+- **Memory trims (P8/P12).** One-pass flatten-and-collect-columns, and the
+ API path decodes its `bytearray` without an intermediate copy.
+- **gunicorn `--timeout 60` (P9)** everywhere, above `API_FETCH_TIMEOUT`.
+- **Chunked Blob for client exports (P13).**
+
+### Added
+
+- `/health/live` and `/health/ready` alongside the unchanged `/health`.
+- **Load more / Load all** over the full dataset, with a 50,000-row DOM guard.
+- **Row filter**: case-insensitive substring across all values.
+- **JSONL and Markdown exports.** Neither is formula-sanitized: JSON has types
+ and nothing evaluates it, and a leading `=` is inert in Markdown — Markdown
+ gets pipe/newline escaping instead. Both write the same flattened columns the
+ table shows (nested objects as dotted keys, nested arrays as JSON strings),
+ because the unflattened rows are never sent to the browser; see
+ "Known limitations".
+- **Column visibility toggle** and **deep-linkable path selection**
+ (`#path=users.0.orders`).
+- `/process` returns `preview_limit`, `total_cells` and `max_export_cells`, so
+ the badge reflects config and the Excel entry is greyed out before the click.
+- CI (`.github/workflows/ci.yml`), `pyproject.toml` (ruff + pytest),
+ `requirements-dev.txt`, `requirements-redis.txt`, `.env.example`, `Makefile`,
+ and Node assertion suites for `static/js/app.js`.
+- The rate-limit topology guard: `RATELIMIT_STORAGE_URI` is configurable at last,
+ and a production deployment must declare `WEB_CONCURRENCY` and `APP_REPLICAS`.
+
+### Changed
+
+- `alert()` About dialog replaced with an in-page modal that reads the version
+ from config; export dropdown is keyboard-accessible.
+- Render auto-deploy now waits for CI (`autoDeployTrigger: checksPass`).
+- License references corrected from MIT to GPL-3.0 (F17).
+
+### Removed
+
+- `find_candidate_arrays` and its four tests — dead code from the old candidates
+ handshake the JSON tree picker replaced (P10/D2).
+
+### Known limitations
+
+- **JSONL export is not a faithful copy of the input document.** Roadmap 4.3
+ called it "lossless — original values". It writes values verbatim (no formula
+ prefixing, which was the security-relevant half of that decision), but over
+ the server's *flattened* projection: `{"tags": [1, 2]}` exports as
+ `"tags": "[1, 2]"`, and `{"meta": {"role": "x"}}` as `"meta.role": "x"`.
+ Only `csv_data` (flattened) reaches the browser for the full dataset —
+ `preview` is truncated and capped at `preview_limit` rows — so a
+ round-tripping export would mean shipping the original rows alongside the
+ flattened ones, doubling the payload and client memory that P2 and P12 exist
+ to reduce. Flagged for a maintainer decision rather than resolved either way.
+
+### Not included
+
+- **Opt-in HTTP Basic Auth gate** (roadmap 4.6). Decision D4 is still open and
+ needs maintainer sign-off. Nothing else in this release depends on it.
+
+## [1.1.0] - 2026-05
+
+- JSON tree picker replaces the multi-array `candidates` handshake.
+- JSONL support across file, paste and API input.
+- Client-side column sorting, light/dark theme, CSV/TSV/Excel export.
+- CSRF protection, SSRF validation with DNS resolution, rate limiting, and a
+ strict CSP with no inline scripts.
diff --git a/CLAUDE.md b/CLAUDE.md
index e38ed2a..450ade2 100644
--- a/CLAUDE.md
+++ b/CLAUDE.md
@@ -12,28 +12,44 @@ A lightweight Flask web application that converts JSON/JSONL data into viewable
```
json-table-tool/
-├── app.py # Flask app factory, middleware registration
-├── config.py # All settings via environment variables
+├── app.py # Flask app factory, startup gates, gzip + error handlers
+├── config.py # All settings via environment variables; is_production()
├── extensions.py # Flask-WTF (CSRF) and Flask-Limiter instances
-├── security.py # SSRF protection (DNS validation) + security headers
-├── helpers.py # Data processing (flatten, extract, JSONL parse, path selector)
+├── security.py # SSRF (port allowlist + bounded DNS) + security headers
+├── helpers.py # Data processing (flatten, extract, JSONL, sanitize, preview)
├── routes.py # Flask Blueprint with all route handlers
├── static/
│ ├── css/
│ │ └── style.css # All CSS (dark/light themes, components, utilities)
│ └── js/
-│ └── app.js # All JavaScript (UI, sorting, export, theme, modal)
+│ └── app.js # All JavaScript (UI, table, exports, theme, modals)
├── templates/
│ └── index.html # HTML structure only (refs external CSS/JS)
├── tests/
│ ├── conftest.py # Shared pytest fixtures (app, client)
│ ├── test_helpers.py # Tests for data processing functions
-│ ├── test_security.py # Tests for SSRF validation
-│ └── test_routes.py # Integration tests for all routes
+│ ├── test_security.py # SSRF validation, port allowlist, bounded DNS resolver
+│ ├── test_routes.py # Integration tests for all routes + config gates
+│ └── js/ # Node assertions against the real app.js (no build step)
+│ ├── dom_stub.mjs
+│ ├── test_export_sanitize.mjs
+│ ├── test_render_caps.mjs
+│ └── test_features.mjs
├── docs/
-│ └── code-health-final.md
-├── requirements.txt # Python dependencies (pinned versions)
+│ ├── code-health-final.md
+│ ├── security-review-v1.2.md
+│ ├── performance-review-v1.2.md
+│ ├── roadmap-v1.2.md
+│ └── export-budget-v1.2.md # How MAX_EXPORT_CELLS was measured
+├── .github/workflows/ci.yml # lint, format, tests, JS assertions, pip-audit
+├── pyproject.toml # ruff + pytest configuration
+├── requirements.txt # Runtime dependencies (exact-pinned)
+├── requirements-dev.txt # Test/lint tooling
+├── requirements-redis.txt # Optional Redis client for shared rate-limit storage
+├── .env.example # Every environment variable with its default
+├── Makefile # test / test-js / lint / format / audit / coverage / run
├── render.yaml # Render.com deployment blueprint
+├── CHANGELOG.md
├── AGENTS.md # Agent contract / quick-reference
├── MEMORY.md # Memory index for AI assistants
├── README.md
@@ -42,15 +58,15 @@ json-table-tool/
## Tech Stack
-- **Backend:** Python 3.11+ (Render deploys with 3.14.5), Flask 3.0.0 (app factory pattern)
+- **Backend:** Python 3.11+ (Render deploys with 3.14.5), Flask 3.1.3 (app factory pattern)
- **Frontend:** Vanilla HTML/CSS/JavaScript (no frameworks, no build step)
- **Security:** Flask-WTF 1.2.1 (CSRF), Flask-Limiter 3.5.0 (rate limiting)
-- **HTTP client:** requests 2.31.0
-- **Excel export:** openpyxl 3.1.2
-- **Production server:** gunicorn 21.2.0
-- **Testing:** pytest 7.4.4
-- **Deployment:** Render.com (free tier, auto-deploy on push)
-- **App version:** 1.1.0 (`config.APP_VERSION`, exposed via `/health`)
+- **HTTP client:** requests 2.33.0
+- **Excel export:** openpyxl 3.1.5
+- **Production server:** gunicorn 22.0.0
+- **Testing:** pytest 9.0.3 (dev deps in `requirements-dev.txt`); Node assertions for `app.js`
+- **Deployment:** Render.com (free tier; auto-deploy gated on CI via `autoDeployTrigger: checksPass`)
+- **App version:** 1.2.0 (`Config.APP_VERSION`, exposed via `/health`)
## Development Setup
@@ -84,20 +100,27 @@ python app.py
**`helpers.py`** — Data processing:
- `flatten_for_csv(data, parent_key, sep, _depth, max_depth)` — Recursively flattens nested dicts (dot notation). Lists are serialized via `json.dumps`. Stops recursing at `max_depth`.
-- `extract_table_data(json_data)` — Extracts tabular rows from various JSON shapes (top-level array, dict containing an array of objects, nested dicts, or a single object).
+- `flatten_rows(rows, max_depth)` — Flattens every row and accumulates column names in one pass; returns `(rows, sorted_columns)`. Replaces flatten-then-`get_all_columns`.
+- `extract_table_data(json_data, _depth, max_depth)` — Extracts tabular rows from various JSON shapes (top-level array, dict containing an array of objects, nested dicts, or a single object). Depth-capped like `flatten_for_csv`.
+- `sanitize_cell(value)` / `serialize_cell_value(value)` / `is_formula_trigger(value)` — Formula-injection defenses for **spreadsheet formats only** (CSV/TSV/XLSX). JSONL and Markdown exports must not use them.
+- `preview_truncate(row)` — Builds a capped **copy** of a preview row (long strings, wide nested objects/arrays). Never mutates the source, so exports stay full-fidelity.
+- `format_size(num_bytes)` — Renders a byte count in the largest non-zero unit. Lives here, not in `app.py`, so `routes.py` can use it without importing `app.py` (that cycle broke every routes-first import).
- `get_all_columns(data)` — Returns sorted unique column names across rows.
- `parse_jsonl(text)` — Parses JSON Lines (one JSON value per non-empty line), raising `ValueError` with line numbers on errors.
-- `find_candidate_arrays(json_data)` — Discovers arrays of objects with their `path`, `length`, and first 5 `sample_keys` (used for the multi-array selector modal).
- `extract_by_path(json_data, path)` — Navigates JSON by dot-notation path (`(root)` returns the document itself).
**`routes.py`** — Flask Blueprint (`bp`):
- `GET /` — Serves `index.html`.
-- `GET /health` — Returns `{"status": "ok", "version": APP_VERSION}` for monitoring.
-- `POST /process` — Parses JSON/JSONL from file/paste/API, returns preview + full CSV-ready data. Returns `{"needs_selection": true, "candidates": [...]}` if multiple arrays found and no `json_path` was provided. Rate-limited via `RATE_LIMIT_PROCESS`.
-- `POST /export-csv` — Server-side CSV generation (fallback). Rate-limited via `RATE_LIMIT_EXPORT`.
-- `POST /export-xlsx` — Server-side Excel generation via openpyxl. Rate-limited via `RATE_LIMIT_EXPORT`.
+- `GET /health` — Returns `{"status": "ok", "version": APP_VERSION}` for monitoring (`version` omitted when `HEALTH_REVEAL_VERSION=0`).
+- `GET /health/live` — Liveness. Checks nothing on purpose, so a failing dependency cannot cause a restart loop.
+- `GET /health/ready` — Readiness. 200, or 503 with a `checks` map when the limiter storage or the Excel writer is unusable.
+- `POST /process` — Parses JSON/JSONL from file/paste/API, returns preview + full CSV-ready data. Returns `{"needs_selection": true, "raw_json": ...}` when no `json_path` was provided, so the client can render the JSON tree picker. Rate-limited via `RATE_LIMIT_PROCESS`. Body is assembled by `_load_input()` and `_select_table_data()`.
+- `POST /export-csv` — Generator-streamed CSV, deliberately **uncapped**. Rate-limited via `RATE_LIMIT_EXPORT`.
+- `POST /export-xlsx` — Server-side Excel via openpyxl, capped by `MAX_EXPORT_CELLS` (400 when exceeded — never truncated). Writes no OS temp files. Rate-limited via `RATE_LIMIT_EXPORT`.
-API-fetch specifics: `requests.get` is called with `stream=True`, `allow_redirects=False`, and a streaming size cap. Errors are logged but the user-facing message is a generic `"API request failed"` to avoid leaking internal hostnames.
+The `/process` success payload is `{success, columns, preview, preview_limit, total_rows, total_cells, max_export_cells, csv_data, csv_columns}`.
+
+API-fetch specifics: `requests.get` is called with `stream=True`, `allow_redirects=False`, and a streaming size cap. The outbound header **name** is checked against an allowlist. Failures log a fixed string with no interpolation — the exception text contains the full URL, and a token can ride in the query string, fragment, userinfo *or* path.
### Frontend
@@ -106,11 +129,12 @@ API-fetch specifics: `requests.get` is called with `stream=True`, `allow_redirec
**`static/js/app.js`** — Vanilla JavaScript:
- CSRF token management (meta tag → FormData / X-CSRFToken header)
- Tab switching, file drag-drop, auth method selection, format toggle (JSON/JSONL)
-- Client-side column sorting (click headers, asc/desc toggle)
-- Client-side CSV/TSV export (no server round-trip needed)
-- Server-side Excel export via `/export-xlsx`
+- Client-side column sorting (click headers, asc/desc toggle), row filtering, "load more" pagination and column visibility toggles
+- Client-side CSV, TSV, JSONL and Markdown export (no server round-trip needed)
+- Server-side Excel export via `/export-xlsx`, greyed out ahead of time when `total_cells > max_export_cells`
- Theme detection (`prefers-color-scheme`) with localStorage override
-- Path selector modal when multiple candidate arrays are detected
+- Lazily-built JSON tree picker modal for choosing which node becomes the table, with `#path=` deep links
+- In-page About modal (no `alert()`), and a keyboard-accessible export dropdown
**`templates/index.html`** — HTML structure only. References external CSS/JS via `url_for('static', ...)`. Includes CSRF meta tag, theme toggle button, format selector, export dropdown, and path selector modal. No inline scripts or styles (CSP enforced).
@@ -118,10 +142,11 @@ API-fetch specifics: `requests.get` is called with `stream=True`, `allow_redirec
1. User provides JSON/JSONL (file / paste / API URL with optional auth).
2. Server validates input (SSRF check for API URLs, UTF-8 decoding, JSON/JSONL parsing, size caps).
-3. If multiple candidate arrays found and no `json_path` is supplied, server returns the candidates so the UI can prompt the user to pick one.
-4. Server returns `preview` (first `PREVIEW_ROW_LIMIT` rows) plus full `csv_data` / `csv_columns`.
-5. Frontend renders the sortable preview table; nested objects render as mini tables.
-6. Export: CSV/TSV generated client-side instantly; Excel via the server endpoint.
+3. If no `json_path` is supplied, the server returns `raw_json` so the UI can render a tree picker and let the user choose a node.
+4. Server returns `preview` (first `PREVIEW_ROW_LIMIT` rows, as a truncated **copy**) plus full-fidelity `csv_data` / `csv_columns` and the export budget.
+5. Frontend renders the sortable preview table; nested objects render as mini tables, with render caps.
+6. Response bodies over `GZIP_MIN_SIZE` are gzipped.
+7. Export: CSV/TSV/JSONL/Markdown generated client-side instantly; Excel via the server endpoint.
## Configuration
@@ -129,18 +154,31 @@ All settings live in `config.py`, configurable via environment variables:
| Setting | Env Var | Default | Description |
|---------|---------|---------|-------------|
-| Secret key | `SECRET_KEY` | `dev-secret-key-change-in-production` | Flask/CSRF secret (change in production) |
+| Secret key | `SECRET_KEY` | `dev-secret-key-change-in-production` | Flask/CSRF secret. The app **refuses to start** on the default when `APP_ENV=production` |
+| Production signal | `APP_ENV` | unset | `production` is the single canonical signal (fail-fast, `Secure` cookie, topology guard). No alias is accepted |
| Max upload size | `MAX_UPLOAD_SIZE` | 10 MB | Request body limit |
| Preview rows | `PREVIEW_ROW_LIMIT` | 25 | Rows shown in preview table |
-| API timeout | `API_FETCH_TIMEOUT` | 30s | Timeout for external API requests |
+| API timeout | `API_FETCH_TIMEOUT` | 30s | Timeout for external API requests. Must stay below gunicorn's `--timeout` |
| API max response | `API_FETCH_MAX_RESPONSE` | 10 MB | Max size for streamed API responses |
-| Flatten depth | `FLATTEN_MAX_DEPTH` | 10 | Max recursion depth for CSV flattening |
+| API ports | `API_ALLOWED_PORTS` | `80,443,8443` | Ports API fetch may connect to. Empty disables the check |
+| DNS wait | `API_DNS_TIMEOUT` | 3s | Bounds how long a **request** waits, not the lookup |
+| DNS workers | `API_DNS_MAX_WORKERS` | 4 | Concurrent lookups; this is the worker-starvation fix |
+| DNS admission | `API_DNS_ADMISSION_TIMEOUT` | 1s | Wait for a permit before rejecting fast |
+| Flatten depth | `FLATTEN_MAX_DEPTH` | 10 | Max recursion depth for flattening and extraction |
+| Excel budget | `MAX_EXPORT_CELLS` | 250000 | XLSX-only cap in cells (`rows × columns`). `0` disables. CSV/TSV stay uncapped |
+| Static cache | `STATIC_MAX_AGE` | 86400 | `Cache-Control` max-age for static assets (URLs carry `?v=APP_VERSION`) |
+| gzip threshold | `GZIP_MIN_SIZE` | 1024 | Smallest body worth compressing |
+| Health version | `HEALTH_REVEAL_VERSION` | on | `0` omits `version` from the health endpoints |
+| Trust proxy | `TRUST_PROXY` | off | `1` installs `ProxyFix` for exactly one hop |
+| Rate limit storage | `RATELIMIT_STORAGE_URI` | `memory://` | Counters are **process-local**; shared storage is required above 1 worker × 1 instance |
+| Workers | `WEB_CONCURRENCY` | 1 | Single source of truth; start commands pass `--workers "$WEB_CONCURRENCY"`. Required under `APP_ENV=production` |
+| Replicas | `APP_REPLICAS` | 1 | Mirrors `render.yaml`'s `numInstances`. Required under `APP_ENV=production` |
| Rate limit (default) | `RATE_LIMIT_DEFAULT` | 120/minute | Global default rate limit |
| Rate limit (process) | `RATE_LIMIT_PROCESS` | 30/minute | Rate limit on `/process` |
| Rate limit (export) | `RATE_LIMIT_EXPORT` | 60/minute | Rate limit on export endpoints |
| Debug mode | `FLASK_DEBUG` | off | Enable Flask debug mode |
-Rate-limiter storage is in-memory (`RATELIMIT_STORAGE_URI = 'memory://'`); switch to Redis if running multiple workers and you want shared counters.
+`.env.example` lists every variable with its default. Rate-limiter storage defaults to `memory://`, whose counters are **process-local** — the effective limit is multiplied by `workers × replicas`, so any deployment above one worker and one instance must set a shared `RATELIMIT_STORAGE_URI` (install `requirements-redis.txt`). Under `APP_ENV=production` the app refuses to start otherwise.
## Security
@@ -154,42 +192,63 @@ Rate-limiter storage is in-memory (`RATELIMIT_STORAGE_URI = 'memory://'`); switc
## Testing
-82 tests using pytest:
+The passing command is the criterion, not a test count:
```bash
-python -m pytest tests/ -v
+python -m pytest tests/ -v # or: make test
```
Test files:
-- `tests/test_helpers.py` (31 tests) — `flatten_for_csv`, `extract_table_data`, `get_all_columns`, `parse_jsonl`, `find_candidate_arrays`, `extract_by_path`.
-- `tests/test_security.py` (16 tests) — URL validation with mocked DNS, private/loopback/link-local IP blocking, scheme checks.
-- `tests/test_routes.py` (35 tests) — All route integration tests, security headers, JSONL, path selection, API-fetch SSRF/size/timeout/error-leak coverage, CSV/Excel export edge cases.
+- `tests/test_helpers.py` — `flatten_for_csv`, `flatten_rows`, `extract_table_data` (incl. the depth guard), `get_all_columns`, `parse_jsonl`, `extract_by_path`, `preview_truncate`.
+- `tests/test_security.py` — URL validation with mocked DNS, private/loopback/link-local IP blocking, scheme checks, the port allowlist, and the bounded DNS resolver (admission limit, permit accounting, fork lifecycle, and the *accepted* unbounded teardown).
+- `tests/test_routes.py` — All route integration tests: security headers, JSONL, path selection, exports, formula injection, log hygiene via `caplog`, gzip, the config gates (`APP_ENV`, SECRET_KEY, integer validation), the rate-limit topology guard, and the health split.
+- `tests/js/*.mjs` — Node assertions that load the real `static/js/app.js` in a stubbed DOM (`dom_stub.mjs`) and exercise the client export, render-cap and feature code. No build step and no npm dependencies; run with `make test-js`.
-Fixtures in `tests/conftest.py` provide `app` (with `TESTING=True` and `WTF_CSRF_ENABLED=False`) and `client`.
+Fixtures in `tests/conftest.py` provide `app` (with `TESTING=True` and `WTF_CSRF_ENABLED=False`) and `client`. `tests/test_routes.py` adds a `fresh_config` fixture that reloads `config` under a patched environment, because `Config` holds class attributes evaluated at import time.
## Linting / Formatting
-No linting tools currently configured. Recommended: `ruff` for linting + formatting.
+`ruff`, configured in `pyproject.toml` (target py311, line length 100, Python files only):
+
+```bash
+ruff check . # or: make lint
+ruff format --check .
+ruff format . # or: make format
+```
+
+CI runs lint, format-check, pytest, the Node assertions, and `pip-audit -r requirements.txt`.
## Deployment
### Production Requirements (All Methods)
-- Set `SECRET_KEY` to a random value (never use the dev default).
+- Set `APP_ENV=production`. This is the single canonical production signal.
+- Set `SECRET_KEY` to a random value. With `APP_ENV=production` the app **refuses
+ to start** on the dev default, an empty value, or an unset one.
- Set `FLASK_DEBUG=0`.
-- Use HTTPS (TLS termination via Nginx, cloud provider, or reverse proxy).
+- Declare `WEB_CONCURRENCY` and `APP_REPLICAS`, and derive the start command's
+ `--workers` from `$WEB_CONCURRENCY`. Above one worker or one instance, set a
+ shared `RATELIMIT_STORAGE_URI` (and install `requirements-redis.txt`) — the app
+ refuses to start otherwise, because `memory://` counters are process-local.
+- Set gunicorn's `--timeout` above `API_FETCH_TIMEOUT` (60 vs 30 by default).
+- Use HTTPS (TLS termination via Nginx, cloud provider, or reverse proxy). Set
+ `TRUST_PROXY=1` behind a proxy you control so rate limiting and `Secure`
+ cookies see the real client and scheme.
### Render.com (PaaS)
Configured via `render.yaml` blueprint:
- Runtime: Python 3.14.5
- Build: `pip install -r requirements.txt`
-- Start: `gunicorn "app:create_app()" --bind 0.0.0.0:$PORT`
-- `SECRET_KEY` is generated by Render; auto-deploy on push; free tier; no persistent storage.
+- Start: `gunicorn "app:create_app()" --bind 0.0.0.0:$PORT --workers "$WEB_CONCURRENCY" --timeout 60`
+- `SECRET_KEY` is generated by Render; `APP_ENV=production`, `WEB_CONCURRENCY=1`
+ and `APP_REPLICAS=1` are declared in the blueprint; `numInstances: 1`.
+- `autoDeployTrigger: checksPass` — a push with failing or missing CI checks does
+ not deploy. Free tier; no persistent storage.
### Own Server (Gunicorn + Systemd + Nginx)
-1. **Gunicorn** runs the app: `gunicorn "app:create_app()" --bind 127.0.0.1:8000 --workers 4`
+1. **Gunicorn** runs the app: `gunicorn "app:create_app()" --bind 127.0.0.1:8000 --workers "$WEB_CONCURRENCY" --timeout 60`
2. **Systemd** manages the process (auto-restart, boot start) — see `README.md` for the unit file.
3. **Nginx** reverse-proxies and handles TLS termination; can also serve `/static/` directly.
@@ -211,7 +270,7 @@ Railway.app and Fly.io are also supported — see `README.md` for CLI commands.
## Common Tasks
### Adding a new route
-Add the handler to `routes.py` on the `bp` Blueprint. Apply `@limiter.limit()` if needed (use a lambda reading from `current_app.config` to keep the limit configurable). Follow existing patterns: `jsonify()` for responses, try/except around external calls, generic error messages with proper HTTP status codes.
+Add the handler to `routes.py` on the `bp` Blueprint. Apply `@limiter.limit()` if needed (use a lambda reading from `current_app.config` to keep the limit configurable). Follow existing patterns: `jsonify()` for responses, try/except around external calls, generic error messages with proper HTTP status codes. Re-raise `HTTPException` before the generic `except Exception` so Flask's JSON error handlers still run. If the route returns payload data, add its endpoint to `security.NO_STORE_ENDPOINTS`.
### Modifying the UI
- **CSS:** Edit `static/css/style.css`. Use existing CSS custom properties. Add light-theme overrides under `:root.light` if needed.
@@ -225,12 +284,40 @@ Add the handler to `routes.py` on the `bp` Blueprint. Apply `@limiter.limit()` i
4. Add the JS visibility toggle in `app.js` (auth-method switching section).
### Adding a new export format
-1. Client-side: add a handler in `app.js` (extend `downloadDelimited()` or add a new function).
-2. Server-side: add a route in `routes.py` and a button in the export dropdown.
-3. Add any new dependency to `requirements.txt` with a pinned version.
+1. **Decide the sanitization policy first.** Spreadsheet-compatible formats
+ (anything a spreadsheet will open and evaluate) must route every cell through
+ `helpers.sanitize_cell`, or pin the cell type if the format has one. Lossless
+ or plain-text formats (JSONL, Markdown) must **not** — a quote prefix would
+ corrupt the data while protecting nothing. See MEMORY.md, 2026-08-21.
+2. Client-side: add a `build*Chunks()` builder in `app.js` and dispatch it from
+ the export-dropdown handler. Return chunks, not one giant string.
+3. Server-side (if needed): add a route in `routes.py`. Stream it with a
+ generator unless the format genuinely cannot be streamed; if it cannot, give
+ it a measured budget the way `MAX_EXPORT_CELLS` works, and never truncate —
+ refuse with a 400 and advertise the limit from `/process`.
+4. Add the button to the export dropdown in `index.html` with
+ `role="menuitem"`.
+5. Add assertions to `tests/js/test_features.mjs` covering the sanitization
+ decision explicitly, in both directions.
+6. Add any new dependency to `requirements.txt` with an exact pin.
### Changing configuration defaults
-Edit `config.py`. All values read from `os.environ.get()` with defaults.
+Edit `config.py`. Read integers through `env_int()` (and integer lists through
+`env_int_set()`) so a typo names the variable instead of raising a bare
+`ValueError` at import. Add the variable to `.env.example`, the table above, and
+the README table. Gate any production-only behavior on `is_production()` — never
+on `not DEBUG`, and never on a second env-var spelling.
### Adding a dependency
-Add to `requirements.txt` with a pinned version (e.g., `package==1.2.3`).
+Add to `requirements.txt` with an exact pin (e.g. `package==1.2.3`). Test and
+lint tooling goes in `requirements-dev.txt`; anything only a specific deployment
+shape needs goes in its own file (see `requirements-redis.txt`). Re-run
+`pip-audit -r requirements.txt` and the full suite in the same commit — the
+convention is a dedicated bump commit, not a bump ridden along with a feature.
+
+### Re-deriving the Excel export budget
+`MAX_EXPORT_CELLS` is a measured number, not a chosen one. Follow the method in
+`docs/export-budget-v1.2.md` (fresh process per measured request, `ru_maxrss`
+converted for the platform, delta across two runs), re-fit against the narrowest
+aspect ratio you care about, and re-run a confirming point at the value you
+intend to ship. Record the new data in that file.
diff --git a/MEMORY.md b/MEMORY.md
index ed82583..8e1154c 100644
--- a/MEMORY.md
+++ b/MEMORY.md
@@ -31,7 +31,7 @@ Keep entries short — if it grows past ~10 lines, it probably belongs in `READM
**What:** The app must never persist user-submitted JSON to disk, database, or any external system. The only writes are stdout logs (and those deliberately omit payloads).
**Why:** Designed as an internal tool for handling potentially sensitive payloads (API responses, exports). The Render free tier deliberately has no persistent disk to enforce this physically.
-**How to apply:** Reject any change that adds a DB driver, file write of payload bytes, third-party analytics, or request-body logging. `logger.warning("API request failed: %s", e)` is fine; `logger.warning("payload was: %s", body)` is not.
+**How to apply:** Reject any change that adds a DB driver, file write of payload bytes, third-party analytics, or request-body logging. Log **fixed strings**: `logger.warning('API request failed')` is fine; interpolating the payload, the URL, or the exception is not — `requests`' exception text embeds the full URL, which can carry a token (see the 2026-08-21 log-hygiene entry).
### 2026-05-12 — Gunicorn must call the factory, not a module-level `app` (area: deploy)
@@ -63,10 +63,10 @@ Keep entries short — if it grows past ~10 lines, it probably belongs in `READM
**Why:** All state-changing routes accept browser form posts, so CSRF is mandatory. Disabling it in tests keeps fixtures simple — production behavior is exercised manually and via the security headers test.
**How to apply:** When adding a route that mutates state or returns sensitive data, it inherits CSRF protection automatically. Don't add `@csrf.exempt` without justification. When testing CSRF behavior, do so in a dedicated test that flips `WTF_CSRF_ENABLED` back on.
-### 2026-05-12 — Multi-array JSON triggers a path-selector handshake (area: backend / ux)
+### 2026-05-12 — Unselected JSON triggers the tree-picker handshake (area: backend / ux)
-**What:** `/process` returns `{"needs_selection": true, "candidates": [...]}` (HTTP 200) when `find_candidate_arrays` reports more than one array of objects in the payload. The frontend opens a modal; the user picks; the request is re-submitted with `json_path` set to the chosen dotted path.
-**Why:** The original heuristic (`extract_table_data`) silently picked the first array it found, which surfaced the wrong data for nested API responses. Returning candidates is more honest than guessing.
+**What:** `/process` returns `{"needs_selection": true, "raw_json": }` (HTTP 200) whenever no `json_path` was supplied. The frontend renders a JSON **tree picker** over `raw_json`; the user clicks any array or object node; the request is re-submitted with `json_path` set to the chosen dotted path.
+**Why:** The original heuristic (`extract_table_data`) silently picked the first array it found, which surfaced the wrong data for nested API responses. Handing the client the document and letting the user point at a node is more honest than guessing — and unlike the earlier `candidates` list it can reach any level, not just arrays of objects.
**How to apply:** Don't "fix" the heuristic by being smarter — the selection prompt *is* the fix. The sentinel `'(root)'` is used when the top-level value is itself a list.
### 2026-05-12 — `flatten_for_csv` has a recursion-depth cap (area: backend)
@@ -101,10 +101,103 @@ Keep entries short — if it grows past ~10 lines, it probably belongs in `READM
### 2026-05-12 — Pinned dependencies are deliberate (area: deploy)
-**What:** `requirements.txt` uses exact `==` pins (Flask 3.0.0, requests 2.31.0, gunicorn 21.2.0, Flask-WTF 1.2.1, Flask-Limiter 3.5.0, pytest 7.4.4, openpyxl 3.1.2).
+**What:** `requirements.txt` uses exact `==` pins (Flask 3.1.3, requests 2.33.0, gunicorn 22.0.0, Flask-WTF 1.2.1, Flask-Limiter 3.5.0, openpyxl 3.1.5). Test tooling lives in `requirements-dev.txt`, and the optional Redis client in `requirements-redis.txt`.
**Why:** Render auto-deploys on push. Loose pins + auto-deploy = surprise breakage. Exact pins keep deploys reproducible and make security audits possible.
**How to apply:** Bump versions intentionally in a dedicated commit, run the full test suite, and verify the Render build before merging. Don't bump on a feature commit "while we're in here".
+### 2026-08-21 — Spreadsheet exports are formula-sanitized; JSONL and Markdown are not (area: security)
+
+**What:** Values starting with `=`, `+`, `-`, `@`, tab, CR or LF are formula triggers (CWE-1236). CSV/TSV prefix them with a single quote; XLSX instead pins the cell's `data_type` to `'s'`, because openpyxl serializes a leading `=` as a *formula cell* and Excel then runs it without the CSV warning. JSONL and Markdown exports are deliberately exempt.
+**Why:** The tool's whole job is turning untrusted API/file JSON into spreadsheets, so an attacker who controls a cell controls the exported file's formulas. The exemptions are not oversights: JSON carries types and nothing evaluates it, so a quote prefix would corrupt data while protecting nothing; Markdown does not evaluate `=` either, but an unescaped pipe or newline breaks the table, so it gets Markdown-specific escaping.
+**How to apply:** Any new **spreadsheet-compatible** export must route cells through `helpers.sanitize_cell` (or pin the data type, for typed formats). Any new **non-spreadsheet** format (JSONL, Markdown, ...) must not. There are four sanitized paths today — two server routes plus the client CSV and TSV builders — and `tests/js/test_export_sanitize.mjs` exists so none of them can regress silently.
+
+### 2026-08-21 — The API-fetch failure log is a fixed string (area: security)
+
+**What:** `logger.warning('API request failed')` — no interpolation, ever.
+**Why:** `requests`' exception text embeds the full URL. With query-param auth the token rides in that URL, so the old `'API request failed: %s'` wrote secrets to stdout. Query strings, fragments, userinfo **and paths** can all carry tokens, so a redaction helper that preserves the path is not sufficient.
+**How to apply:** Never add the URL, the exception, or any request field to a log line on this path. `caplog` tests assert no URL component reaches the logs; keep them passing.
+
+**Owner / source:** security review F3/F9.
+
+### 2026-08-21 — DNS is bounded in *concurrency*, not in execution (area: security / performance)
+
+**What:** Lookups run on a shared, fixed-size `ThreadPoolExecutor` created lazily inside the worker (it records its pid, so a pool inherited across a fork is replaced). The admission permit is taken *before* `submit` and released from the future's **done-callback**, never from the caller's `finally`.
+**Why:** `getaddrinfo` takes no timeout and cannot be cancelled. `API_DNS_TIMEOUT` bounds only how long the *request* waits; the lookup keeps running. Releasing the permit on caller timeout would re-admit work into an already-blocked pool, which is exactly how it saturates under repeated slow-DNS requests. **Teardown is not bounded by anything this code controls** — glibc's defaults are ~5s per nameserver × 2 attempts × every nameserver in `resolv.conf`, so tens of seconds is the realistic worst case.
+**How to apply:** Do **not** describe teardown as bounded in any doc, comment or test. Assert the caller wait and the admission error instead. `options timeout:2 attempts:1` in the container's `resolv.conf` is a best-effort narrowing; a killable subprocess resolver is the only real bound and is not in v1.2.
+
+**Owner / source:** security review F6.1, performance review P7.
+
+### 2026-08-21 — `APP_ENV=production` is the only production signal (area: deploy / security)
+
+**What:** One env var, read through one helper (`config.is_production()`), gating the SECRET_KEY fail-fast, `SESSION_COOKIE_SECURE`, and the rate-limit topology guard.
+**Why:** Two accepted spellings let a deployment satisfy one gate and silently miss another — e.g. passing the secret-key check with `Secure` cookies still off, a live vulnerability produced purely by the inconsistency. And production is never inferred from `not DEBUG`: the documented local run `python app.py` has `DEBUG=False`, so that would block ordinary development.
+**How to apply:** New production-only behavior calls `is_production()`. Never add `PRODUCTION=true`, `ENV=prod`, or any alias — a test asserts `PRODUCTION=true` alone is *not* honored.
+
+**Owner / source:** security review F7/F16.
+
+### 2026-08-21 — `memory://` rate-limit counters multiply by workers × replicas (area: deploy)
+
+**What:** The default storage is process-local, so N workers on M instances enforce N×M times the configured limit. `RATELIMIT_STORAGE_URI` is configurable (it was hardcoded before v1.2). Under `APP_ENV=production` the app refuses to start unless `WEB_CONCURRENCY` and `APP_REPLICAS` are declared, the start command's `--workers` agrees with `WEB_CONCURRENCY`, and storage is shared whenever either count exceeds one.
+**Why:** Defaults of 1 fail *open* — an undeclared 4-worker deployment reads as single-worker, which is exactly where the guard matters most. `WEB_CONCURRENCY` is the single source of truth because gunicorn reads it natively, so the number the app validates cannot drift from the number gunicorn runs.
+**How to apply:** To run more than one worker or instance: install `requirements-redis.txt`, set `RATELIMIT_STORAGE_URI=redis://…`, then raise the counts. Never write a bare `--workers N` into a start command — derive it from `$WEB_CONCURRENCY`. `APP_REPLICAS` must mirror `render.yaml`'s `numInstances`.
+
+**Owner / source:** security review F12, roadmap 2.10.
+
+### 2026-08-21 — API fetch is restricted to ports 80, 443 and 8443 (area: security)
+
+**What:** `API_ALLOWED_PORTS` (default `80,443,8443`), checked before DNS so a rejected URL costs no lookup. An empty value disables the check.
+**Why:** Only the resolved IP was validated, so `http://public.example.com:22` or `:6379` passed and the tool would connect to any port on any public host.
+**How to apply:** Widen the list via env rather than in code, and keep the check ahead of resolution.
+
+**Owner / source:** security review F6.2, decision D5.
+
+### 2026-08-21 — The XLSX export budget is measured, in cells, and on by default (area: performance)
+
+**What:** `MAX_EXPORT_CELLS` (default 250,000) caps Excel exports only. CSV/TSV are generator-streamed and stay uncapped.
+**Why:** openpyxl memory tracks `rows × columns`, not rows — at equal cell counts a narrow, tall sheet costs *more* (84.3 MiB at 50k×3 vs 73.6 at 15k×10), so a row limit says almost nothing about the footprint. The default comes from the measurement in `docs/export-budget-v1.2.md`, not from feel. An unlimited default would leave a High finding unmitigated; uncapped CSV/TSV is what keeps the export contract as wide as the input contract.
+**How to apply:** Re-derive the number whenever the measurement is re-run — do not round it to something tidy. Never truncate an oversized export: `/process` advertises `total_cells`/`max_export_cells` so the client greys Excel out beforehand, and `/export-xlsx` returns 400 for direct callers. And keep exports diskless: openpyxl's `write_only` mode writes worksheet parts to OS temp files, and `SpooledTemporaryFile` is either pointless (its default `max_size=0` never rolls over) or disk-backed.
+
+**Owner / source:** performance review P3, decision D6.
+
+### 2026-08-21 — gunicorn's `--timeout` must exceed `API_FETCH_TIMEOUT` (area: deploy)
+
+**What:** Every documented invocation sets `--timeout 60` against a default `API_FETCH_TIMEOUT` of 30s.
+**Why:** gunicorn's default timeout is also 30s, so a slow API fetch raced the worker kill: the worker was SIGKILLed mid-response and the client saw a 502 instead of the timeout message.
+**How to apply:** If you raise `API_FETCH_TIMEOUT`, raise `--timeout` with it — roughly double is the documented margin.
+
+**Owner / source:** performance review P9.
+
+### 2026-08-21 — The preview is a truncated copy; exports are not (area: backend)
+
+**What:** `helpers.preview_truncate` builds a *new* row capping long strings, nested objects and nested arrays. `table_data` and `csv_data` are never mutated.
+**Why:** Preview rows used to carry full-fidelity nested structures, so a 50k-key object or a 5 MB string cell froze the tab. Truncating in place would have silently corrupted every export.
+**How to apply:** Anything that trims data for display must build a projection. Tests assert the server CSV and XLSX exports still contain the untruncated values — keep them.
+
+**Owner / source:** performance review P2.2/P5.
+
+
+---
+
+### 2026-08-22 — `routes.py` must never import `app.py` (area: backend)
+
+**What:** Shared utilities go in `helpers.py` (or another leaf module), never in `app.py`. `helpers.format_size` is there for exactly this reason.
+**Why:** `create_app()` imports `bp` from `routes.py`, so a module-level `from app import ...` in `routes.py` makes `import routes` re-enter a half-initialized module and raise `ImportError`. It hid for a while because gunicorn's `app:create_app()` imports app-first, which happens to work — only routes-first entry points broke.
+**How to apply:** `TestNoImportCycle` in `tests/test_routes.py` imports each module first in a fresh subprocess and AST-checks that `routes.py` does not import `app`. If you need something from `app.py` in a route, move it down, don't import up.
+
+**Owner / source:** CodeRabbit review on the v1.2.0 PR.
+
+
+---
+
+### 2026-08-22 — A blank element in an integer-list env var is an error, not a default (area: deploy / security)
+
+**What:** `config.env_int_set` rejects `80,,443`, `80,443,` and a lone `,`. Only an *unset* variable selects the default; only a fully empty value disables the check.
+**Why:** `security.validate_url` reads an empty allowlist as "no port restriction". Skipping blank elements meant `API_ALLOWED_PORTS=,` silently removed the outbound port restriction, and `80,,443` silently narrowed it — both from a typo, with no startup error.
+**How to apply:** Any list-valued setting whose empty state weakens a check must fail loudly on a malformed element. Never `if part.strip()` your way past bad input in a security setting.
+
+**Owner / source:** CodeRabbit review on the v1.2.0 PR.
+
+
---
## Conventions for Adding Entries
diff --git a/Makefile b/Makefile
new file mode 100644
index 0000000..da5ebc8
--- /dev/null
+++ b/Makefile
@@ -0,0 +1,57 @@
+# Developer entry points. Everything here is also what CI runs.
+
+VENV ?= venv
+PY ?= $(VENV)/bin/python
+PIP ?= $(VENV)/bin/pip
+
+.PHONY: help venv install test test-js lint format audit coverage run check clean
+
+help:
+ @echo "make install - create the venv and install dev dependencies"
+ @echo "make test - run the Python test suite"
+ @echo "make test-js - run the Node assertions for static/js/app.js"
+ @echo "make lint - ruff check + ruff format --check"
+ @echo "make format - ruff format (rewrites files)"
+ @echo "make audit - pip-audit against the runtime requirements"
+ @echo "make coverage - test suite with a coverage report"
+ @echo "make run - start the development server on :5000"
+ @echo "make check - lint + test + test-js + audit (what CI runs)"
+
+venv:
+ test -d $(VENV) || python3 -m venv $(VENV)
+
+install: venv
+ $(PIP) install --upgrade pip
+ $(PIP) install -r requirements-dev.txt
+
+test:
+ $(PY) -m pytest tests/ -v
+
+test-js:
+ node tests/js/test_export_sanitize.mjs
+ node tests/js/test_render_caps.mjs
+ node tests/js/test_features.mjs
+
+lint:
+ $(VENV)/bin/ruff check .
+ $(VENV)/bin/ruff format --check .
+
+format:
+ $(VENV)/bin/ruff format .
+ $(VENV)/bin/ruff check . --fix
+
+audit:
+ $(VENV)/bin/pip-audit -r requirements.txt
+
+coverage:
+ $(VENV)/bin/coverage run -m pytest tests/
+ $(VENV)/bin/coverage report -m
+
+run:
+ $(PY) app.py
+
+check: lint test test-js audit
+
+clean:
+ find . -type d -name __pycache__ -prune -exec rm -rf {} +
+ rm -rf .pytest_cache .coverage .ruff_cache
diff --git a/README.md b/README.md
index 3a3b1ac..eccfb76 100644
--- a/README.md
+++ b/README.md
@@ -4,7 +4,7 @@ A lightweight web tool to convert JSON data into viewable tables with CSV export


-
+
## Features
@@ -20,10 +20,11 @@ A lightweight web tool to convert JSON data into viewable tables with CSV export
- Query Parameter Token
- **Data Processing**
- - Handles nested JSON objects
- - Displays nested data as expandable tables
- - Preview first 25 rows
- - Export ALL rows to CSV
+ - Handles nested JSON objects and JSON Lines
+ - Displays nested data as expandable tables (with render caps, so a huge cell cannot freeze the tab)
+ - Preview the first `PREVIEW_ROW_LIMIT` rows, then "Load next 500" / "Load all"
+ - Filter rows, sort columns, hide columns
+ - Export **all** rows to CSV, TSV, JSONL, Markdown or Excel
- **Privacy First**
- No data storage - everything processed in-memory
@@ -111,7 +112,11 @@ git push -u origin main
- **Branch**: `main`
- **Runtime**: `Python 3`
- **Build Command**: `pip install -r requirements.txt`
- - **Start Command**: `gunicorn "app:create_app()" --bind 0.0.0.0:$PORT`
+ - **Start Command**: `gunicorn "app:create_app()" --bind 0.0.0.0:$PORT --workers "$WEB_CONCURRENCY" --timeout 60`
+ - **Environment**: `SECRET_KEY` (click *Generate* — the app **refuses to start**
+ under `APP_ENV=production` with the development default), `APP_ENV=production`,
+ `WEB_CONCURRENCY=1`, `APP_REPLICAS=1`
+ (see [Deployment topology](#deployment-topology-and-rate-limiting))
4. Select **Free** plan
5. Click **"Create Web Service"**
@@ -181,12 +186,31 @@ python -m venv venv
source venv/bin/activate
pip install -r requirements.txt
-# Set production environment variables
-export SECRET_KEY="your-random-secret-key-here"
+# Set production environment variables.
+#
+# SECRET_KEY must be a random value you generate, not a literal copied from this
+# README. The startup gate rejects the dev default and an empty value, but it
+# cannot tell a real secret from a memorable one someone pasted:
+#
+# python -c "import secrets; print(secrets.token_urlsafe(48))"
+#
+# Every SECRET_KEY placeholder below means "the output of that command", kept
+# out of the shell history and out of version control.
+export SECRET_KEY="$(python -c 'import secrets; print(secrets.token_urlsafe(48))')"
export FLASK_DEBUG=0
-
-# Run with gunicorn
-gunicorn "app:create_app()" --bind 0.0.0.0:8000 --workers 4
+export APP_ENV=production
+
+# Deployment topology. memory:// rate-limit counters are process-local, so the
+# effective limit is multiplied by workers x replicas. One worker and one
+# instance is the default; see "Deployment topology and rate limiting" below
+# before raising either.
+export WEB_CONCURRENCY=1
+export APP_REPLICAS=1
+
+# Run with gunicorn. --workers comes from WEB_CONCURRENCY so the running count
+# and the declared count cannot drift, and --timeout stays above
+# API_FETCH_TIMEOUT (default 30s).
+gunicorn "app:create_app()" --bind 0.0.0.0:8000 --workers "$WEB_CONCURRENCY" --timeout 60
```
#### Systemd Service (Auto-Start on Boot)
@@ -202,9 +226,12 @@ After=network.target
User=www-data
Group=www-data
WorkingDirectory=/opt/json-table-tool
-Environment="SECRET_KEY=your-random-secret-key-here"
+Environment="SECRET_KEY=