From 1432c8ea1069d105f5b080aaea554045cf69c7e8 Mon Sep 17 00:00:00 2001 From: Anmol Baranwal Date: Tue, 22 Sep 2026 17:25:44 +0530 Subject: [PATCH 1/2] feat(release-bot): announce releases and videos to Discord Posts new GitHub releases and YouTube videos into the CopilotKit, AG-UI and OpenBot channels. Runs as a Railway cron job. Release notes are not forwarded. Each release is paired with the commits since the previous one and rewritten into a few lines about what changed, with outside contributors credited by name. There is no state file. Every announcement ends with its source URL, so the bot reads its own recent messages to find what it has already posted and announces only what is newer, oldest first. Running twice posts nothing the second time, and a crash mid-batch cannot cause a repeat. Sources are configured in src/sources.ts, one entry per repository, with the tag filter and the reason for it written together. Adding a repo is an entry there and a channel id in the environment. Covered by 75 tests across the watermark, the 2000-character budget, the GitHub and YouTube parsing, and the summarizer's skip handling. --- apps/release-bot/.env.example | 39 ++ apps/release-bot/Dockerfile | 54 +++ apps/release-bot/README.md | 276 +++++++++++++ apps/release-bot/package.json | 25 ++ apps/release-bot/railway.toml | 25 ++ .../release-bot/src/__tests__/discord.test.ts | 82 ++++ apps/release-bot/src/__tests__/github.test.ts | 93 +++++ .../release-bot/src/__tests__/sources.test.ts | 82 ++++ .../src/__tests__/summarize.test.ts | 48 +++ .../src/__tests__/watermark.test.ts | 85 ++++ .../release-bot/src/__tests__/youtube.test.ts | 47 +++ apps/release-bot/src/discord.ts | 330 ++++++++++++++++ apps/release-bot/src/github.ts | 368 ++++++++++++++++++ apps/release-bot/src/http.ts | 20 + apps/release-bot/src/index.ts | 198 ++++++++++ apps/release-bot/src/sources.ts | 93 +++++ apps/release-bot/src/summarize.ts | 200 ++++++++++ apps/release-bot/src/watermark.ts | 63 +++ apps/release-bot/src/youtube.ts | 70 ++++ apps/release-bot/tsconfig.json | 18 + apps/release-bot/vitest.config.ts | 13 + pnpm-lock.yaml | 15 + 22 files changed, 2244 insertions(+) create mode 100644 apps/release-bot/.env.example create mode 100644 apps/release-bot/Dockerfile create mode 100644 apps/release-bot/README.md create mode 100644 apps/release-bot/package.json create mode 100644 apps/release-bot/railway.toml create mode 100644 apps/release-bot/src/__tests__/discord.test.ts create mode 100644 apps/release-bot/src/__tests__/github.test.ts create mode 100644 apps/release-bot/src/__tests__/sources.test.ts create mode 100644 apps/release-bot/src/__tests__/summarize.test.ts create mode 100644 apps/release-bot/src/__tests__/watermark.test.ts create mode 100644 apps/release-bot/src/__tests__/youtube.test.ts create mode 100644 apps/release-bot/src/discord.ts create mode 100644 apps/release-bot/src/github.ts create mode 100644 apps/release-bot/src/http.ts create mode 100644 apps/release-bot/src/index.ts create mode 100644 apps/release-bot/src/sources.ts create mode 100644 apps/release-bot/src/summarize.ts create mode 100644 apps/release-bot/src/watermark.ts create mode 100644 apps/release-bot/src/youtube.ts create mode 100644 apps/release-bot/tsconfig.json create mode 100644 apps/release-bot/vitest.config.ts diff --git a/apps/release-bot/.env.example b/apps/release-bot/.env.example new file mode 100644 index 0000000..cd9f068 --- /dev/null +++ b/apps/release-bot/.env.example @@ -0,0 +1,39 @@ +# Required. Discord application (dev portal -> your app -> Bot -> Reset Token). +# The bot needs Send Messages, Embed Links, View Channel and Read Message History: +# reading the channel is how it knows what it has already announced. +DISCORD_BOT_TOKEN= + +# Required. Release notes are never posted unsummarized, so without this nothing +# is announced. A failed call stops that source for the run and is retried next. +OPENAI_API_KEY= + +# Required, and needs read:org. This is how the bot tells the team from outside +# contributors. A token without org visibility does not fail, it returns an empty +# member list, so the bot credits nobody rather than thanking colleagues as if +# they were community. A classic token with only read:org ticked is enough. +GITHUB_TOKEN= + +# Where each source posts. A source with no channel is skipped, so these can be +# filled in one at a time. +AGUI_CHANNEL_ID= +CPK_CHANNEL_ID= +# Optional: OpenBot posts to CPK_CHANNEL_ID unless given its own channel. +OPENBOT_CHANNEL_ID= +YOUTUBE_CHANNEL_DISCORD_ID= + +# The CopilotKit YouTube channel. +YOUTUBE_CHANNEL_ID=UCbC2DjohfqaUcXK_XmXBVUg + +# Optional. Unset means announcements are silent, which is the default. +AGUI_PING_ROLE_ID= +CPK_PING_ROLE_ID= +# Optional: falls back to CPK_PING_ROLE_ID. +OPENBOT_PING_ROLE_ID= +YOUTUBE_PING_ROLE_ID= + +# Optional. Comma-separated logins treated as team, on top of org membership. +# Extends the team list; it cannot substitute for org visibility. +CORE_LOGINS= + +# Optional. Defaults to gpt-5.4. +OPENAI_MODEL= diff --git a/apps/release-bot/Dockerfile b/apps/release-bot/Dockerfile new file mode 100644 index 0000000..62bf6cd --- /dev/null +++ b/apps/release-bot/Dockerfile @@ -0,0 +1,54 @@ +# ── Stage 1: prune the monorepo to only what @copilotkit/outpost-release-bot needs ── +FROM node:20-alpine AS pruner +RUN apk add --no-cache libc6-compat +WORKDIR /app + +# No pnpm in this stage: pruning runs on turbo alone. +# Pinned in step with the root devDependency, so the pruner cannot drift. +RUN npm install -g turbo@2.3.0 + +COPY . . +RUN turbo prune @copilotkit/outpost-release-bot --docker + +# ── Stage 2: install dependencies and build ────────────────────────────────── +FROM node:20-alpine AS installer +RUN apk add --no-cache libc6-compat +WORKDIR /app + +# Matches the root package.json's packageManager field; corepack honours that +# field, so a different pin here would be silently ignored at best. +RUN corepack enable && corepack prepare pnpm@10.33.4 --activate + +COPY --from=pruner /app/out/json/ . +RUN pnpm install --frozen-lockfile + +COPY --from=pruner /app/out/full/ . +COPY --from=pruner /app/tsconfig.json ./tsconfig.json +RUN pnpm turbo run build --filter=@copilotkit/outpost-release-bot + +# ── Stage 3: production image ──────────────────────────────────────────────── +# No Prisma, no database, no health server: this service talks to GitHub, YouTube, +# OpenAI and Discord over HTTPS, announces what is new, and exits. +FROM node:20-alpine AS runner + +ENV NODE_ENV=production + +RUN addgroup --system --gid 1001 outpost && \ + adduser --system --uid 1001 outpost + +WORKDIR /app + +COPY --from=installer --chown=outpost:outpost /app/node_modules ./node_modules +COPY --from=installer --chown=outpost:outpost /app/apps/release-bot/dist ./apps/release-bot/dist +COPY --from=installer --chown=outpost:outpost /app/apps/release-bot/package.json ./apps/release-bot/package.json +# pnpm's isolated layout keeps an app's own deps here, symlinked into the root +# store, so this has to be copied for the first runtime dependency added not to +# fail only in production. The destination is named explicitly: a COPY whose +# source is a directory copies its *contents*, so a bare destination would +# scatter the packages across apps/release-bot/ instead. +COPY --from=installer --chown=outpost:outpost /app/apps/release-bot/node_modules ./apps/release-bot/node_modules +COPY --from=installer --chown=outpost:outpost /app/package.json ./package.json + +USER outpost + +CMD ["node", "apps/release-bot/dist/index.js"] diff --git a/apps/release-bot/README.md b/apps/release-bot/README.md new file mode 100644 index 0000000..b2f070b --- /dev/null +++ b/apps/release-bot/README.md @@ -0,0 +1,276 @@ +# @copilotkit/outpost-release-bot + +Announces new releases and new YouTube videos in the CopilotKit, AG-UI and +OpenBot Discord channels. A scheduled job: it works out what shipped since its last announcement, +writes each one up, posts, and finishes. + +Forwarding release notes verbatim does not work, which is the reason this app +exists rather than a GitHub webhook. CopilotKit's notes are often a single +sentence (`v1.72.0` was 156 characters) and AG-UI's run to thousands of characters +of package tables, well past Discord's 2000-character limit. Neither is something +a reader can skim. So every release is paired with the commits since the previous +release, rewritten into a few lines about what a developer can now do, and +credited to whoever outside the team worked on it. + +## How it works + +One pass over three sources, each independent of the others: + +``` +┌──────────────────────────────────────────────────────────────────────┐ +│ 1. read the channel what has this bot already announced here? │ +│ -> the newest announcement is a watermark │ +├──────────────────────────────────────────────────────────────────────┤ +│ 2. list the source GitHub releases / the YouTube feed │ +│ -> drop drafts, prereleases, other tags │ +│ -> keep only what shipped after the mark │ +├──────────────────────────────────────────────────────────────────────┤ +│ 3. gather context commits since the previous release, and │ +│ (releases only) their authors │ +├──────────────────────────────────────────────────────────────────────┤ +│ 4. write it up OpenAI, with the notes and the commit list │ +│ (releases only) -> a few lines, or SKIP if nothing shipped │ +├──────────────────────────────────────────────────────────────────────┤ +│ 5. post plain text, source URL last, mentions off │ +└──────────────────────────────────────────────────────────────────────┘ +``` + +### The files + +``` +src/ +├── sources.ts what is watched: repo, channel, which tags, how the title reads +├── index.ts runs one pass over the sources and decides what to post +├── watermark.ts given a channel's history, which items are still pending +├── github.ts releases, the commits between them, and who is on the team +├── youtube.ts the channel's RSS feed +├── summarize.ts turns a release into a few lines, or says to skip it +├── discord.ts reads the channel, builds the message, posts it +└── http.ts timeouts and JSON parsing shared by the above +``` + +`sources.ts` is the file to edit for anything about coverage. `index.ts` never +names a repository. + +There is no database, no queue and no shared package. It is HTTPS calls and the +decisions between them. + +### Knowing what has already been announced + +The channel is the record. Every announcement ends with its source URL, so the +bot reads back its own recent messages, collects those URLs, and treats the newest +as a watermark. Only items published after the watermark are announced, oldest +first, so the watermark advances one step at a time. + +Two details carry most of the correctness: + +**Announce forward, never backwards.** "Newer than the last announcement" is not +the same as "anything the channel does not mention". The second walks backwards +through history and announces releases that predate the bot entirely. + +**Drain oldest first.** Taking the newest pending items instead moves the +watermark straight to the top, and everything between is dropped permanently +rather than caught up later. That selection lives in `watermark.ts`, apart from +the entry point so it can be tested without starting a run, and it is covered by +tests for exactly that reason. + +**Count posts, not candidates.** A skipped release leaves no trace in the +channel, so when skips consumed the per-run budget two skippable releases in a +row stalled a source until they aged out of the window. The cap is on +announcements made; a separate cap bounds how many releases are examined. + +What follows from using the channel as the record: + +- Running twice in a row posts nothing the second time. +- A crash halfway through a batch cannot cause a repeat, because what was posted + is visibly in the channel and what was not is still absent. +- A failed run needs no recovery. The next run picks up what was missed. +- A channel with no messages from this bot starts at the newest items rather than + replaying history. + +Two costs. Deleting the bot's messages resets its memory of that channel. And the +search is bounded at 300 messages: past that, the oldest message actually read +becomes the floor, so anything published before it is assumed announced rather +than posted again. + +### Writing the announcement + +`summarize.ts` sends the release notes plus the newest 60 commit subjects, and +asks for lines describing what a developer can now do or must now change, with +breaking changes called out first and short headings when a release spans several +areas. Length follows the release: a patch gets two lines, a large release gets +more. + +Three behaviours are worth knowing before changing the prompt: + +**Nothing is ever posted unsummarized.** If the OpenAI call fails, the source +stops there for this run rather than falling back to the raw notes. The raw notes +are the failure case this step exists to avoid: the AG-UI release that shipped 1.0 +opens with four lines of "publish the declared MIT license". + +**A transient failure stops the source; a permanent one does not.** The watermark +is a high-water mark, so announcing a newer release would move it past a failed +one and it would never be retried. But a release that can never be summarized +(a rejected model id, a body the provider refuses) would then block everything +behind it, so those are announced with the link and no summary instead. + +**`SKIP` is checked against the commits.** The model can answer `SKIP` when a +release is only dependency bumps, CI or version metadata. It is not consistent +about this, and in testing the same release was summarized on one run and skipped +on the next. So a `SKIP` is only accepted when no commit subject looks like a +feature, fix or perf change, docs scopes excluded; otherwise the model is asked +again with `SKIP` ruled out. + +### Crediting contributors + +Authors come from the commits between the two releases, with bots and noise +commits filtered out, and contributors outside the org are thanked by name, up to +six with the rest counted. + +Team membership is read from the GitHub orgs rather than a list in the code, which +means `GITHUB_TOKEN` needs `read:org`. A token without it does not fail: it +returns HTTP 200 and an empty member list. So the lookup is all or nothing across +both orgs, and an empty org counts as unresolved. If any page of any org cannot +be read, no credit line is added at all, because publicly thanking colleagues as +though they were outside contributors is worse than saying nothing. `CORE_LOGINS` +extends the team list but cannot assert that the lookup worked. + +GitHub logins cannot be resolved to Discord accounts, so credit is plain text. +Tagging would mean either guessing or pinging people who never joined the server. + +## Sources + +Configured in `src/sources.ts`, one entry per repository. Adding a source is an +entry there plus a channel id in the environment; nothing else needs touching. + +| Source | Announced | +| ----------------------- | ----------------------------------------------- | +| `ag-ui-protocol/ag-ui` | `release/YYYY-MM-DD` tags | +| `CopilotKit/CopilotKit` | `vX.Y.Z`, `channels/vX.Y.Z`, `angular/vX.Y.Z` | +| `CopilotKit/OpenBot` | `vX.Y.Z` | +| CopilotKit on YouTube | Every published video, as a bare link to unfurl | + +Channel ids live in the environment because they differ per server and per +deployment. Tag filters and titles live in the source file because they are +decisions about what is worth announcing and how it should read, and each has its +reason written next to it. + +Sources can share a channel: CopilotKit and OpenBot both post to the CopilotKit +community's releases channel by default, which is why every announcement names +its product on the first line rather than leaving the channel to imply it. + +``` +CopilotKit 1.73.0 Channels SDK 0.10.0 +OpenBot 0.0.15 AG-UI 2026-09-17 +``` + +Give OpenBot `OPENBOT_CHANNEL_ID` if it should have a channel of its own. Note +that two sources in one channel can each post up to the per-run cap. + +AG-UI aggregates a day's package publishes into one dated release, so the tag +shape is all the filter needs to be. + +CopilotKit publishes several release lines from one repo. Announced are the ones +that are both a product people install and still shipping: the main line, the +Channels SDK and the Angular SDK. Patches count, since a two-line release that +fixes something people are hitting is worth saying. + +The rest are skipped, with their share of the last 100 releases: + +| Line | Why | +| ------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------- | +| `channels-teams/`, `-slack/`, `-whatsapp/`, `-telegram/`, `-discord/`, `-intelligence/` (8) | Per-adapter packages, all last released 2026-07-10 and superseded by the `channels/` umbrella | +| `bot/`, `bot-slack/`, `bot-teams/` (6) | Last released 2026-06-25; OpenBot now lives in its own repo | +| `intelligence-mastra/`, `intelligence-langgraph/` (4) | Version alignment. `intelligence-mastra/v1.71.2`'s notes say the API and implementation are unchanged | +| `python-sdk/` (3) | Still shipping, but the release notes are only a PyPI link | +| `pr-*`, `vundefined`, `PR` (5+) | Preview and junk tags that exist in the repo | + +The distinction that matters: a `channels/` release is the SDK shipping, while +`channels-teams/` is one adapter's version moving. + +Each of those is one `include` predicate in `src/sources.ts`, with the reason +written above it. The README table and that comment say the same thing on +purpose: the comment is for whoever changes the regex, the table for whoever +won't open the file. + +## Edge cases + +| Situation | Behaviour | +| ------------------------------------- | --------------------------------------------------------------- | +| Run twice in a row | Second run posts nothing | +| Bot switched off for a fortnight | 2 items per source per run, oldest first, catching up over days | +| OpenAI call fails | That source stops for the run, nothing posted raw, retried next | +| `GITHUB_TOKEN` cannot see the org | Announced with no credit line, never crediting the team | +| Release is only dependency bumps | Skipped, unless the commits show real work | +| Summary longer than Discord allows | Body trimmed; the title and source URL always survive | +| A channel is not configured | That source is skipped, the others still run | +| A source fails outright | Logged, the others still run, and the run exits non-zero | +| Discord rate limit or 5xx | Up to 3 attempts honouring `retry-after`, reads included | +| Release has no previous release | Announced from its notes alone, with no commit context | +| Upcoming premiere in the YouTube feed | Ignored until it has actually aired | + +## Setup + +```bash +# 1. Install + build +pnpm install +pnpm --filter @copilotkit/outpost-release-bot build + +# 2. Configure (gitignored) +cp apps/release-bot/.env.example apps/release-bot/.env + +# 3. See what it would post, without posting +pnpm --filter @copilotkit/outpost-release-bot dry +``` + +`dry` skips the POST and nothing else: it still reads the channel, so it needs a +`DISCORD_BOT_TOKEN` with read access, and it still calls OpenAI for every pending +release, so it still costs money. That is the point of it, since the summary is +usually what you want to check. + +## Environment + +| Variable | Required | Purpose | +| ---------------------------- | ---------- | ---------------------------------------------------------------------------------------------------------- | +| `DISCORD_BOT_TOKEN` | yes | Posting, and reading the channel to see what was already announced | +| `OPENAI_API_KEY` | yes | Writing the announcements. Without it nothing is announced | +| `GITHUB_TOKEN` | yes | Needs `read:org`, to tell the team from outside contributors. Missing, the run fails rather than degrading | +| `AGUI_CHANNEL_ID` | per source | Channel for AG-UI releases | +| `CPK_CHANNEL_ID` | per source | Channel for CopilotKit releases | +| `YOUTUBE_CHANNEL_DISCORD_ID` | per source | Channel for video announcements | +| `YOUTUBE_CHANNEL_ID` | per source | The YouTube channel to watch | +| `AGUI_PING_ROLE_ID` | no | Role to ping for AG-UI releases. Unset means silent | +| `CPK_PING_ROLE_ID` | no | Role to ping for CopilotKit releases. Unset means silent | +| `YOUTUBE_PING_ROLE_ID` | no | Role to ping for videos. Unset means silent | +| `CORE_LOGINS` | no | Comma-separated logins treated as team, on top of org membership | +| `OPENAI_MODEL` | no | Defaults to `gpt-5.4` | + +Announcements are silent by default. At roughly eight a week across the three +repositories, a ping on every one is how a channel gets muted. + +`DISCORD_BOT_TOKEN` is a separate bot identity from +[`apps/discord-bot`](../discord-bot/) and [`apps/discord-mcp`](../discord-mcp/). +This one only posts announcements, so it should not carry the ingest bot's +permissions or the MCP reader's intents. + +## Discord permissions + +**Send Messages**, **View Channel** and **Read Message History**. The last two are +not optional: reading the channel is how the bot knows what it has already +announced, and without them the run fails before posting anything. + +**Embed Links** is not needed to post, since announcements are plain text, but +without it Discord will not unfurl the trailing link into a preview. The bot +never reads those preview embeds back: it unfurls every link in a message, +including any the model wrote into the summary, which is not a safe identity. + +## Deployment + +A Railway cron service. `railway.toml` carries the build config, the restart +policy, the schedule and the watch patterns, so a recreated service is still +scheduled rather than running once at deploy and never again, and an unrelated +push elsewhere in the monorepo does not trigger an extra run. + +`restartPolicyType` is `NEVER`, unlike the long-running services in this repo. A +completed run exits, and an `ALWAYS` policy would read that as a crash and restart +it in a loop. diff --git a/apps/release-bot/package.json b/apps/release-bot/package.json new file mode 100644 index 0000000..8812836 --- /dev/null +++ b/apps/release-bot/package.json @@ -0,0 +1,25 @@ +{ + "name": "@copilotkit/outpost-release-bot", + "version": "0.1.0", + "private": true, + "type": "module", + "main": "./dist/index.js", + "scripts": { + "build": "tsc", + "dev": "tsx --env-file-if-exists=.env src/index.ts", + "dry": "tsx --env-file-if-exists=.env src/index.ts --dry-run", + "start": "node dist/index.js", + "typecheck": "tsc --noEmit", + "lint": "eslint src/", + "test": "vitest run" + }, + "devDependencies": { + "@types/node": "^22.10.0", + "tsx": "^4.19.0", + "typescript": "^5.7.0", + "vitest": "^4.1.4" + }, + "engines": { + "node": ">=20.12.0" + } +} diff --git a/apps/release-bot/railway.toml b/apps/release-bot/railway.toml new file mode 100644 index 0000000..9efe933 --- /dev/null +++ b/apps/release-bot/railway.toml @@ -0,0 +1,25 @@ +# Railway config — outpost-release-bot (scheduled community announcements) +# In Railway: set this service's "Config file path" to apps/release-bot/railway.toml +# and leave Root Directory empty (build context must be the repo root). +# +# This service is a CRON JOB, not a long-running process: it announces whatever is +# new and exits. +# +# cronSchedule is set here rather than in the dashboard so a recreated service is +# still scheduled. Without it, the service runs once at deploy and then never +# again, with no error anywhere to say so. +# +# restartPolicyType = "NEVER" because a completed run exiting 0 is success, not a +# crash. ALWAYS would restart it in a loop and re-announce on every boot. + +# watchPatterns keeps unrelated pushes from redeploying this service. A deploy +# runs the job once immediately, so without it every merge anywhere in the +# monorepo fires an extra announcement run outside the schedule. +[build] +builder = "DOCKERFILE" +dockerfilePath = "apps/release-bot/Dockerfile" +watchPatterns = ["apps/release-bot/**", "pnpm-lock.yaml", "tsconfig.json", "turbo.json"] + +[deploy] +restartPolicyType = "NEVER" +cronSchedule = "0 8 * * *" diff --git a/apps/release-bot/src/__tests__/discord.test.ts b/apps/release-bot/src/__tests__/discord.test.ts new file mode 100644 index 0000000..5e7282a --- /dev/null +++ b/apps/release-bot/src/__tests__/discord.test.ts @@ -0,0 +1,82 @@ +import { describe, expect, it } from 'vitest'; +import { MESSAGE_LIMIT, compose, sourceUrlOf } from '../discord.js'; + +const url = 'https://github.com/CopilotKit/CopilotKit/releases/tag/v1.73.0'; + +describe('compose', () => { + it('keeps the message within Discord’s limit and ends with the source URL', () => { + const content = compose({ + title: 'v1.73.0', + body: 'a'.repeat(4000), + footer: 'thanks someone for contributing :)', + url, + }); + + expect(content.length).toBeLessThanOrEqual(MESSAGE_LIMIT); + expect(content.endsWith(url)).toBe(true); + }); + + it('still ends with the source URL when a role is pinged', () => { + // The mention is ~24 characters. Budgeting without it used to push the + // message over the limit and truncate the URL, which is the dedup key. + const content = compose({ + title: 'v1.73.0', + body: 'a'.repeat(4000), + footer: 'thanks someone for contributing :)', + url, + pingRoleId: '1550206847087288480', + }); + + expect(content.length).toBeLessThanOrEqual(MESSAGE_LIMIT); + expect(content.endsWith(url)).toBe(true); + expect(content.startsWith('<@&1550206847087288480> ')).toBe(true); + }); + + it('keeps the URL when the title and credit line leave no room for anything else', () => { + // Sacrifice order: body, then credit line, then title. Never the URL. + const content = compose({ + title: 'x'.repeat(1200), + body: 'a'.repeat(500), + footer: `thanks ${'contributor, '.repeat(60)}for contributing :)`, + url, + }); + + expect(content.length).toBeLessThanOrEqual(MESSAGE_LIMIT); + expect(content.endsWith(url)).toBe(true); + }); + + it('keeps the URL even when the title alone exceeds the limit', () => { + const content = compose({ title: 'x'.repeat(5000), body: 'body', url }); + expect(content.length).toBeLessThanOrEqual(MESSAGE_LIMIT); + expect(content.endsWith(url)).toBe(true); + }); + + it('leaves a short message untouched', () => { + const content = compose({ title: 'v1.73.0', body: '- one thing shipped', url }); + expect(content).toBe(`**v1.73.0**\n\n- one thing shipped\n\n${url}`); + }); + + it('does not split a surrogate pair when trimming', () => { + const content = compose({ title: 'v1.73.0', body: '🚀'.repeat(2000), url }); + expect(content.length).toBeLessThanOrEqual(MESSAGE_LIMIT); + // A lone high surrogate would be an unpaired code unit in the output. + expect(/[\uD800-\uDBFF](?![\uDC00-\uDFFF])/.test(content)).toBe(false); + }); +}); + +describe('sourceUrlOf', () => { + it('reads the trailing URL', () => { + expect(sourceUrlOf(`**v1.73.0**\n\n- shipped\n\n${url}`)).toBe(url); + }); + + it('ignores links inside the summary body', () => { + // The summary is model-written and may mention other releases. Treating + // those as announced would silently suppress them later. + const content = `**v1.73.0**\n\nsee https://github.com/CopilotKit/CopilotKit/releases/tag/v1.72.0\n\n${url}`; + expect(sourceUrlOf(content)).toBe(url); + }); + + it('returns nothing when the message does not end in a URL', () => { + expect(sourceUrlOf('just a chat message')).toBeUndefined(); + }); +}); diff --git a/apps/release-bot/src/__tests__/github.test.ts b/apps/release-bot/src/__tests__/github.test.ts new file mode 100644 index 0000000..932ddea --- /dev/null +++ b/apps/release-bot/src/__tests__/github.test.ts @@ -0,0 +1,93 @@ +import { afterEach, describe, expect, it, vi } from 'vitest'; +import { team } from '../github.js'; + +const ORIGINAL_FETCH = globalThis.fetch; + +function respondWith(handler: (url: string) => { status?: number; body: unknown }) { + const calls: string[] = []; + vi.stubGlobal('fetch', (input: string | URL) => { + const url = String(input); + calls.push(url); + const { status = 200, body } = handler(url); + return Promise.resolve( + new Response(JSON.stringify(body), { + status, + headers: { 'content-type': 'application/json' }, + }), + ); + }); + return calls; +} + +afterEach(() => { + vi.stubGlobal('fetch', ORIGINAL_FETCH); + vi.resetModules(); +}); + +/** `team()` memoizes for the process, so each case needs a fresh module. */ +async function freshTeam() { + vi.resetModules(); + return (await import('../github.js')).team; +} + +describe('team', () => { + it('reads every org, not just the first', async () => { + // Returning instead of breaking meant ag-ui-protocol was never queried, + // and its members were then thanked publicly as outside contributors. + process.env.GITHUB_TOKEN = 'test'; + const calls = respondWith((url) => + url.includes('/orgs/CopilotKit/') + ? { body: [{ login: 'NathanTarbert' }] } + : { body: [{ login: 'mme' }] }, + ); + + const result = await (await freshTeam())(); + + expect(calls.filter((c) => c.includes('/orgs/CopilotKit/members')).length).toBe(1); + expect(calls.filter((c) => c.includes('/orgs/ag-ui-protocol/members')).length).toBe(1); + expect(result.resolved).toBe(true); + expect([...result.members].sort()).toEqual(['mme', 'nathantarbert']); + }); + + it('is unresolved when no org can be read', async () => { + // A token without read:org does not fail, it returns 200 and []. Reading + // that as "this org has nobody" credits the whole team as outsiders. + process.env.GITHUB_TOKEN = 'test'; + process.env.CORE_LOGINS = 'someone'; + respondWith(() => ({ body: [] })); + + const result = await (await freshTeam())(); + + expect(result.resolved).toBe(false); + delete process.env.CORE_LOGINS; + }); + + it('still resolves when one org is readable and the other is not', async () => { + // ag-ui-protocol returns an empty list for a CopilotKit-scoped token, + // and requiring both made credit impossible in practice. + process.env.GITHUB_TOKEN = 'test'; + respondWith((url) => + url.includes('/orgs/CopilotKit/') + ? { body: [{ login: 'NathanTarbert' }] } + : { status: 403, body: { message: 'Forbidden' } }, + ); + + const result = await (await freshTeam())(); + + expect(result.resolved).toBe(true); + expect(result.members.has('nathantarbert')).toBe(true); + }); + + it('refuses to run without a token rather than degrading silently', async () => { + delete process.env.GITHUB_TOKEN; + respondWith(() => ({ body: [] })); + + await expect((await freshTeam())()).rejects.toThrow(/GITHUB_TOKEN/); + }); +}); + +describe('module surface', () => { + it('exports team for callers that need the resolved flag', () => { + expect(typeof team).toBe('function'); + }); +}); diff --git a/apps/release-bot/src/__tests__/sources.test.ts b/apps/release-bot/src/__tests__/sources.test.ts new file mode 100644 index 0000000..3560b2a --- /dev/null +++ b/apps/release-bot/src/__tests__/sources.test.ts @@ -0,0 +1,82 @@ +import { describe, expect, it } from 'vitest'; +import { SOURCES } from '../sources.js'; +import type { Release } from '../github.js'; + +const sourceFor = (name: string) => { + const source = SOURCES.find((s) => s.name === name); + if (!source) throw new Error(`no source named ${name}`); + return source; +}; + +const filterFor = (name: string) => sourceFor(name).include; + +const titleFor = (name: string, tag: string) => + sourceFor(name).title({ tag, name: tag } as Release); + +describe('copilotkit', () => { + const include = filterFor('copilotkit'); + + it.each(['v1.73.0', 'v1.72.1', 'channels/v0.10.0', 'channels/v0.9.2', 'angular/v0.5.2'])( + 'announces %s', + (tag) => { + expect(include(tag)).toBe(true); + }, + ); + + it.each([ + // Active, but the notes are only a PyPI link. + 'python-sdk/v0.1.96', + // Superseded by the channels/ umbrella, all last released 2026-07-10. + 'channels-teams/v0.1.2', + 'channels-slack/v0.1.1', + // Moved to its own repo. + 'bot-slack/v0.1.0', + // Version alignment only. + 'intelligence-mastra/v1.71.2', + 'intelligence-langgraph/v0.1.0', + // Tags that exist in the repo and are not releases at all. + 'PR', + 'vundefined', + 'pr-6517-visuals', + ])('skips %s', (tag) => { + expect(include(tag)).toBe(false); + }); +}); + +describe('ag-ui', () => { + const include = filterFor('ag-ui'); + + it('announces a dated release', () => { + expect(include('release/2026-09-17')).toBe(true); + }); + + it.each(['release/visual-qa', 'release/2026-9-1', 'v1.0.0'])('skips %s', (tag) => { + expect(include(tag)).toBe(false); + }); +}); + +describe('openbot', () => { + const include = filterFor('openbot'); + + it.each(['v0.0.15', 'v0.0.8'])('announces %s', (tag) => { + expect(include(tag)).toBe(true); + }); + + it.each(['desktop/v0.0.15', 'nightly'])('skips %s', (tag) => { + expect(include(tag)).toBe(false); + }); +}); + +describe('titles', () => { + it.each([ + ['copilotkit', 'v1.73.0', 'CopilotKit 1.73.0'], + ['copilotkit', 'channels/v0.10.0', 'Channels SDK 0.10.0'], + ['copilotkit', 'angular/v0.5.2', 'Angular SDK 0.5.2'], + ['openbot', 'v0.0.15', 'OpenBot 0.0.15'], + ['ag-ui', 'release/2026-09-17', 'AG-UI 2026-09-17'], + ])('names the product for %s %s', (source, tag, expected) => { + // Sources share a channel, so a bare version number would not say which + // product shipped. + expect(titleFor(source, tag)).toBe(expected); + }); +}); diff --git a/apps/release-bot/src/__tests__/summarize.test.ts b/apps/release-bot/src/__tests__/summarize.test.ts new file mode 100644 index 0000000..fe6f404 --- /dev/null +++ b/apps/release-bot/src/__tests__/summarize.test.ts @@ -0,0 +1,48 @@ +import { describe, expect, it } from 'vitest'; +import { SKIP_REPLY, SUBSTANTIVE } from '../summarize.js'; + +describe('SUBSTANTIVE', () => { + it.each([ + 'feat: add useAgent', + 'fix(runtime): stop dropping tool results', + 'perf: cut replay allocations', + // Breaking changes are the releases that must never be silently skipped, + // and `feat!:` has no parenthesis for the pattern to key on. + 'feat!: drop the useRenderTool shim', + 'fix!: rename the flag', + ])('counts %s as shipped work', (subject) => { + expect(SUBSTANTIVE.test(subject)).toBe(true); + }); + + it.each([ + 'fix(docs): typo', + 'feat(doc): rewrite the guide', + // Compound docs scopes reached the channel as feature announcements. + 'fix(docs/api): correct the example', + 'feat(docs,website): new landing page', + 'fix(deps): bump zod', + 'chore(deps): bump vitest', + 'ci: pin the runner', + 'docs: explain threads', + ])('does not count %s', (subject) => { + expect(SUBSTANTIVE.test(subject)).toBe(false); + }); +}); + +describe('SKIP_REPLY', () => { + it.each(['SKIP', 'skip', 'SKIP.', '**SKIP**', ' SKIP ', '`skip`', '- Skip.'])( + 'reads %s as the sentinel', + (reply) => { + expect(SKIP_REPLY.test(reply)).toBe(true); + }, + ); + + it.each([ + 'SKIP this release because nothing shipped', + '- You can now skip the setup step', + 'Skipping is now configurable', + ])('does not read %s as the sentinel', (reply) => { + // A near miss used to be posted verbatim as the announcement body. + expect(SKIP_REPLY.test(reply)).toBe(false); + }); +}); diff --git a/apps/release-bot/src/__tests__/watermark.test.ts b/apps/release-bot/src/__tests__/watermark.test.ts new file mode 100644 index 0000000..3cfc9b3 --- /dev/null +++ b/apps/release-bot/src/__tests__/watermark.test.ts @@ -0,0 +1,85 @@ +import { describe, expect, it } from 'vitest'; +import { pending, type Item } from '../watermark.js'; +import type { Announced } from '../discord.js'; + +const item = (n: number, publishedAt: string): Item => ({ url: `https://x/${n}`, publishedAt }); + +const seen = (urls: string[], foundOwn = true, searchedFrom?: string): Announced => ({ + urls: new Set(urls), + foundOwn, + searchedFrom, +}); + +describe('pending', () => { + it('drains a backlog oldest-first so nothing is skipped past', () => { + const items = [ + item(1, '2026-09-01T00:00:00Z'), + item(2, '2026-09-02T00:00:00Z'), + item(3, '2026-09-03T00:00:00Z'), + item(4, '2026-09-04T00:00:00Z'), + item(5, '2026-09-05T00:00:00Z'), + ]; + + // Announced item 1; items 2..5 are pending, oldest first, so a caller + // taking the first N walks forward one step at a time. + const first = pending(items, seen(['https://x/1'])); + expect(first.map((i) => i.url)).toEqual([ + 'https://x/2', + 'https://x/3', + 'https://x/4', + 'https://x/5', + ]); + + // Next run continues from there rather than jumping to the newest and + // abandoning the middle. + const second = pending(items, seen(['https://x/1', 'https://x/2', 'https://x/3'])); + expect(second.map((i) => i.url)).toEqual(['https://x/4', 'https://x/5']); + }); + + it('announces nothing when the newest item is already announced', () => { + const items = [item(1, '2026-09-01T00:00:00Z'), item(2, '2026-09-02T00:00:00Z')]; + expect(pending(items, seen(['https://x/1', 'https://x/2']))).toEqual([]); + }); + + it('announces only the newest item when the bot has never posted here', () => { + const items = [ + item(1, '2026-09-01T00:00:00Z'), + item(2, '2026-09-02T00:00:00Z'), + item(3, '2026-09-03T00:00:00Z'), + ]; + expect(pending(items, seen([], false)).map((i) => i.url)).toEqual(['https://x/3']); + }); + + it('announces only what postdates the history it could read', () => { + // The bot has posted here, but nothing it announced is in range any + // more. Anything published after the oldest message read would have been + // visible if announced, so only those are pending; older ones are done. + const items = [item(1, '2026-09-01T00:00:00Z'), item(2, '2026-09-20T00:00:00Z')]; + const result = pending(items, seen(['https://x/older'], true, '2026-09-10T00:00:00Z')); + expect(result.map((i) => i.url)).toEqual(['https://x/2']); + }); + + it('falls back to the newest item when the history gives no floor', () => { + const items = [item(1, '2026-09-01T00:00:00Z'), item(2, '2026-09-02T00:00:00Z')]; + expect(pending(items, seen(['https://x/older'], true)).map((i) => i.url)).toEqual([ + 'https://x/2', + ]); + }); + + it('ignores an item whose timestamp cannot be parsed', () => { + // One bad timestamp used to poison the watermark through Math.max and + // take the source silent with no explanation. + const items = [ + item(1, '2026-09-01T00:00:00Z'), + { url: 'https://x/bad', publishedAt: 'not a date' }, + item(3, '2026-09-03T00:00:00Z'), + ]; + expect(pending(items, seen(['https://x/1'])).map((i) => i.url)).toEqual(['https://x/3']); + }); + + it('announces an item published in the same second as the watermark', () => { + const stamp = '2026-09-14T13:03:25Z'; + const items = [item(1, stamp), item(2, stamp)]; + expect(pending(items, seen(['https://x/1'])).map((i) => i.url)).toEqual(['https://x/2']); + }); +}); diff --git a/apps/release-bot/src/__tests__/youtube.test.ts b/apps/release-bot/src/__tests__/youtube.test.ts new file mode 100644 index 0000000..dd7dd69 --- /dev/null +++ b/apps/release-bot/src/__tests__/youtube.test.ts @@ -0,0 +1,47 @@ +import { describe, expect, it } from 'vitest'; +import { parseFeed } from '../youtube.js'; + +const entry = (id: string, published: string, attrs = '') => ` + + ${id} + A video + ${published} + `; + +describe('parseFeed', () => { + it('reads video ids and timestamps', () => { + const videos = parseFeed( + `${entry('abc123', '2026-09-11T19:33:39+00:00')}${entry('def456', '2026-09-10T16:23:21+00:00')}`, + ); + + expect(videos).toEqual([ + { + id: 'abc123', + url: 'https://www.youtube.com/watch?v=abc123', + publishedAt: '2026-09-11T19:33:39+00:00', + }, + { + id: 'def456', + url: 'https://www.youtube.com/watch?v=def456', + publishedAt: '2026-09-10T16:23:21+00:00', + }, + ]); + }); + + it('still parses entries that carry attributes', () => { + // Splitting on the literal `` returned nothing the moment the + // feed added a namespace, which read as an empty channel. + const videos = parseFeed( + `${entry('abc123', '2026-09-11T19:33:39+00:00', ' xml:lang="en"')}`, + ); + expect(videos.map((v) => v.id)).toEqual(['abc123']); + }); + + it('drops entries with no id or timestamp', () => { + expect(parseFeed('broken')).toEqual([]); + }); + + it('returns nothing for an empty feed', () => { + expect(parseFeed('')).toEqual([]); + }); +}); diff --git a/apps/release-bot/src/discord.ts b/apps/release-bot/src/discord.ts new file mode 100644 index 0000000..11df91a --- /dev/null +++ b/apps/release-bot/src/discord.ts @@ -0,0 +1,330 @@ +/** + * Reading and posting, over REST. + * + * A gateway connection is only needed to *receive* events (slash commands, + * reactions). Posting needs nothing but the token, so this stays a scheduled job + * until the bot has a reason to listen. + */ + +import { TIMEOUT_MS, parseJson } from './http.js'; + +const API = 'https://discord.com/api/v10'; + +/** Discord's hard cap on a message. */ +export const MESSAGE_LIMIT = 2000; + +/** Pages of 100 messages to look back through for this bot's own announcements. */ +const HISTORY_PAGES = 3; + +const MAX_RETRIES = 3; + +/** Ceiling on an honoured `retry_after`, so a global limit cannot park the run for an hour. */ +const MAX_RETRY_WAIT_MS = 60_000; + +function auth() { + const token = process.env.DISCORD_BOT_TOKEN; + if (!token) throw new Error('DISCORD_BOT_TOKEN is not set'); + return { + Authorization: `Bot ${token}`, + 'User-Agent': 'DiscordBot (https://github.com/CopilotKit/outpost, 1.0)', + }; +} + +let botIdPromise: Promise | undefined; + +function selfId(): Promise { + botIdPromise ??= (async () => { + // Through request() so this read gets the same retries as the others; a + // single 429 here used to fail whichever source reached it first. + const res = await request('/users/@me', { method: 'GET' }); + if (!res.ok) throw new Error(`Discord ${res.status} reading own identity`); + return (await parseJson<{ id: string }>(res, 'Discord')).id; + })().catch((error) => { + botIdPromise = undefined; + throw error; + }); + return botIdPromise; +} + +type DiscordMessage = { + id: string; + timestamp?: string; + author?: { id: string }; + content?: string; +}; + +export type Announced = { + /** Source URLs this bot has already announced in the channel. */ + urls: Set; + /** False when no message from this bot was found in the pages searched. */ + foundOwn: boolean; + /** Timestamp of the oldest message read, which bounds how far back we looked. */ + searchedFrom?: string; +}; + +/** + * What this bot has already announced in a channel. + * + * The channel is the record, rather than a state file: nothing to commit, + * nothing to migrate when the repo moves, and no way for the file and reality to + * drift apart. Each announcement ends with its source URL, so that URL is the + * identity we match on. + * + * Only that trailing URL counts. Discord's auto-generated preview embeds are + * deliberately ignored: it unfurls every link in a message, including ones the + * model wrote into the summary, and reading those back re-introduced the bug + * `sourceUrlOf` exists to prevent. + * + * Paginating matters: with one page of 50, a channel where people actually talk + * pushed the bot's last announcement out of the window, which read as "never + * posted here" and re-announced the latest release. + */ +export async function announced(channelId: string): Promise { + const me = await selfId(); + const urls = new Set(); + let foundOwn = false; + let searchedFrom: string | undefined; + let before: string | undefined; + + for (let page = 0; page < HISTORY_PAGES; page++) { + const query = new URLSearchParams({ limit: '100' }); + if (before) query.set('before', before); + + const res = await request(`/channels/${channelId}/messages?${query}`, { method: 'GET' }); + + if (res.status === 403) { + throw new Error( + `Discord 403 reading ${channelId}. The bot needs View Channel and Read Message History.`, + ); + } + if (!res.ok) throw new Error(`Discord ${res.status} reading ${channelId}`); + + const batch = await parseJson(res, 'Discord'); + if (!batch.length) break; + + for (const message of batch) { + if (message.author?.id !== me) continue; + foundOwn = true; + const trailing = sourceUrlOf(message.content ?? ''); + if (trailing) urls.add(trailing); + } + + const oldest = batch[batch.length - 1]; + searchedFrom = oldest?.timestamp ?? searchedFrom; + before = oldest?.id; + if (batch.length < 100) break; + } + + if (!foundOwn) { + console.warn( + `${channelId}: no messages from this bot in the last ${HISTORY_PAGES * 100}; ` + + 'treating the channel as new.', + ); + } + + return { urls, foundOwn, searchedFrom }; +} + +/** + * The source URL of an announcement, which is its last line. + * + * Only the trailing URL counts. Harvesting every link in the message swept up + * URLs the model wrote into the summary, and a summary that mentioned another + * release marked that release as already announced. + */ +export function sourceUrlOf(content: string): string | undefined { + const lastLine = content.trimEnd().split('\n').pop() ?? ''; + const match = lastLine.trim().match(/^]+)>?$/); + return match?.[1]; +} + +export type Announcement = { + channelId: string; + /** Bold first line. */ + title: string; + /** The summary. Trimmed if the message would not otherwise fit. */ + body: string; + /** Credit line, kept whole. */ + footer?: string; + /** Ends the message, and is the dedup identity, so it always survives. */ + url: string; + /** Role id to ping. Omitted means the post is silent. */ + pingRoleId?: string; +}; + +/** + * Assembles a message that fits Discord's limit with the source URL intact. + * + * The budget lives here rather than in the caller because the role mention is + * added here: the caller used to budget to exactly 2000 characters, then this + * function prepended a ~24-character mention and truncated the overflow off the + * end, taking the trailing URL with it. A release whose URL was cut looked + * unannounced forever and was re-posted on every run. + */ +export function compose({ + title, + body, + footer, + url, + pingRoleId, +}: Omit): string { + const prefix = pingRoleId ? `${withPing('', pingRoleId)}` : ''; + const urlPart = `\n\n${url}`; + + // Sacrifice order, most expendable first: the body, then the credit line, + // then the title. The URL is never sacrificed, because losing it means the + // announcement can never be recognised again and gets re-posted forever. + const titleRoom = MESSAGE_LIMIT - prefix.length - urlPart.length - '****'.length; + const shownTitle = title.length <= titleRoom ? title : truncate(title, titleRoom); + const head = shownTitle ? `${prefix}**${shownTitle}**` : prefix.trimEnd(); + + let left = MESSAGE_LIMIT - head.length - urlPart.length; + + const footerPart = footer ? `\n\n${footer}` : ''; + const keepFooter = footerPart.length > 0 && footerPart.length <= left; + if (keepFooter) left -= footerPart.length; + + return head + section(body, left) + (keepFooter ? footerPart : '') + urlPart; +} + +/** A `\n\n`-separated section, trimmed to what is left, or nothing if it cannot fit. */ +function section(text: string, room: number): string { + const available = room - '\n\n'.length; + if (!text || available <= 1) return ''; + return `\n\n${text.length <= available ? text : truncate(text, available)}`; +} + +/** Trims to `max` characters including the ellipsis, without splitting a surrogate pair. */ +function truncate(text: string, max: number): string { + if (max <= 1) return ''; + let end = max - 1; + const code = text.charCodeAt(end - 1); + if (code >= 0xd800 && code <= 0xdbff) end -= 1; // don't leave a lone high surrogate + return `${text.slice(0, end).trimEnd()}…`; +} + +export async function announce(announcement: Announcement) { + const content = compose(announcement); + await send(announcement.channelId, content, announcement.pingRoleId); +} + +/** A plain message. Used for videos, where Discord's own unfurl is the preview. */ +export async function postText(channelId: string, content: string, pingRoleId?: string) { + await send(channelId, withPing(content, pingRoleId), pingRoleId); +} + +/** + * The role mention, in one place. + * + * Two sites used to build this independently, and the mention being applied + * outside a caller's character budget is what sheared the trailing URL off + * announcements. + */ +export function withPing(content: string, pingRoleId?: string): string { + return pingRoleId ? `<@&${pingRoleId}> ${content}` : content; +} + +/** A stable per-message id, so the same content cannot post twice in a run. */ +function nonceFor(content: string): string { + let hash = 0; + for (let i = 0; i < content.length; i++) hash = (hash * 31 + content.charCodeAt(i)) | 0; + return `rb-${(hash >>> 0).toString(36)}`; +} + +async function send(channelId: string, content: string, pingRoleId?: string) { + // A message over the limit would be truncated from the end, which is where + // the dedup URL lives. Refusing is safer than posting an announcement that + // can never be recognised again. + if (content.length > MESSAGE_LIMIT) { + throw new Error( + `Message for ${channelId} is ${content.length} characters, over Discord's ${MESSAGE_LIMIT}.`, + ); + } + + const res = await request(`/channels/${channelId}/messages`, { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify({ + // Discord de-duplicates by nonce for a short window, which covers a + // retry that lands after the message was already created. + nonce: nonceFor(content), + content, + // `parse: []` on both paths. The body is model-written text derived + // from release notes, so an @everyone in a summary must be + // structurally impossible rather than left to an API default. + allowed_mentions: { parse: [], roles: pingRoleId ? [pingRoleId] : [] }, + }), + }); + + if (!res.ok) { + throw new Error(`Discord ${res.status}: ${(await res.text()).slice(0, 300)}`); + } +} + +/** + * One request, with bounded retries. + * + * Reads are retried as well as writes: a rate limit while reading the channel + * used to abort the whole run, taking the other sources with it. + * + * What is retried differs by method. A 429 is safe everywhere, because a + * rate-limited request never executed. A 5xx is only retried on reads: Discord + * can accept a message and then fail the response, so retrying a POST risks + * announcing the same release twice, which is worse than failing the run. + * Thrown errors (timeout, reset) follow the same rule. + */ +async function request(path: string, init: RequestInit): Promise { + const isRead = (init.method ?? 'GET') === 'GET'; + + for (let attempt = 1; ; attempt++) { + const last = attempt >= MAX_RETRIES; + + let res: Response; + try { + res = await fetch(`${API}${path}`, { + ...init, + headers: { ...auth(), ...(init.headers ?? {}) }, + signal: AbortSignal.timeout(TIMEOUT_MS), + }); + } catch (error) { + if (last || !isRead) throw error; + await pause(backoff(attempt)); + continue; + } + + const retryable = res.status === 429 || (isRead && res.status >= 500); + if (!retryable || last) return res; + + await pause(await retryDelay(res, attempt)); + } +} + +const pause = (ms: number) => new Promise((resolve) => setTimeout(resolve, ms)); + +const backoff = (attempt: number) => Math.min(1000 * 2 ** (attempt - 1), MAX_RETRY_WAIT_MS); + +/** + * How long to wait before retrying. + * + * The header is read first because a Cloudflare-level 429 returns HTML, and + * parsing that body as JSON threw a SyntaxError that surfaced as + * "Unexpected token '<'" with nothing to say it was a rate limit. + */ +async function retryDelay(res: Response, attempt: number): Promise { + const header = Number(res.headers.get('retry-after')); + if (Number.isFinite(header) && header > 0) { + return Math.min(header * 1000 + 100, MAX_RETRY_WAIT_MS); + } + + const body = await res.text().catch(() => ''); + try { + const parsed = JSON.parse(body) as { retry_after?: number }; + if (typeof parsed.retry_after === 'number') { + return Math.min(parsed.retry_after * 1000 + 100, MAX_RETRY_WAIT_MS); + } + } catch { + // Not JSON, which is itself the signal that this is an edge rate limit. + } + + return backoff(attempt); +} diff --git a/apps/release-bot/src/github.ts b/apps/release-bot/src/github.ts new file mode 100644 index 0000000..6d1d418 --- /dev/null +++ b/apps/release-bot/src/github.ts @@ -0,0 +1,368 @@ +/** + * GitHub source: new releases, plus the context needed to write about them. + * + * Release notes alone are not enough. CopilotKit's are often one sentence, and + * AG-UI's are thousands of characters of package tables. So for every release we + * also pull the commits since the previous release, which gives us the real PR + * list and, more importantly, who wrote them. + */ + +import { TIMEOUT_MS, parseJson } from './http.js'; + +const API = 'https://api.github.com'; + +/** Org listings this bot treats as "the team". */ +const ORGS = ['CopilotKit', 'ag-ui-protocol']; + +/** Pages of 100 members to read before giving up. */ +const MEMBER_PAGES = 20; + +/** Pages of 100 releases to walk back through while still inside the lookback window. */ +const RELEASE_PAGES = 5; + +/** Pages of 100 commits to read from a compare range. */ +const COMPARE_PAGES = 10; + +export type Release = { + repo: string; + tag: string; + name: string; + url: string; + body: string; + publishedAt: string; +}; + +export type Contributor = { + login: string; + external: boolean; +}; + +export type ReleaseContext = Release & { + commits: string[]; + contributors: Contributor[]; +}; + +/** Only the fields this app reads, not the full GitHub payloads. */ +type GhRelease = { + draft: boolean; + prerelease: boolean; + published_at: string | null; + tag_name: string; + name: string | null; + html_url: string; + body: string | null; +}; + +type GhCommit = { + commit: { message: string }; + author: { login: string } | null; +}; + +type GhCompare = { commits?: GhCommit[]; total_commits?: number }; + +type GhMember = { login: string }; + +function headers() { + const token = process.env.GITHUB_TOKEN; + // Unauthenticated, GitHub allows 60 requests an hour and hides org members, + // so the run would die partway through with an opaque 403 and credit nobody. + // Failing here names the actual problem instead. + if (!token) { + throw new Error( + 'GITHUB_TOKEN is not set. It needs read:org to tell the team from contributors.', + ); + } + return { + Accept: 'application/vnd.github+json', + 'User-Agent': 'copilotkit-release-bot', + Authorization: `Bearer ${token}`, + }; +} + +const MAX_RETRIES = 3; + +/** + * A GitHub read, with bounded retries. + * + * Reads are retried for the same reason Discord's are: a secondary rate limit + * or a 502 on one of several calls per run would otherwise fail a whole source. + * `tolerate` returns null instead of throwing, for the caller that needs to + * degrade rather than abort. + */ +async function gh(path: string, options: { tolerate?: boolean } = {}): Promise { + // Outside the loop, so a missing token throws rather than being tolerated + // as though it were an HTTP failure. A config error is not a bad network. + const requestHeaders = headers(); + + for (let attempt = 1; ; attempt++) { + const last = attempt >= MAX_RETRIES; + + let res: Response; + try { + res = await fetch(`${API}${path}`, { + headers: requestHeaders, + signal: AbortSignal.timeout(TIMEOUT_MS), + }); + } catch (error) { + if (last) { + if (options.tolerate) return null as T; + throw error; + } + await new Promise((r) => setTimeout(r, 1000 * 2 ** (attempt - 1))); + continue; + } + + if (res.ok) return parseJson(res, 'GitHub'); + + const retryable = res.status === 429 || res.status === 403 || res.status >= 500; + if (retryable && !last) { + const after = Number(res.headers.get('retry-after')); + const wait = + Number.isFinite(after) && after > 0 ? after * 1000 : 1000 * 2 ** (attempt - 1); + await new Promise((r) => setTimeout(r, Math.min(wait, 60_000))); + continue; + } + + if (options.tolerate) return null as T; + throw new Error(`GitHub ${res.status} on ${path}: ${(await res.text()).slice(0, 200)}`); + } +} + +/** + * Releases published since `since`, oldest first. + * + * Paginates rather than reading one page: the API orders by creation, not + * publication, and a busy week of per-package releases pushed main-line releases + * out of a single 30-item page while they were still inside the lookback window. + * A release that falls out of the window is not deferred, it is lost, because the + * watermark has already moved past it. + */ +export async function listReleases(repo: string, since: string): Promise { + const collected: Release[] = []; + const cutoff = Date.parse(since); + let exhausted = true; + + for (let page = 1; page <= RELEASE_PAGES; page++) { + const batch = await gh(`/repos/${repo}/releases?per_page=100&page=${page}`); + if (!batch.length) break; + + for (const r of batch) { + if (r.draft || r.prerelease || !r.published_at) continue; + // By instant, not by string: `since` carries milliseconds and + // GitHub's timestamps do not, so a lexicographic compare disagrees + // inside the boundary second. + if (Date.parse(r.published_at) <= cutoff) continue; + collected.push({ + repo, + tag: r.tag_name, + name: r.name || r.tag_name, + url: r.html_url, + body: r.body || '', + publishedAt: r.published_at, + }); + } + + // Ordering is by creation, so only stop once a whole page is older than + // the window rather than on the first old entry. + const allOlder = batch.every( + (r) => !r.published_at || Date.parse(r.published_at) <= cutoff, + ); + if (allOlder || batch.length < 100) { + exhausted = false; + break; + } + } + + // Anything still inside the window but past this many pages is invisible, + // and invisible means lost rather than deferred once the watermark moves. + if (exhausted) { + console.warn(`${repo}: more than ${RELEASE_PAGES} pages of releases inside the window`); + } + + return collected.sort((a, b) => Date.parse(a.publishedAt) - Date.parse(b.publishedAt)); +} + +/** + * Logins that belong to automation. + * + * Anchored, because unanchored substrings dropped real people: a contributor + * called `renovate-fan` is not Renovate. + */ +const BOT_LOGINS = [ + /\[bot\]$/i, + /^renovate(-bot)?$/i, + /^dependabot$/i, + /^claude$/i, + /^copilot$/i, + /^cursoragent$/i, + /^coderabbitai$/i, + /^sweep-ai$/i, + /^[\w-]*devops-bot$/i, +]; + +const isBot = (login: string) => BOT_LOGINS.some((pattern) => pattern.test(login)); + +/** + * Who counts as the team. + * + * Read from the orgs rather than kept in a list here, because most of the team's + * membership is private and the public endpoint reports colleagues as outsiders. + * + * A token without org visibility does not fail, it returns 200 and an empty + * list, and a failure mid-pagination used to pass silently with a partial list. + * Either outcome credits colleagues publicly as outside contributors, which is + * worse than crediting nobody, so `resolved` is false unless at least one org + * was read completely and returned members. + * + * It is not all-or-nothing across every org, because that made credit + * impossible in practice: `ag-ui-protocol` returns an empty list for a token + * scoped to `CopilotKit`, and the maintainers of both repos are in the + * CopilotKit org anyway. An org that cannot be read is warned about loudly, and + * `CORE_LOGINS` covers anyone it would otherwise have missed. + */ +export type Team = { members: Set; resolved: boolean }; + +let teamPromise: Promise | undefined; + +export function team(): Promise { + teamPromise ??= readTeam().catch((error) => { + // Do not cache the rejection: one transient network error would + // otherwise poison every later call in the run. + teamPromise = undefined; + throw error; + }); + return teamPromise; +} + +async function readTeam(): Promise { + const members = new Set( + (process.env.CORE_LOGINS?.split(',') ?? []) + .map((l) => l.trim().toLowerCase()) + .filter(Boolean), + ); + + const unreadable: string[] = []; + + for (const org of ORGS) { + let complete = false; + let seen = 0; + + for (let page = 1; page <= MEMBER_PAGES && !complete; page++) { + const res = await gh( + `/orgs/${org}/members?per_page=100&page=${page}`, + { tolerate: true }, + ); + + if (!res) { + console.warn( + `Could not read ${org} members (page ${page}). Anyone only in that org ` + + 'may be credited as an outside contributor; add them to CORE_LOGINS. ' + + 'GITHUB_TOKEN needs read:org.', + ); + unreadable.push(org); + break; + } + + for (const m of res) members.add(m.login.toLowerCase()); + seen += res.length; + // `break` here, not `return`: returning meant the second org was + // never read at all, and its members were then thanked publicly as + // outside contributors. + if (res.length < 100) complete = true; + } + + if (unreadable.includes(org)) continue; + + if (!complete) { + console.warn(`${org} has more members than ${MEMBER_PAGES} pages; not crediting.`); + return { members, resolved: false }; + } + + // 200 with an empty list is what a token lacking visibility returns. + // Treating it as "this org has nobody" is what credits a whole team as + // outsiders, so it counts as unreadable rather than as an answer. + if (!seen) { + console.warn( + `${org} returned no members; GITHUB_TOKEN cannot see it. Anyone only in that ` + + 'org may be credited as an outside contributor; add them to CORE_LOGINS.', + ); + unreadable.push(org); + } + } + + // Credit needs at least one org actually read. With none, every contributor + // would look external and the whole team would be thanked publicly. + const resolved = unreadable.length < ORGS.length; + if (!resolved) console.warn('No org membership visible; contributors will not be credited.'); + + return { members, resolved }; +} + +/** Commit subjects that are noise in an announcement and in the credit line. */ +const NOISE = + /^((chore|build|fix|ci|test|docs)\((deps|deps-dev|release)\)|chore\(release\)|chore: bump|release:|(ci|test|docs)[(:]|Merge )/i; + +/** + * Commits between the previous release and this one, and who authored them. + * + * `previous` comes from the caller's release list rather than from `GET /tags`. + * Tag adjacency looked right and was not: the tag list is ordered by refname, + * carries junk tags (`vundefined` sorts above `v1.73.0`), and mixes per-package + * tags in, so `tags[index + 1]` could pick a baseline from an unrelated line and + * credit people for commits they had nothing to do with. + */ +export async function contextFor(release: Release, previous?: Release): Promise { + if (!previous) { + console.warn( + `${release.tag}: no previous release in the window, announcing without commit context`, + ); + return { ...release, commits: [], contributors: [] }; + } + + const range = `${encodeURIComponent(previous.tag)}...${encodeURIComponent(release.tag)}`; + const first = await gh(`/repos/${release.repo}/compare/${range}?per_page=100`); + + const all = [...(first.commits ?? [])]; + const total = first.total_commits ?? all.length; + + // The compare endpoint caps a page at 250 and returns oldest first, so + // without paging the newest work in a large release is simply absent, and + // any contributor who only appears there goes unthanked. + for (let page = 2; all.length < total && page <= COMPARE_PAGES; page++) { + const next = await gh( + `/repos/${release.repo}/compare/${range}?per_page=100&page=${page}`, + ); + const batch = next.commits ?? []; + if (!batch.length) break; + all.push(...batch); + } + + if (all.length < total) { + console.warn( + `${release.tag}: read ${all.length} of ${total} commits; ` + + 'the summary and credit line may be incomplete.', + ); + } + + const substantive = all.filter((c) => !NOISE.test(c.commit.message.split('\n')[0])); + const commits = substantive.map((c) => c.commit.message.split('\n')[0]); + + // Authors come from the filtered commits: someone whose only commits in the + // range were dependency bumps should not be thanked for the release. + const logins = new Set(); + for (const c of substantive) { + const login = c.author?.login; + if (login && !isBot(login)) logins.add(login); + } + + const { members, resolved } = await team(); + + const contributors = [...logins].map((login) => ({ + login, + // Unknown team means unknown provenance, so nobody is marked external + // and the announcement carries no credit line at all. + external: resolved && !members.has(login.toLowerCase()), + })); + + return { ...release, commits, contributors }; +} diff --git a/apps/release-bot/src/http.ts b/apps/release-bot/src/http.ts new file mode 100644 index 0000000..4539fec --- /dev/null +++ b/apps/release-bot/src/http.ts @@ -0,0 +1,20 @@ +/** + * Shared fetch concerns. + * + * Node's fetch has no default timeout, and this runs as a cron container with a + * restart policy of NEVER: one hung socket would leave the run alive until a + * later scheduled run overlapped it, and two concurrent runs reading the same + * channel can both decide the same release is unannounced. + */ + +export const TIMEOUT_MS = 20_000; + +/** Parses a JSON body, naming the service when the body turns out not to be JSON. */ +export async function parseJson(res: Response, what: string): Promise { + const text = await res.text(); + try { + return JSON.parse(text) as T; + } catch (cause) { + throw new Error(`${what} returned a non-JSON body: ${text.slice(0, 200)}`, { cause }); + } +} diff --git a/apps/release-bot/src/index.ts b/apps/release-bot/src/index.ts new file mode 100644 index 0000000..546ec40 --- /dev/null +++ b/apps/release-bot/src/index.ts @@ -0,0 +1,198 @@ +/** + * One pass: look at what each channel already says, find what shipped since, + * announce the difference. + * + * The channel is the source of truth. Every announcement ends with its source + * URL, so "have we said this already" is answered by reading the bot's own + * recent messages rather than by trusting a file to still be accurate. + * + * Run it on any schedule. Running it twice in a row posts nothing the second + * time, and a crash halfway through a batch cannot cause a repeat. + */ + +import { pathToFileURL } from 'node:url'; +import { listReleases, contextFor, type Release } from './github.js'; +import { listVideos, type Video } from './youtube.js'; +import { summarize } from './summarize.js'; +import { announce, announced, postText, withPing } from './discord.js'; +import { pending } from './watermark.js'; +import { SOURCES, type Source } from './sources.js'; + +const DRY_RUN = process.argv.includes('--dry-run'); + +/** + * How far back to look for things to announce. + * + * This also bounds the watermark: if the last announced item is older than the + * window, the run cannot see it and falls back to announcing the newest items. + */ +const LOOKBACK_DAYS = 30; + +/** + * Most announcements *posted* per source per run. A source that has been off for + * a fortnight catches up over several runs instead of emptying its backlog into + * the channel at once. Two sources pointed at one channel can each post this + * many. + * + * Counted on posts, not on candidates: a skipped release leaves no trace in the + * channel, so when skips consumed the budget two skippable releases in a row + * stalled a source until they aged out of the window. + */ +const MAX_PER_RUN = 2; + +/** + * Most releases examined per source per run, however few of them post. Bounds + * the OpenAI spend on a long run of skippable releases. + */ +const MAX_CONSIDERED = 10; + +function lookback(): string { + return new Date(Date.now() - LOOKBACK_DAYS * 864e5).toISOString(); +} + +/** Contributors outside the org, named in the announcement. */ +function credit(contributors: { login: string; external: boolean }[]) { + const external = contributors.filter((c) => c.external); + if (!external.length) return undefined; + + const shown = external.slice(0, 6).map((c) => c.login); + const rest = external.length - shown.length; + const others = rest === 1 ? 'other' : 'others'; + const names = rest > 0 ? `${shown.join(', ')} and ${rest} ${others}` : shown.join(', '); + return `thanks ${names} for contributing :)`; +} + +async function announceReleases(source: Source) { + if (!source.channelId) { + console.log(`${source.name}: no channel configured, skipping`); + return; + } + + const seen = await announced(source.channelId); + const releases = (await listReleases(source.repo, lookback())).filter((r) => + source.include(r.tag), + ); + + let posted = 0; + + for (const release of pending(releases, seen).slice(0, MAX_CONSIDERED)) { + if (posted >= MAX_PER_RUN) break; + + // The previous release of the same line, from this list: tag adjacency + // could pick a baseline from an unrelated tag line. + const previous = releases[releases.indexOf(release) - 1]; + const context = await contextFor(release, previous); + const summary = await summarize(context); + + if (summary.kind === 'failed' && summary.retryable) { + // Stop this source here, holding its position. Announcing a newer + // release would move the watermark past this one and it would never + // be retried, even though asking again would have worked. + throw new Error(`${release.tag}: ${summary.reason}`); + } + + if (summary.kind === 'skip') { + console.log(`${release.tag}: nothing user-facing, skipped`); + continue; + } + + // A permanent failure still gets announced, with the link instead of a + // summary. Raw notes are never posted, but silence is not the answer + // either: a release that can never be summarized would otherwise block + // every release behind it for as long as it stays in the window. + const body = + summary.kind === 'text' ? summary.text : 'Summary unavailable. See the release notes.'; + + if (summary.kind === 'failed') { + console.error(`${release.tag}: ${summary.reason}; announcing with the link only`); + } + + const announcement = { + channelId: source.channelId, + title: source.title(release), + body, + footer: credit(context.contributors), + url: release.url, + pingRoleId: source.pingRoleId, + }; + + if (DRY_RUN) { + console.log(`\n--- ${source.name} ${release.tag} ---\n${body}`); + if (announcement.footer) console.log(announcement.footer); + posted++; + continue; + } + + await announce(announcement); + posted++; + console.log(`${source.name} ${release.tag}: posted`); + } +} + +/** + * Videos post as a line of text and a bare link. Discord unfurls the link into a + * player, which is a better preview than anything we could assemble, so the + * message stays out of its way. + */ +async function announceVideos() { + const channelId = process.env.YOUTUBE_CHANNEL_DISCORD_ID; + const youtubeChannel = process.env.YOUTUBE_CHANNEL_ID; + + if (!channelId || !youtubeChannel) { + console.log('youtube: no channel configured, skipping'); + return; + } + + const seen = await announced(channelId); + const videos = await listVideos(youtubeChannel, lookback()); + const pingRoleId = process.env.YOUTUBE_PING_ROLE_ID; + + for (const video of pending