From b93182bf1cfca9dc706bee3c534faf096e830b1c Mon Sep 17 00:00:00 2001 From: SBrandeis <33657802+SBrandeis@users.noreply.github.com> Date: Fri, 18 Sep 2026 17:34:33 +0200 Subject: [PATCH 1/4] new python-release workflow --- .github/workflows/CI.yml | 210 -------------- .github/workflows/python-release.yml | 403 +++++++++++++++------------ bindings/python/Makefile | 6 +- 3 files changed, 225 insertions(+), 394 deletions(-) delete mode 100644 .github/workflows/CI.yml diff --git a/.github/workflows/CI.yml b/.github/workflows/CI.yml deleted file mode 100644 index 9c0df4e410..0000000000 --- a/.github/workflows/CI.yml +++ /dev/null @@ -1,210 +0,0 @@ -# This file is autogenerated by maturin v1.7.4 -# To update, run -# -# maturin generate-ci github -m bindings/python/Cargo.toml -# -name: CI - -on: - push: - branches: - - main - - master - tags: - - '*' - pull_request: - # The bindings do not build against the v1 integration branch yet: `PipelineTokenizer` is - # read-only and the trainers left the umbrella crate, so a PR into `feat/train_encode_split` - # fails here for the reasons REQUIRED_FOR_V1.md §2 already tracks -- 195 references to wrapper - # types that no longer exist. Drop this filter together with the mutable pipeline builder. - branches-ignore: - - feat/train_encode_split - workflow_dispatch: - -permissions: - contents: read - -jobs: - linux: - runs-on: ${{ matrix.platform.runner }} - strategy: - fail-fast: false - matrix: - platform: - - runner: ubuntu-latest - arch: x86_64 - target: x86_64-unknown-linux-gnu - - runner: ubuntu-latest - arch: x86 - target: i686-unknown-linux-gnu - - runner: ubuntu-latest - arch: aarch64 - target: aarch64-unknown-linux-gnu - - runner: ubuntu-latest - arch: armv7 - target: armv7-unknown-linux-gnueabihf - - runner: ubuntu-latest - arch: s390x - target: s390x-unknown-linux-gnu - - runner: ubuntu-latest - arch: ppc64le - target: powerpc64le-unknown-linux-gnu - - runner: ubuntu-latest - arch: riscv64 - target: riscv64gc-unknown-linux-gnu - steps: - - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - - uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 - with: - python-version: 3.x - - name: Build wheels - uses: PyO3/maturin-action@e83996d129638aa358a18fbd1dfb82f0b0fb5d3b # v1.51.0 - with: - target: ${{ matrix.platform.target }} - args: --release --out dist --manifest-path bindings/python/Cargo.toml - sccache: 'true' - manylinux: auto - - name: Upload wheels - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 - with: - name: wheels-linux-${{ matrix.platform.arch }} - path: dist - - musllinux: - runs-on: ${{ matrix.platform.runner }} - strategy: - fail-fast: false - matrix: - platform: - - runner: ubuntu-latest - target: x86_64 - - runner: ubuntu-latest - target: x86 - - runner: ubuntu-latest - target: aarch64 - - runner: ubuntu-latest - target: armv7 - steps: - - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - - uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 - with: - python-version: 3.x - - name: Build wheels - uses: PyO3/maturin-action@e83996d129638aa358a18fbd1dfb82f0b0fb5d3b # v1.51.0 - with: - target: ${{ matrix.platform.target }} - args: --release --out dist --manifest-path bindings/python/Cargo.toml - sccache: 'true' - manylinux: musllinux_1_2 - - name: Upload wheels - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 - with: - name: wheels-musllinux-${{ matrix.platform.target }} - path: dist - - windows: - runs-on: ${{ matrix.platform.runner }} - strategy: - fail-fast: false - matrix: - platform: - - runner: windows-latest - target: x64 - architecture: x64 - - runner: windows-latest - target: x86 - architecture: x86 - - runner: windows-11-arm - target: aarch64 - architecture: arm64 - steps: - - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - - uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 - with: - python-version: 3.x - architecture: ${{ matrix.platform.architecture }} - - name: Build wheels - uses: PyO3/maturin-action@e83996d129638aa358a18fbd1dfb82f0b0fb5d3b # v1.51.0 - with: - target: ${{ matrix.platform.target }} - args: --release --out dist --manifest-path bindings/python/Cargo.toml - sccache: 'true' - - name: Upload wheels - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 - with: - name: wheels-windows-${{ matrix.platform.target }} - path: dist - - macos: - runs-on: ${{ matrix.platform.runner }} - strategy: - fail-fast: false - matrix: - platform: - - runner: macos-15-intel - target: x86_64 - - runner: macos-14 - target: aarch64 - steps: - - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - - uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 - with: - python-version: 3.x - - name: Build wheels - uses: PyO3/maturin-action@e83996d129638aa358a18fbd1dfb82f0b0fb5d3b # v1.51.0 - with: - target: ${{ matrix.platform.target }} - args: --release --out dist --manifest-path bindings/python/Cargo.toml - sccache: 'true' - env: - # abi3 extension modules resolve Python symbols at runtime; - # explicit --target triggers cross-compilation mode which - # doesn't pass this flag automatically on macOS. - RUSTFLAGS: "-C link-arg=-undefined -C link-arg=dynamic_lookup" - - name: Upload wheels - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 - with: - name: wheels-macos-${{ matrix.platform.target }} - path: dist - - sdist: - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - - name: Build sdist - uses: PyO3/maturin-action@e83996d129638aa358a18fbd1dfb82f0b0fb5d3b # v1.51.0 - with: - command: sdist - args: --out dist --manifest-path bindings/python/Cargo.toml - - name: Upload sdist - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 - with: - name: wheels-sdist - path: dist - - release: - name: Release - runs-on: ubuntu-latest - if: ${{ startsWith(github.ref, 'refs/tags/') || github.event_name == 'workflow_dispatch' }} - needs: [linux, musllinux, windows, macos, sdist] - permissions: - # Use to sign the release artifacts - id-token: write - # Used to upload release artifacts - contents: write - # Used to generate artifact attestation - attestations: write - steps: - - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 - - name: Generate artifact attestation - uses: actions/attest-build-provenance@4d101475d8b20a2381f78447822ac1eab6504dd8 # v4.2.2 - with: - subject-path: 'wheels-*/*' - - name: Publish to PyPI - if: "startsWith(github.ref, 'refs/tags/')" - uses: PyO3/maturin-action@e83996d129638aa358a18fbd1dfb82f0b0fb5d3b # v1.51.0 - env: - MATURIN_PYPI_TOKEN: ${{ secrets.PYPI_TOKEN_DIST}} - with: - command: upload - args: --non-interactive --skip-existing wheels-*/* diff --git a/.github/workflows/python-release.yml b/.github/workflows/python-release.yml index 776f40480b..67de875d1f 100644 --- a/.github/workflows/python-release.yml +++ b/.github/workflows/python-release.yml @@ -3,223 +3,262 @@ on: push: tags: - v* - -env: - AWS_DEFAULT_REGION: us-east-1 - DIST_DIR: ${{ github.sha }} + # ^ trigger when a v* tag gets pushed + # Manual runs build and test everything but never publish: see the `if` on `publish`. + workflow_dispatch: permissions: {} jobs: - lock_exists: + version-match: + name: Assert the tag and version match permissions: contents: read runs-on: ubuntu-latest - name: Cargo.lock steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - - name: Cargo.lock lock exists - run: cat Cargo.lock - working-directory: ./bindings/python + - uses: astral-sh/setup-uv@bec219d24cd3e171d82865faccec33120bb574f4 # v10.1.0 + # Step-level, not job-level: a skipped job would skip everything that `needs` it. + - if: startsWith(github.ref, 'refs/tags/v') + run: | + cargo_version=$( + cargo metadata --no-deps \ + --manifest-path bindings/python/Cargo.toml \ + --format-version 1 \ + | jq -r '.packages[0].version' + ) + uvx --from packaging python -c "\ + import sys + from packaging.version import Version + + assert Version(sys.argv[1]) == Version(sys.argv[2]), sys.argv[1:] + " "$cargo_version" "${GITHUB_REF_NAME#v}" + + build-sdist: + needs: ["version-match"] + name: Make sdist + runs-on: ubuntu-latest + permissions: + contents: read + # The from-sdist build compiles the whole extension cold, without sccache. + timeout-minutes: 30 + defaults: + run: + shell: bash + working-directory: bindings/python + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + - uses: astral-sh/setup-uv@bec219d24cd3e171d82865faccec33120bb574f4 # v10.1.0 + - name: build sdist + uses: PyO3/maturin-action@e83996d129638aa358a18fbd1dfb82f0b0fb5d3b # v1.51.0 + with: + working-directory: bindings/python + command: sdist + rust-toolchain: stable + args: --out dist + - name: upload sdist + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: wheels-sdist + path: bindings/python/dist + if-no-files-found: error + - name: twine check + run: | + ls -lh dist + uvx twine check --strict dist/* + - name: build from the sdist + run: | + uv venv --python 3.14 .venv + source .venv/bin/activate + + # --no-binary tokenizers: refuse any wheel for this name, so the tarball is what gets built. + # Dependencies still come as wheels from PyPI. + # This runs maturin from PyPI under PEP 517 build isolation, the same path a user without a wheel takes. + uv pip install --no-binary tokenizers dist/*.tar.gz + python -c "import tokenizers; print(tokenizers.__version__)" - build: + build-wheel: + needs: ["version-match"] + name: Build Python wheel - ${{ matrix.config.runner }} - ${{ matrix.config.target }} - ${{ matrix.config.python }} + runs-on: ${{ matrix.config.runner }} permissions: contents: read - name: build on ${{ matrix.platform || matrix.os }} (${{ matrix.target }} - ${{ matrix.manylinux || 'auto' }} - ${{ matrix.flavor == 'ft' && '3.14t' || matrix.interpreter || '3.14' }}) - # only run on push to main and on release - needs: [lock_exists] - if: startsWith(github.ref, 'refs/tags/') || github.ref == 'refs/heads/main' || contains(github.event.pull_request.labels.*.name, 'Full Build') + timeout-minutes: 30 + defaults: + run: + shell: bash + working-directory: bindings/python strategy: fail-fast: false matrix: - os: [ubuntu, macos, windows] - target: [x86_64, aarch64] - manylinux: [auto] - # `flavor` discriminates regular abi3 wheels from free-threaded - # (3.14t, non-abi3) wheels so the include: entries below for - # 3.14t create new matrix cells instead of merging with the abi3 - # combos. - flavor: [abi3] - include: - - os: ubuntu - platform: linux - - os: windows - ls: dir - interpreter: "3.14" - - os: windows - ls: dir - target: x86_64 - python-architecture: x64 - python-install: "3.14" - interpreter: "3.14" - - os: windows - ls: dir - target: i686 - python-architecture: x86 - python-install: "3.14" - interpreter: "3.14" - - os: windows-11-arm - ls: dir - target: aarch64 - python-architecture: arm64 - # 3.14t arm64-freethreaded is currently broken upstream: - # actions/python-versions ships a 0-byte python.exe so pip - # install fails with `ModuleNotFoundError: encodings`. Drop - # 3.14t here until the upstream package is fixed. - python-install: "3.14" - interpreter: "3.14" - # - os: windows - # ls: dir - # target: aarch64 - # interpreter: 3.11 3.12 - - os: macos - target: aarch64 - interpreter: "3.14" - - os: ubuntu - platform: linux - target: i686 - - os: ubuntu - platform: linux - target: aarch64 - - - os: ubuntu - platform: linux - target: armv7 - interpreter: "3.14" - # musllinux - - os: ubuntu - platform: linux - target: x86_64 - manylinux: musllinux_1_1 - - os: ubuntu - platform: linux - target: aarch64 - manylinux: musllinux_1_1 - - os: ubuntu - platform: linux - target: ppc64le - interpreter: "3.14" - - os: ubuntu - platform: linux - target: s390x - interpreter: "3.14" - - # --- Free-threaded Python 3.14t wheels ------------------------- - # `flavor: ft` switches the build to non-abi3 (`--no-default-features - # --features ext-module`) and `--interpreter 3.14t`, derived from - # `flavor` in the maturin invocation below. linux container builds - # (manylinux/musllinux) get 3.14t from the docker image, so - # `python-install` is unset for those; macOS/windows host builds - # need it set explicitly so setup-python actually installs 3.14t. - # windows-11-arm 3.14t is intentionally absent — see the abi3 - # windows-11-arm entry above (upstream-broken package). - - { os: ubuntu, platform: linux, target: x86_64, manylinux: auto, flavor: ft } - - { os: ubuntu, platform: linux, target: aarch64, manylinux: auto, flavor: ft } - - { os: ubuntu, platform: linux, target: x86_64, manylinux: musllinux_1_1, flavor: ft } - - { os: ubuntu, platform: linux, target: aarch64, manylinux: musllinux_1_1, flavor: ft } - - { os: macos, target: x86_64, manylinux: auto, flavor: ft, python-install: "3.14t" } - - { os: macos, target: aarch64, manylinux: auto, flavor: ft, python-install: "3.14t" } - - { os: windows, ls: dir, target: x86_64, manylinux: auto, python-architecture: x64, python-install: "3.14t", flavor: ft } - exclude: - - os: windows - target: aarch64 - # # Optimized PGO builds for x86_64 manylinux and windows follow a different matrix, - # # maybe in future maturin-action can support this automatically - # - os: ubuntu - # target: x86_64 - # manylinux: auto - # - os: windows - # target: x86_64 - # Windows on arm64 only supports Python 3.11+ - - runs-on: ${{ - matrix.os == 'windows-11-arm' && matrix.os || - format('{0}-latest', matrix.os) - }} + config: + # ---- Linux glibc, abi3 (one wheel for 3.10+) ------------------------------------- + - { runner: ubuntu-24.04, target: x86_64, manylinux: auto, python: "3.14" } + - { runner: ubuntu-24.04-arm, target: aarch64, manylinux: auto, python: "3.14" } + - { runner: ubuntu-24.04, target: i686, manylinux: auto, python: "3.14" } + - { runner: ubuntu-24.04, target: armv7, manylinux: auto, python: "3.14" } + - { runner: ubuntu-24.04, target: ppc64le, manylinux: auto, python: "3.14" } + - { runner: ubuntu-24.04, target: s390x, manylinux: auto, python: "3.14" } + - { runner: ubuntu-24.04, target: riscv64, manylinux: auto, python: "3.14" } + + # ---- Linux glibc, free-threaded (cp314t) ------------------------------------------ + - { runner: ubuntu-24.04, target: x86_64, manylinux: auto, python: "3.14t" } + - { runner: ubuntu-24.04-arm, target: aarch64, manylinux: auto, python: "3.14t" } + + # ---- Linux musl ------------------------------------------------------------------ + - { runner: ubuntu-24.04, target: x86_64, manylinux: musllinux_1_2, python: "3.14" } + - { runner: ubuntu-24.04, target: aarch64, manylinux: musllinux_1_2, python: "3.14" } + - { runner: ubuntu-24.04, target: i686, manylinux: musllinux_1_2, python: "3.14" } + - { runner: ubuntu-24.04, target: armv7, manylinux: musllinux_1_2, python: "3.14" } + - { runner: ubuntu-24.04, target: x86_64, manylinux: musllinux_1_2, python: "3.14t" } + - { runner: ubuntu-24.04, target: aarch64, manylinux: musllinux_1_2, python: "3.14t" } + + # ---- macOS ----------------------------------------------------------------------- + # macos-latest is macOS 26 on arm64; macos-15-intel is the only current x86_64 box. + - { runner: macos-15-intel, target: x86_64, python: "3.14" } + - { runner: macos-latest, target: aarch64, python: "3.14" } + - { runner: macos-15-intel, target: x86_64, python: "3.14t" } + - { runner: macos-latest, target: aarch64, python: "3.14t" } + + # ---- Windows --------------------------------------------------------------------- + - { runner: windows-latest, target: x86_64, python: "3.14" } + - { runner: windows-latest, target: i686, python: "3.14", python-arch: x86 } + - { runner: windows-11-arm, target: aarch64, python: "3.14", python-arch: arm64 } + - { runner: windows-latest, target: x86_64, python: "3.14t" } steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - - - name: set up python + - uses: astral-sh/setup-uv@bec219d24cd3e171d82865faccec33120bb574f4 # v10.1.0 + - name: setup python + if: runner.os != 'Linux' uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 with: - python-version: ${{ matrix.python-install || '3.14' }} - architecture: ${{ matrix.python-architecture || 'x64' }} - - - run: pip install -U twine - + python-version: ${{ matrix.config.python }} + architecture: ${{ matrix.config.python-arch || 'x64' }} - name: build wheels uses: PyO3/maturin-action@e83996d129638aa358a18fbd1dfb82f0b0fb5d3b # v1.51.0 with: - target: ${{ matrix.target }} - working-directory: ./bindings/python - manylinux: ${{ matrix.manylinux || 'auto' }} - container: ${{ matrix.container }} - # `flavor=ft` builds drop the abi3 cargo feature so the resulting - # wheel is non-abi3 (free-threaded Python can't load limited-API - # extensions). `flavor=abi3` builds use defaults. - args: >- - --release --out dist - --interpreter ${{ matrix.flavor == 'ft' && '3.14t' || matrix.interpreter || '3.14' }} - ${{ matrix.flavor == 'ft' && '--no-default-features --features ext-module' || '' }} + working-directory: bindings/python + target: ${{ matrix.config.target }} + manylinux: ${{ matrix.config.manylinux || 'auto' }} rust-toolchain: stable - sccache: false - docker-options: -e CI - - - run: ${{ matrix.ls || 'ls -lh' }} dist/ - working-directory: ./bindings/python - - - run: twine check --strict dist/* - working-directory: ./bindings/python - - - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + # Never cache the build on a tag + sccache: ${{ !startsWith(github.ref, 'refs/tags/') }} + args: >- + --release + --locked + --compatibility pypi + --out dist + -i ${{ matrix.config.python }} + env: + RUSTFLAGS: ${{ runner.os == 'macOS' && '-C link-arg=-undefined -C link-arg=dynamic_lookup' || '' }} + - name: upload wheels + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 with: - name: pypi_files-${{ matrix.os }}-${{ matrix.target }}-${{ matrix.manylinux }}-${{ matrix.flavor || 'abi3' }} - path: ./bindings/python/dist - build-sdist: + name: wheels-${{ matrix.config.runner }}-${{ matrix.config.target }}-${{ matrix.config.manylinux || 'host' }}-${{ matrix.config.python }} + path: bindings/python/dist + if-no-files-found: error + - name: twine check + run: | + ls -lh dist + uvx twine check --strict dist/* + + test-wheel: + needs: ["build-wheel"] + name: Test Python wheel - ${{ matrix.config.runner }} - ${{ matrix.config.python }} + runs-on: ${{ matrix.config.runner }} permissions: contents: read - name: build sdist - needs: [lock_exists] - runs-on: ubuntu-latest + timeout-minutes: 20 + defaults: + run: + shell: bash + working-directory: bindings/python + strategy: + fail-fast: false + matrix: + config: + # ---- Linux x86_64 ------------------------------------------------------------------ + - { runner: ubuntu-latest, python: "3.10" } + - { runner: ubuntu-latest, python: "3.14" } + - { runner: ubuntu-latest, python: "3.14t" } + # ---- macOS arm64 ------------------------------------------------------------------- + - { runner: macos-latest, python: "3.10" } + - { runner: macos-latest, python: "3.14" } + - { runner: macos-latest, python: "3.14t" } + # ---- Other native architectures ---------------------------------------------------- + - { runner: ubuntu-24.04-arm, python: "3.10" } + - { runner: ubuntu-24.04-arm, python: "3.14" } + - { runner: ubuntu-24.04-arm, python: "3.14t" } + - { runner: macos-15-intel, python: "3.10" } + - { runner: macos-15-intel, python: "3.14" } + - { runner: macos-15-intel, python: "3.14t" } steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - - uses: PyO3/maturin-action@e83996d129638aa358a18fbd1dfb82f0b0fb5d3b # v1.51.0 + - uses: astral-sh/setup-uv@bec219d24cd3e171d82865faccec33120bb574f4 # v10.1.0 + - name: download all wheels + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 with: - working-directory: ./bindings/python - command: sdist - args: --out dist - rust-toolchain: stable - - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + pattern: wheels-* + path: bindings/python/dist + merge-multiple: true + - name: Cache HF test data + uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 with: - name: pypi_files-srt - path: ./bindings/python/dist + path: bindings/python/data + key: hf-test-data-python-${{ hashFiles('bindings/python/Makefile') }} + restore-keys: hf-test-data-python- + - name: test the wheel + env: + HF_TOKEN: ${{ secrets.HF_TOKEN }} + run: | + ls dist + # Setup virtual env + uv venv --python ${{ matrix.config.python }} .venv + + source .venv/bin/activate - upload_package: - permissions: - contents: read - name: Upload package to PyPi - runs-on: ubuntu-latest - needs: [build, build-sdist] - env: - PYPI_TOKEN: ${{ secrets.PYPI_TOKEN_DIST }} + # Install the built wheel. + # --no-build: a missing wheel is an error, not a source build + uv pip install tokenizers --no-index --no-deps --no-build --find-links dist --reinstall - steps: - - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + # Install deps and pytest: tokenizers is already satisfied, so this only pulls its + # dependencies and pytest from PyPI. + uv pip install tokenizers pytest - - name: Install Python - uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 - with: - python-version: "3.14" - architecture: x64 + # Download data + make test-data HF="uvx --from huggingface_hub hf" + + # Disable the GIL on the free-threaded interpreter, so a module that re-enables it fails + # instead of passing single-threaded. + if [[ "${{ matrix.config.python }}" == *t ]]; then + export PYTHON_GIL=0 + fi + # Run tests + pytest tests -m "not network" + + publish: + # Tag pushes only + if: startsWith(github.ref, 'refs/tags/v') + needs: ["test-wheel", "build-sdist"] + name: Publish to PyPI + runs-on: ubuntu-latest + environment: + name: pypi + url: https://pypi.org/p/tokenizers + permissions: + id-token: write + steps: - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 with: - path: ./bindings/python/dist + pattern: wheels-* + path: dist merge-multiple: true - - - name: Upload to PyPi - working-directory: ./bindings/python - run: | - pip install twine - twine upload dist/* -u __token__ -p "$PYPI_TOKEN" + # Defaults left on: verify-metadata runs twine check over dist/, attestations uploads a + # PEP 740 attestation per file (only possible under Trusted Publishing). + - uses: pypa/gh-action-pypi-publish@dc37677b2e1c63e2034f94d8a5b11f265b73ba33 # v1.14.2 + with: + packages-dir: dist + # PyPI files are immutable, so a re-run after a half-failed upload must skip what landed. + skip-existing: true diff --git a/bindings/python/Makefile b/bindings/python/Makefile index d6e27684bf..1fa087b994 100644 --- a/bindings/python/Makefile +++ b/bindings/python/Makefile @@ -14,10 +14,12 @@ stubs: ruff check --fix --select I python/tokenizers/tokenizers.pyi ruff format python/tokenizers/tokenizers.pyi -test: develop $(TESTS_RESOURCES) +test-data: $(TESTS_RESOURCES) + +test: develop test-data python -m pytest -s -v tests -examples: develop $(TESTS_RESOURCES) +examples: develop test-data for example in examples/*.py; do python $$example; done $(DATA_DIR)/%: From 5fea79aa68f1d1f392e06920cedb291d02af2adb Mon Sep 17 00:00:00 2001 From: SBrandeis <33657802+SBrandeis@users.noreply.github.com> Date: Fri, 18 Sep 2026 17:39:59 +0200 Subject: [PATCH 2/4] remove RELEASE.md (was badly outdated) --- RELEASE.md | 90 ------------------------------------------------------ 1 file changed, 90 deletions(-) delete mode 100644 RELEASE.md diff --git a/RELEASE.md b/RELEASE.md deleted file mode 100644 index bbd6b0e787..0000000000 --- a/RELEASE.md +++ /dev/null @@ -1,90 +0,0 @@ -## How to release - -# Before the release - -Simple checklist on how to make releases for `tokenizers`. - -- Freeze `master` branch. -- Run all tests (Check CI has properly run) -- If any significant work, check benchmarks: - - `cd tokenizers && cargo bench` (needs to be run on latest release tag to measure difference if it's your first time) -- Run all `transformers` tests. (`transformers` is a big user of `tokenizers` we need - to make sure we don't break it, testing is one way to make sure nothing unforeseen - has been done.) - - Run all fast tests at the VERY least (not just the tokenization tests). (`RUN_PIPELINE_TESTS=1 CUDA_VISIBLE_DEVICES=-1 pytest -sv tests/`) - - When all *fast* tests work, then we can also (it's recommended) run the whole `transformers` - test suite. - - Rebase this [PR](https://github.com/huggingface/transformers/pull/16708). - This will create new docker images ready to run the tests suites with `tokenizers` from the main branch. - - Wait for actions to finish - - Rebase this [PR](https://github.com/huggingface/transformers/pull/16712) - This will run the actual full test suite. - - Check the results. -- **If any breaking change has been done**, make sure the version can safely be increased for transformers users (`tokenizers` version need to make sure users don't upgrade before `transformers` has). [link](https://github.com/huggingface/transformers/blob/main/setup.py#L154) - For instance `tokenizers>=0.10,<0.11` so we can safely upgrade to `0.11` without impacting - current users -- Then start a new PR containing all desired code changes from the following steps. -- You will `Create release` after the code modifications are on `master`. - -# Rust - -- `tokenizers` (rust, python & node) versions don't have to be in sync but it's - very common to release for all versions at once for new features. -- Edit `Cargo.toml` to reflect new version -- Edit `CHANGELOG.md`: - - Add relevant PRs that were added (python PRs do not belong for instance). - - Add links at the end of the files. -- Go to [Releases](https://github.com/huggingface/tokenizers/releases) -- Create new Release: - - Mark it as pre-release - - Use new version name with a new tag (create on publish) `vX.X.X`. - - Copy paste the new part of the `CHANGELOG.md` -- ⚠️ Click on `Publish release`. This will start the whole process of building a uploading - the new version on `crates.io`, there's no going back after this -- Go to the [Actions](https://github.com/huggingface/tokenizers/actions) tab and check everything works smoothly. -- If anything fails, you need to fix the CI/CD to make it work again. Since your package was not uploaded to the repository properly, you can try again. - - -# Python - -- Edit `bindings/python/setup.py` to reflect new version. -- Edit `bindings/python/py_src/tokenizers/__init__.py` to reflect new version. -- Edit `CHANGELOG.md`: - - Add relevant PRs that were added (node PRs do not belong for instance). - - Add links at the end of the files. -- Go to [Releases](https://github.com/huggingface/tokenizers/releases) -- Create new Release: - - Mark it as pre-release - - Use new version name with a new tag (create on publish) `python-vX.X.X`. - - Copy paste the new part of the `CHANGELOG.md` -- ⚠️ Click on `Publish release`. This will start the whole process of building a uploading - the new version on `pypi`, there's no going back after this -- Go to the [Actions](https://github.com/huggingface/tokenizers/actions) tab and check everything works smoothly. -- If anything fails, you need to fix the CI/CD to make it work again. Since your package was not uploaded to the repository properly, you can try again. -- This CI/CD has 3 distinct builds, `Pypi`(normal), `conda` and `extra`. `Extra` is REALLY slow (~4h), this is normal since it has to rebuild many things, but enables the wheel to be available for old Linuxes - -# Node - -- Edit `bindings/node/package.json` to reflect new version. -- Edit `CHANGELOG.md`: - - Add relevant PRs that were added (python PRs do not belong for instance). - - Add links at the end of the files. -- Go to [Releases](https://github.com/huggingface/tokenizers/releases) -- Create new Release: - - Mark it as pre-release - - Use new version name with a new tag (create on publish) `node-vX.X.X`. - - Copy paste the new part of the `CHANGELOG.md` -- ⚠️ Click on `Publish release`. This will start the whole process of building a uploading - the new version on `npm`, there's no going back after this -- Go to the [Actions](https://github.com/huggingface/tokenizers/actions) tab and check everything works smoothly. -- If anything fails, you need to fix the CI/CD to make it work again. Since your package was not uploaded to the repository properly, you can try again. - - -# Testing the CI/CD for release - - -If you want to make modifications to the CI/CD of the release GH actions, you need -to : -- **Comment the part that uploads the artifacts** to `crates.io`, `PyPi` or `npm`. -- Change the trigger mechanism so it can trigger every time you push to your branch. -- Keep pushing your changes until the artifacts are properly created. From 07911f6dee41af928bfe6ee797f7def153fc5c2d Mon Sep 17 00:00:00 2001 From: SBrandeis <33657802+SBrandeis@users.noreply.github.com> Date: Fri, 18 Sep 2026 17:46:47 +0200 Subject: [PATCH 3/4] rename bitsplit --- .../workflows/{bitsplit.yml => bitcanon.yml} | 18 ++++----- .github/workflows/rust-release.yml | 8 ++-- README.md | 12 +++--- bindings/node/Cargo.lock | 20 +++++----- bindings/python/Cargo.lock | 20 +++++----- tokenizers/Cargo.lock | 22 +++++------ tokenizers/Cargo.toml | 2 +- tokenizers/README.md | 2 +- tokenizers/{bitsplit => bitcanon}/Cargo.toml | 8 ++-- .../mask_splitter_spec.md | 16 ++++---- .../{bitsplit => bitcanon}/src/classes.rs | 0 .../src/classify/atom_tables.rs | 0 .../src/classify/avx.rs | 0 .../src/classify/mod.rs | 2 +- .../src/classify/neon.rs | 0 .../src/classify/tables.rs | 0 .../src/classify/wasm.rs | 0 tokenizers/{bitsplit => bitcanon}/src/lib.rs | 20 +++++----- .../{bitsplit => bitcanon}/src/literal.rs | 0 .../src/models/cl100k.rs | 4 +- .../src/models/deepseek.rs | 6 +-- .../src/models/family_gpt.rs | 0 .../src/models/family_o200k.rs | 0 .../{bitsplit => bitcanon}/src/models/gpt2.rs | 4 +- .../{bitsplit => bitcanon}/src/models/kimi.rs | 2 +- .../{bitsplit => bitcanon}/src/models/mod.rs | 0 .../src/models/o200k.rs | 2 +- .../src/models/tekken.rs | 2 +- .../{bitsplit => bitcanon}/src/regexes.rs | 4 +- .../{bitsplit => bitcanon}/src/simd/mod.rs | 0 .../{bitsplit => bitcanon}/src/simd/neon.rs | 2 +- .../{bitsplit => bitcanon}/src/simd/x86.rs | 0 .../{bitsplit => bitcanon}/tests/classes.rs | 6 +-- .../{bitsplit => bitcanon}/tests/parity.rs | 20 +++++----- tokenizers/bitmap_gen/Cargo.toml | 2 +- tokenizers/bitmap_gen/src/lib.rs | 4 +- tokenizers/bitmap_gen/src/main.rs | 6 +-- tokenizers/src/lib.rs | 2 +- tokenizers/tk-convert/Cargo.toml | 4 +- tokenizers/tk-convert/src/convert.rs | 2 +- tokenizers/tk-convert/tests/convert.rs | 2 +- tokenizers/tk-encode/Cargo.toml | 2 +- tokenizers/tk-encode/README.md | 2 +- tokenizers/tk-encode/src/lib.rs | 2 +- .../tk-encode/src/pre_tokenizers/bert.rs | 10 ++--- .../tk-encode/src/pre_tokenizers/delimiter.rs | 6 +-- .../tk-encode/src/pre_tokenizers/digits.rs | 10 ++--- .../src/pre_tokenizers/punctuation.rs | 6 +-- .../tk-encode/src/pre_tokenizers/sequence.rs | 6 +-- .../tk-encode/src/pre_tokenizers/split.rs | 10 ++--- .../src/pre_tokenizers/whitespace.rs | 18 ++++----- tokenizers/tk-encode/src/tokenizer/pattern.rs | 4 +- .../tk-encode/src/tokenizer/pipeline/mod.rs | 2 +- .../src/tokenizer/pipeline/pre_tokenizer.rs | 8 ++-- .../src/tokenizer/pipeline/scratch_pool.rs | 2 +- tokenizers/tk-encode/src/utils/byte_level.rs | 4 +- tokenizers/tk-encode/src/utils/mod.rs | 8 ++-- tokenizers/tk-encode/src/utils/no_regex.rs | 4 +- tokenizers/tk-encode/src/utils/search.rs | 2 +- .../tk-encode/src/utils/unrolled_regex.rs | 38 +++++++++---------- tokenizers/tk-serialize/Cargo.toml | 4 +- 61 files changed, 186 insertions(+), 186 deletions(-) rename .github/workflows/{bitsplit.yml => bitcanon.yml} (82%) rename tokenizers/{bitsplit => bitcanon}/Cargo.toml (91%) rename tokenizers/{bitsplit => bitcanon}/mask_splitter_spec.md (97%) rename tokenizers/{bitsplit => bitcanon}/src/classes.rs (100%) rename tokenizers/{bitsplit => bitcanon}/src/classify/atom_tables.rs (100%) rename tokenizers/{bitsplit => bitcanon}/src/classify/avx.rs (100%) rename tokenizers/{bitsplit => bitcanon}/src/classify/mod.rs (99%) rename tokenizers/{bitsplit => bitcanon}/src/classify/neon.rs (100%) rename tokenizers/{bitsplit => bitcanon}/src/classify/tables.rs (100%) rename tokenizers/{bitsplit => bitcanon}/src/classify/wasm.rs (100%) rename tokenizers/{bitsplit => bitcanon}/src/lib.rs (98%) rename tokenizers/{bitsplit => bitcanon}/src/literal.rs (100%) rename tokenizers/{bitsplit => bitcanon}/src/models/cl100k.rs (99%) rename tokenizers/{bitsplit => bitcanon}/src/models/deepseek.rs (98%) rename tokenizers/{bitsplit => bitcanon}/src/models/family_gpt.rs (100%) rename tokenizers/{bitsplit => bitcanon}/src/models/family_o200k.rs (100%) rename tokenizers/{bitsplit => bitcanon}/src/models/gpt2.rs (95%) rename tokenizers/{bitsplit => bitcanon}/src/models/kimi.rs (97%) rename tokenizers/{bitsplit => bitcanon}/src/models/mod.rs (100%) rename tokenizers/{bitsplit => bitcanon}/src/models/o200k.rs (97%) rename tokenizers/{bitsplit => bitcanon}/src/models/tekken.rs (96%) rename tokenizers/{bitsplit => bitcanon}/src/regexes.rs (97%) rename tokenizers/{bitsplit => bitcanon}/src/simd/mod.rs (100%) rename tokenizers/{bitsplit => bitcanon}/src/simd/neon.rs (99%) rename tokenizers/{bitsplit => bitcanon}/src/simd/x86.rs (100%) rename tokenizers/{bitsplit => bitcanon}/tests/classes.rs (96%) rename tokenizers/{bitsplit => bitcanon}/tests/parity.rs (95%) diff --git a/.github/workflows/bitsplit.yml b/.github/workflows/bitcanon.yml similarity index 82% rename from .github/workflows/bitsplit.yml rename to .github/workflows/bitcanon.yml index c049ea353c..9c0b430cd3 100644 --- a/.github/workflows/bitsplit.yml +++ b/.github/workflows/bitcanon.yml @@ -1,10 +1,10 @@ -name: bitsplit +name: bitcanon on: push: - paths: ["tokenizers/bitsplit/**", "tokenizers/bitmap_gen/**", ".github/workflows/bitsplit.yml"] + paths: ["tokenizers/bitcanon/**", "tokenizers/bitmap_gen/**", ".github/workflows/bitcanon.yml"] pull_request: - paths: ["tokenizers/bitsplit/**", "tokenizers/bitmap_gen/**", ".github/workflows/bitsplit.yml"] + paths: ["tokenizers/bitcanon/**", "tokenizers/bitmap_gen/**", ".github/workflows/bitcanon.yml"] defaults: run: @@ -22,11 +22,11 @@ jobs: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - uses: dtolnay/rust-toolchain@29eef336d9b2848a0b548edc03f92a220660cdb8 # stable with: { components: rustfmt, clippy } - - run: cargo fmt -p bitsplit -p bitmap_gen -- --check - - run: cargo clippy -p bitsplit --all-targets -- -D warnings - - run: cargo test -p bitsplit # unit + parity gates (cl100k/deepseek/byte_level vs onig) + - run: cargo fmt -p bitcanon -p bitmap_gen -- --check + - run: cargo clippy -p bitcanon --all-targets -- -D warnings + - run: cargo test -p bitcanon # unit + parity gates (cl100k/deepseek/byte_level vs onig) # committed classify tables must match the generator - - run: cargo run -p bitmap_gen && git diff --exit-code -- bitsplit/src/classify/atom_tables.rs + - run: cargo run -p bitmap_gen && git diff --exit-code -- bitcanon/src/classify/atom_tables.rs # AVX-512 classify path — GitHub runners often lack AVX-512, so emulate it with Intel SDE and run the # SAME byte-exact test binaries under it. @@ -48,7 +48,7 @@ jobs: echo "$PWD/${SDE_VER}" >> "$GITHUB_PATH" - name: Run tests under SDE (-future = AVX-512) run: | - cargo test -p bitsplit --no-run --message-format=json \ + cargo test -p bitcanon --no-run --message-format=json \ | jq -r 'select(.profile.test == true) | .executable | select(. != null)' \ | while read -r bin; do echo "SDE: $bin"; sde64 -future -- "$bin"; done @@ -66,4 +66,4 @@ jobs: env: RUSTFLAGS: "-C target-feature=+simd128" CARGO_TARGET_WASM32_WASIP1_RUNNER: "wasmtime run --" - run: cargo test -p bitsplit --target wasm32-wasip1 + run: cargo test -p bitcanon --target wasm32-wasip1 diff --git a/.github/workflows/rust-release.yml b/.github/workflows/rust-release.yml index 67614b222a..75c61c4920 100644 --- a/.github/workflows/rust-release.yml +++ b/.github/workflows/rust-release.yml @@ -28,7 +28,7 @@ jobs: path: ~/.cargo/registry key: ubuntu-latest-cargo-registry-${{ hashFiles('**/Cargo.toml') }} - # bitsplit ships committed classify tables (no build script). Regenerate them here — this + # bitcanon ships committed classify tables (no build script). Regenerate them here — this # self-validates every codepoint against the reference atom() and fails the release if the # committed table is stale (someone changed the scheme without `cargo run -p bitmap_gen`). # The `--` matters: without it a path that no longer exists is read as a revision, and @@ -37,11 +37,11 @@ jobs: working-directory: ./tokenizers run: | cargo run -p bitmap_gen - git diff --exit-code -- bitsplit/src/classify/atom_tables.rs \ - || { echo "::error::bitsplit/src/classify/atom_tables.rs is stale — run 'cargo run -p bitmap_gen' and commit"; exit 1; } + git diff --exit-code -- bitcanon/src/classify/atom_tables.rs \ + || { echo "::error::bitcanon/src/classify/atom_tables.rs is stale — run 'cargo run -p bitmap_gen' and commit"; exit 1; } # One command, not one step per crate: `--workspace` publishes the members in dependency - # order (bitsplit, then tk-encode/tk-convert, then tk-serialize, then tokenizers) and waits + # order (bitcanon, then tk-encode/tk-convert, then tk-serialize, then tokenizers) and waits # for each to appear on the index before starting the next, so tk-serialize can resolve the # tk-encode it just uploaded. bitmap_gen is skipped for us: it is `publish = false`. # No rc gate: a caret requirement never resolves a prerelease unless asked for by name. diff --git a/README.md b/README.md index c51e072ad5..481a1d87e8 100644 --- a/README.md +++ b/README.md @@ -67,7 +67,7 @@ tokenizer.padding = None # get/set padding ### Remaining before 1.0.0 -- Improve `bitsplit` +- Improve `bitcanon` - Bring training back - Apply performance improvement to trainer - Bring back offset output @@ -137,7 +137,7 @@ the 325 KB story.
-bitsplit — SIMD pre-tokenization +bitcanon — SIMD pre-tokenization Unicode atom classification plus pre-tokenization as a **bitstream program** rather than a scalar FSM. Follows *Interleaved Bitstream Execution for Multi-Pattern Regex Matching on GPUs* @@ -158,7 +158,7 @@ feeds every grammar, and a grammar pays only for the distinctions it actually as regex engine has no such shared vocabulary to compile against. **Zero Unicode dependencies at runtime.** Those tables are committed source, baked offline by -`bitmap_gen` (below) from `unicode-properties`. `bitsplit`'s entire runtime dependency list is +`bitmap_gen` (below) from `unicode-properties`. `bitcanon`'s entire runtime dependency list is `ahash` — no `unicode-*` crate, no build script, so nothing that ships carries a Unicode table crate or rebuilds one. @@ -202,14 +202,14 @@ They have opposite constraints, so they get opposite dependency budgets.
bitmap_gen — dev-only table generator -`cargo run -p bitmap_gen` regenerates `bitsplit`'s committed classify tables from +`cargo run -p bitmap_gen` regenerates `bitcanon`'s committed classify tables from `unicode-properties`, emitting one `Atom` tag per codepoint. **Why separate — this is what buys the zero Unicode dependency.** `unicode-properties` is a dependency of *this* crate and of nothing else: the tables it produces are checked into -`bitsplit/src/classify/atom_tables.rs` as ordinary source, so the Unicode data is resolved once, at +`bitcanon/src/classify/atom_tables.rs` as ordinary source, so the Unicode data is resolved once, at development time, by a crate that is never linked into anything that ships and is never published. -No build script either — `bitsplit` compiles with no code generation step, and a release binary +No build script either — `bitcanon` compiles with no code generation step, and a release binary contains the tags without containing a Unicode crate to derive them. The release workflow re-runs the generator and fails if the committed table differs, so "baked" cannot silently mean "stale".
diff --git a/bindings/node/Cargo.lock b/bindings/node/Cargo.lock index 4eb80733a0..134102c63e 100644 --- a/bindings/node/Cargo.lock +++ b/bindings/node/Cargo.lock @@ -67,18 +67,18 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "72b3254f16251a8381aa12e40e3c4d2f0199f8c6508fbecb9d91f575e0fbb8c6" [[package]] -name = "bitflags" -version = "2.13.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b4388bee8683e3d04af747c73422af53102d2bd24d9eadb6cbc100baef4b43f8" - -[[package]] -name = "bitsplit" +name = "bitcanon" version = "0.1.0-dev.0" dependencies = [ "ahash", ] +[[package]] +name = "bitflags" +version = "2.13.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b4388bee8683e3d04af747c73422af53102d2bd24d9eadb6cbc100baef4b43f8" + [[package]] name = "bitvec" version = "1.1.1" @@ -1192,7 +1192,7 @@ checksum = "1f3ccbac311fea05f86f61904b462b55fb3df8837a366dfc601a0161d0532f20" name = "tk-convert" version = "0.1.0-dev.0" dependencies = [ - "bitsplit", + "bitcanon", "serde_json", "thiserror", ] @@ -1202,7 +1202,7 @@ name = "tk-encode" version = "0.1.0-dev.0" dependencies = [ "ahash", - "bitsplit", + "bitcanon", "dary_heap", "itertools 0.15.0", "libc", @@ -1228,7 +1228,7 @@ name = "tk-serialize" version = "0.1.0-dev.0" dependencies = [ "base64 0.22.1", - "bitsplit", + "bitcanon", "hifijson", "tk-encode", ] diff --git a/bindings/python/Cargo.lock b/bindings/python/Cargo.lock index 5bad694efe..5d0376b133 100644 --- a/bindings/python/Cargo.lock +++ b/bindings/python/Cargo.lock @@ -67,18 +67,18 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "72b3254f16251a8381aa12e40e3c4d2f0199f8c6508fbecb9d91f575e0fbb8c6" [[package]] -name = "bitflags" -version = "2.13.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b588b76d00fde79687d7646a9b5bdf3cc0f655e0bbd080335a95d7e96f3587da" - -[[package]] -name = "bitsplit" +name = "bitcanon" version = "0.1.0-dev.0" dependencies = [ "ahash", ] +[[package]] +name = "bitflags" +version = "2.13.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b588b76d00fde79687d7646a9b5bdf3cc0f655e0bbd080335a95d7e96f3587da" + [[package]] name = "bitvec" version = "1.1.1" @@ -1181,7 +1181,7 @@ checksum = "1f3ccbac311fea05f86f61904b462b55fb3df8837a366dfc601a0161d0532f20" name = "tk-convert" version = "0.1.0-dev.0" dependencies = [ - "bitsplit", + "bitcanon", "serde_json", "thiserror", ] @@ -1191,7 +1191,7 @@ name = "tk-encode" version = "0.1.0-dev.0" dependencies = [ "ahash", - "bitsplit", + "bitcanon", "dary_heap", "itertools 0.15.0", "libc", @@ -1217,7 +1217,7 @@ name = "tk-serialize" version = "0.1.0-dev.0" dependencies = [ "base64 0.22.1", - "bitsplit", + "bitcanon", "hifijson", "tk-encode", ] diff --git a/tokenizers/Cargo.lock b/tokenizers/Cargo.lock index f19be2200f..2c161841b3 100644 --- a/tokenizers/Cargo.lock +++ b/tokenizers/Cargo.lock @@ -187,6 +187,14 @@ version = "0.8.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "5e764a1d40d510daf35e07be9eb06e75770908c27d411ee6c92109c9840eaaf7" +[[package]] +name = "bitcanon" +version = "0.1.0-dev.0" +dependencies = [ + "ahash", + "onig", +] + [[package]] name = "bitflags" version = "2.13.1" @@ -200,14 +208,6 @@ dependencies = [ "unicode-properties", ] -[[package]] -name = "bitsplit" -version = "0.1.0-dev.0" -dependencies = [ - "ahash", - "onig", -] - [[package]] name = "bitvec" version = "1.1.1" @@ -3147,7 +3147,7 @@ checksum = "1f3ccbac311fea05f86f61904b462b55fb3df8837a366dfc601a0161d0532f20" name = "tk-convert" version = "0.1.0-dev.0" dependencies = [ - "bitsplit", + "bitcanon", "criterion 0.8.2", "rayon", "serde_json", @@ -3163,7 +3163,7 @@ version = "0.1.0-dev.0" dependencies = [ "ahash", "assert_approx_eq", - "bitsplit", + "bitcanon", "daachorse 5.0.0", "dary_heap", "fancy-regex", @@ -3198,7 +3198,7 @@ name = "tk-serialize" version = "0.1.0-dev.0" dependencies = [ "base64 0.22.1", - "bitsplit", + "bitcanon", "criterion 0.8.2", "hifijson", "serde_json", diff --git a/tokenizers/Cargo.toml b/tokenizers/Cargo.toml index 542da36a1b..59c71e144f 100644 --- a/tokenizers/Cargo.toml +++ b/tokenizers/Cargo.toml @@ -1,6 +1,6 @@ [workspace] resolver = "3" -members = ["bitmap_gen", "bitsplit", "tk-encode", "tk-serialize", "tk-convert"] +members = ["bitmap_gen", "bitcanon", "tk-encode", "tk-serialize", "tk-convert"] exclude = ["tk-train"] [package] diff --git a/tokenizers/README.md b/tokenizers/README.md index 0cb71ec4f9..3abd0f21be 100644 --- a/tokenizers/README.md +++ b/tokenizers/README.md @@ -18,7 +18,7 @@ The 🤗 Tokenizers library. The implementation is split across crates (each built on internal engines — `tk_encode` on the -`bitsplit` SIMD pre-tokenizer, and the shared `bitmap_gen` tables): +`bitcanon` SIMD pre-tokenizer, and the shared `bitmap_gen` tables): - [`tk_encode`] — inference: the model engines and the full pipeline components ([`Normalizer`], [`PreTokenizer`], [`Model`], [`PostProcessor`], [`Decoder`]). diff --git a/tokenizers/bitsplit/Cargo.toml b/tokenizers/bitcanon/Cargo.toml similarity index 91% rename from tokenizers/bitsplit/Cargo.toml rename to tokenizers/bitcanon/Cargo.toml index 9f0a98997b..ee22f22b6a 100644 --- a/tokenizers/bitsplit/Cargo.toml +++ b/tokenizers/bitcanon/Cargo.toml @@ -1,5 +1,5 @@ [package] -name = "bitsplit" +name = "bitcanon" version = "0.1.0-dev.0" edition = "2024" rust-version = "1.89" # AVX-512 VBMI intrinsics (simd_avx_classify) are stable only since 1.89 @@ -10,7 +10,7 @@ authors = [ ] homepage = "https://github.com/huggingface/tokenizers" repository = "https://github.com/huggingface/tokenizers" -documentation = "https://docs.rs/bitsplit/" +documentation = "https://docs.rs/bitcanon/" license = "Apache-2.0" keywords = ["tokenizer", "nlp", "pretokenizer", "simd", "unicode"] categories = ["text-processing", "parsing"] @@ -21,13 +21,13 @@ exclude = ["benches/data/"] all-features = true [lib] -name = "bitsplit" +name = "bitcanon" path = "src/lib.rs" [dependencies] ahash = "0.8" # NOTE: classify tables live in src/classify/atom_tables.rs (committed, generated). Regenerate after any atom -# scheme change with `cargo run -p bitmap_gen`. No build script / build-dep — bitsplit builds clean. +# scheme change with `cargo run -p bitmap_gen`. No build script / build-dep — bitcanon builds clean. # onig is C (Oniguruma) and has no wasi libc to build against, so the parity oracle is gated off wasm32. [target.'cfg(not(target_arch = "wasm32"))'.dev-dependencies] diff --git a/tokenizers/bitsplit/mask_splitter_spec.md b/tokenizers/bitcanon/mask_splitter_spec.md similarity index 97% rename from tokenizers/bitsplit/mask_splitter_spec.md rename to tokenizers/bitcanon/mask_splitter_spec.md index ebe8a8a345..d55e1a0e7e 100644 --- a/tokenizers/bitsplit/mask_splitter_spec.md +++ b/tokenizers/bitcanon/mask_splitter_spec.md @@ -1,4 +1,4 @@ -# bitsplit — mask splitter spec +# bitcanon — mask splitter spec A Rust library for **pre-tokenizer splitting as a bitstream program**. Not a general regex engine: it covers exactly what `tokenizers` needs — the GPT-family regexes, the class-run family, and @@ -10,8 +10,8 @@ processors*) for the operator set and carry discipline; *Interleaved Bitstream E Multi-Pattern Regex Matching on GPUs* (MICRO'25, doi 10.1145/3725843.3756052) for the execution model — fuse every instruction into ONE block-wise loop instead of one pass per instruction. -Status: proven in `scratch/bitsplit` for three grammars, all byte-exact against their -`bitsplit` grammar counterparts. +Status: proven in `scratch/bitcanon` for three grammars, all byte-exact against their +`bitcanon` grammar counterparts. --- @@ -25,12 +25,12 @@ program decides 64 bytes per register op, branchlessly, so its cost is flat. fsm_deepseek ████████████████████████████████████████████████ 2.93 4.7 B/tok fsm_cl100k ██████████████████████████████ 1.86 fsm_byte_level ████████████████████████████ 1.71 - bitsplit ds █████████ 0.58 - bitsplit cl100k████████ 0.50 - bitsplit bl ██████ 0.41 + bitcanon ds █████████ 0.58 + bitcanon cl100k████████ 0.50 + bitcanon bl ██████ 0.41 ``` -The FSM spread across grammars is 71% (1.71 → 2.93); bitsplit's is 20%. **Flat cost across grammars +The FSM spread across grammars is 71% (1.71 → 2.93); bitcanon's is 20%. **Flat cost across grammars is the design invariant** — it means the per-grammar layer is thin, which is what makes the library worth having. @@ -42,7 +42,7 @@ worth having. text ─────────────────────────────────────────────────────────────┐ │ │ ▼ │ - classify (bitsplit) one Atom tag per byte │ + classify (bitcanon) one Atom tag per byte │ │ │ ▼ ▼ ┌─────────────────────── per 64-byte block, fused ────────────────────────┐ diff --git a/tokenizers/bitsplit/src/classes.rs b/tokenizers/bitcanon/src/classes.rs similarity index 100% rename from tokenizers/bitsplit/src/classes.rs rename to tokenizers/bitcanon/src/classes.rs diff --git a/tokenizers/bitsplit/src/classify/atom_tables.rs b/tokenizers/bitcanon/src/classify/atom_tables.rs similarity index 100% rename from tokenizers/bitsplit/src/classify/atom_tables.rs rename to tokenizers/bitcanon/src/classify/atom_tables.rs diff --git a/tokenizers/bitsplit/src/classify/avx.rs b/tokenizers/bitcanon/src/classify/avx.rs similarity index 100% rename from tokenizers/bitsplit/src/classify/avx.rs rename to tokenizers/bitcanon/src/classify/avx.rs diff --git a/tokenizers/bitsplit/src/classify/mod.rs b/tokenizers/bitcanon/src/classify/mod.rs similarity index 99% rename from tokenizers/bitsplit/src/classify/mod.rs rename to tokenizers/bitcanon/src/classify/mod.rs index ca8c7b4ed5..24d958d438 100644 --- a/tokenizers/bitsplit/src/classify/mod.rs +++ b/tokenizers/bitcanon/src/classify/mod.rs @@ -137,7 +137,7 @@ pub fn classify(text: &[u8], tags: &mut [u8]) { // check guards every arch path below. assert!( tags.len() >= text.len(), - "bitsplit::classify: `tags` shorter than `text`" + "bitcanon::classify: `tags` shorter than `text`" ); #[cfg(target_arch = "aarch64")] // SAFETY: `tags.len() >= text.len()` (asserted above); NEON vld1q/vst1q are alignment-free. diff --git a/tokenizers/bitsplit/src/classify/neon.rs b/tokenizers/bitcanon/src/classify/neon.rs similarity index 100% rename from tokenizers/bitsplit/src/classify/neon.rs rename to tokenizers/bitcanon/src/classify/neon.rs diff --git a/tokenizers/bitsplit/src/classify/tables.rs b/tokenizers/bitcanon/src/classify/tables.rs similarity index 100% rename from tokenizers/bitsplit/src/classify/tables.rs rename to tokenizers/bitcanon/src/classify/tables.rs diff --git a/tokenizers/bitsplit/src/classify/wasm.rs b/tokenizers/bitcanon/src/classify/wasm.rs similarity index 100% rename from tokenizers/bitsplit/src/classify/wasm.rs rename to tokenizers/bitcanon/src/classify/wasm.rs diff --git a/tokenizers/bitsplit/src/lib.rs b/tokenizers/bitcanon/src/lib.rs similarity index 98% rename from tokenizers/bitsplit/src/lib.rs rename to tokenizers/bitcanon/src/lib.rs index 9455161fc9..8bcf4d7b11 100644 --- a/tokenizers/bitsplit/src/lib.rs +++ b/tokenizers/bitcanon/src/lib.rs @@ -1,4 +1,4 @@ -//! `bitsplit` — GPT-family pre-tokenization as a **bitstream program**, replacing the scalar FSMs. +//! `bitcanon` — GPT-family pre-tokenization as a **bitstream program**, replacing the scalar FSMs. //! //! Follows *Interleaved Bitstream Execution for Multi-Pattern Regex Matching on GPUs* //! (MICRO'25, doi 10.1145/3725843.3756052). The paper's two ideas that carry over to a CPU: @@ -21,7 +21,7 @@ //! builder folds those 16 atoms into a grammar-specific **dense 3-bit code** and extracts 3 //! bit-planes of it; every class stream is then a 2–3 op boolean function of the planes. //! -//! Grammars: [`bitsplit_deepseek`], [`bitsplit_byte_level`] (GPT-2), [`bitsplit_cl100k`]. All three +//! Grammars: [`bitcanon_deepseek`], [`bitcanon_byte_level`] (GPT-2), [`bitcanon_cl100k`]. All three //! byte-exact with the oniguruma oracle over a block-phase sweep — see `tests/parity.rs`. /// Declares a grammar's block-local class streams, and with them the two carried shifts every @@ -203,12 +203,12 @@ pub mod models; pub mod regexes; mod simd; -pub use models::cl100k::{bitsplit_cl100k, bitsplit_qwen}; -pub use models::deepseek::bitsplit_deepseek; -pub use models::gpt2::bitsplit_byte_level; -pub use models::kimi::bitsplit_kimi; -pub use models::o200k::bitsplit_o200k; -pub use models::tekken::bitsplit_tekken; +pub use models::cl100k::{bitcanon_cl100k, bitcanon_qwen}; +pub use models::deepseek::bitcanon_deepseek; +pub use models::gpt2::bitcanon_byte_level; +pub use models::kimi::bitcanon_kimi; +pub use models::o200k::bitcanon_o200k; +pub use models::tekken::bitcanon_tekken; /// A token span: byte offsets `[start, end)` into the input. `#[repr(C)]` so the output buffer has a /// stable `[start, end]` layout — the pipeline reuses it with zero conversion, and it can be @@ -734,11 +734,11 @@ pub fn build_only(text: &[u8], tags: &[u8]) -> u64 { acc } -/// Convenience wrapper: classify + deepseek bitsplit over caller-owned scratch. +/// Convenience wrapper: classify + deepseek bitcanon over caller-owned scratch. #[must_use] pub fn pre_tokenize(text: &[u8], tags: &mut [u8], starts: &mut [u64], out: &mut [Span]) -> usize { classify::classify(text, tags); - bitsplit_deepseek(text, tags, starts, out) + bitcanon_deepseek(text, tags, starts, out) } #[cfg(test)] diff --git a/tokenizers/bitsplit/src/literal.rs b/tokenizers/bitcanon/src/literal.rs similarity index 100% rename from tokenizers/bitsplit/src/literal.rs rename to tokenizers/bitcanon/src/literal.rs diff --git a/tokenizers/bitsplit/src/models/cl100k.rs b/tokenizers/bitcanon/src/models/cl100k.rs similarity index 99% rename from tokenizers/bitsplit/src/models/cl100k.rs rename to tokenizers/bitcanon/src/models/cl100k.rs index dc09d6f68a..6623c662dc 100644 --- a/tokenizers/bitsplit/src/models/cl100k.rs +++ b/tokenizers/bitcanon/src/models/cl100k.rs @@ -14,7 +14,7 @@ use crate::{ /// cl100k_base / Llama-3 / GLM-4.6 — rule 3 is `\p{N}{1,3}`. #[must_use] -pub fn bitsplit_cl100k( +pub fn bitcanon_cl100k( text: &[u8], tags: &[u8], starts: &mut [u64], @@ -26,7 +26,7 @@ pub fn bitsplit_cl100k( /// Qwen2 / Qwen3 — cl100k character-for-character except rule 3 is a bare `\p{N}`. #[must_use] -pub fn bitsplit_qwen( +pub fn bitcanon_qwen( text: &[u8], tags: &[u8], starts: &mut [u64], diff --git a/tokenizers/bitsplit/src/models/deepseek.rs b/tokenizers/bitcanon/src/models/deepseek.rs similarity index 98% rename from tokenizers/bitsplit/src/models/deepseek.rs rename to tokenizers/bitcanon/src/models/deepseek.rs index e02888dd5d..8917165782 100644 --- a/tokenizers/bitsplit/src/models/deepseek.rs +++ b/tokenizers/bitcanon/src/models/deepseek.rs @@ -1,6 +1,6 @@ //! DeepSeek-V3/V4 pre-tokenization: the `Sequence` of `\p{N}{1,3}` → `[一-龥぀-ゟ゠-ヿ]+` → //! the big regex, all `Isolated`, as one bitstream program. Byte-exact with -//! `bitsplit::bitsplit_deepseek`. +//! `bitcanon::bitcanon_deepseek`. use crate::{ AUX_CJK, Anl, CODE_CONT, Digits, Out, Span, blocks, build_block, emit, later_in_run, scanthru, @@ -110,10 +110,10 @@ fn cls( } /// Pre-tokenize `text` (well-formed UTF-8) with the DeepSeek grammar: writes token spans into `out` -/// and returns the count. `tags` is `bitsplit::classify`'s output (len ≥ `text.len()`), `starts` +/// and returns the count. `tags` is `bitcanon::classify`'s output (len ≥ `text.len()`), `starts` /// is scratch for the token-start bitmap (len ≥ `text.len().div_ceil(64)`). #[must_use] -pub fn bitsplit_deepseek(text: &[u8], tags: &[u8], starts: &mut [u64], out: &mut [Span]) -> usize { +pub fn bitcanon_deepseek(text: &[u8], tags: &[u8], starts: &mut [u64], out: &mut [Span]) -> usize { let ntext = text.len(); if ntext == 0 { return 0; diff --git a/tokenizers/bitsplit/src/models/family_gpt.rs b/tokenizers/bitcanon/src/models/family_gpt.rs similarity index 100% rename from tokenizers/bitsplit/src/models/family_gpt.rs rename to tokenizers/bitcanon/src/models/family_gpt.rs diff --git a/tokenizers/bitsplit/src/models/family_o200k.rs b/tokenizers/bitcanon/src/models/family_o200k.rs similarity index 100% rename from tokenizers/bitsplit/src/models/family_o200k.rs rename to tokenizers/bitcanon/src/models/family_o200k.rs diff --git a/tokenizers/bitsplit/src/models/gpt2.rs b/tokenizers/bitcanon/src/models/gpt2.rs similarity index 95% rename from tokenizers/bitsplit/src/models/gpt2.rs rename to tokenizers/bitcanon/src/models/gpt2.rs index f38d9e9028..3bb1a3d013 100644 --- a/tokenizers/bitsplit/src/models/gpt2.rs +++ b/tokenizers/bitcanon/src/models/gpt2.rs @@ -11,9 +11,9 @@ use super::family_gpt::{cls, contractions}; use crate::{CODE_CONT, Out, Span, blocks, emit, to_lead}; /// GPT-2 / byte-level pre-tokenization. `starts` and `flag` are scratch bitmaps -/// (len ≥ `text.len().div_ceil(64)`); byte-exact with `bitsplit::fsm_byte_level`. +/// (len ≥ `text.len().div_ceil(64)`); byte-exact with `bitcanon::fsm_byte_level`. #[must_use] -pub fn bitsplit_byte_level( +pub fn bitcanon_byte_level( text: &[u8], tags: &[u8], starts: &mut [u64], diff --git a/tokenizers/bitsplit/src/models/kimi.rs b/tokenizers/bitcanon/src/models/kimi.rs similarity index 97% rename from tokenizers/bitsplit/src/models/kimi.rs rename to tokenizers/bitcanon/src/models/kimi.rs index 12592ecc7c..17b3aaadad 100644 --- a/tokenizers/bitsplit/src/models/kimi.rs +++ b/tokenizers/bitcanon/src/models/kimi.rs @@ -10,7 +10,7 @@ use crate::{AUX_NONE, Span}; /// kimi-k2 pre-tokenization. #[must_use] -pub fn bitsplit_kimi( +pub fn bitcanon_kimi( text: &[u8], tags: &[u8], starts: &mut [u64], diff --git a/tokenizers/bitsplit/src/models/mod.rs b/tokenizers/bitcanon/src/models/mod.rs similarity index 100% rename from tokenizers/bitsplit/src/models/mod.rs rename to tokenizers/bitcanon/src/models/mod.rs diff --git a/tokenizers/bitsplit/src/models/o200k.rs b/tokenizers/bitcanon/src/models/o200k.rs similarity index 97% rename from tokenizers/bitsplit/src/models/o200k.rs rename to tokenizers/bitcanon/src/models/o200k.rs index 5a3ffc3862..3f20507c74 100644 --- a/tokenizers/bitsplit/src/models/o200k.rs +++ b/tokenizers/bitcanon/src/models/o200k.rs @@ -9,7 +9,7 @@ use crate::{AUX_SLASH, Span}; /// o200k_base / GPT-4o — and byte-for-byte the same regex Llama-4, gpt-oss and MiniMax-M2 ship. #[must_use] -pub fn bitsplit_o200k( +pub fn bitcanon_o200k( text: &[u8], tags: &[u8], starts: &mut [u64], diff --git a/tokenizers/bitsplit/src/models/tekken.rs b/tokenizers/bitcanon/src/models/tekken.rs similarity index 96% rename from tokenizers/bitsplit/src/models/tekken.rs rename to tokenizers/bitcanon/src/models/tekken.rs index 9a876d4cc8..0a2b15037b 100644 --- a/tokenizers/bitsplit/src/models/tekken.rs +++ b/tokenizers/bitcanon/src/models/tekken.rs @@ -7,7 +7,7 @@ use crate::{AUX_SLASH, Span}; /// Mistral tekken pre-tokenization. #[must_use] -pub fn bitsplit_tekken( +pub fn bitcanon_tekken( text: &[u8], tags: &[u8], starts: &mut [u64], diff --git a/tokenizers/bitsplit/src/regexes.rs b/tokenizers/bitcanon/src/regexes.rs similarity index 97% rename from tokenizers/bitsplit/src/regexes.rs rename to tokenizers/bitcanon/src/regexes.rs index d13dbd3247..6eefffad8a 100644 --- a/tokenizers/bitsplit/src/regexes.rs +++ b/tokenizers/bitcanon/src/regexes.rs @@ -2,7 +2,7 @@ //! under an `Isolated` split. This is the single source of truth: the parity oracle //! (`tests/parity.rs`) and tk-encode's runtime recognizer both reference these consts, so the pattern a //! tokenizer ships, the pattern the FSM is tested against, and the pattern the pipeline recognizes can -//! never drift apart. `bitsplit` never *runs* these at runtime — it works off the tag stream; the +//! never drift apart. `bitcanon` never *runs* these at runtime — it works off the tag stream; the //! consts only document (and gate the tests of) the contract the FSMs implement. /// GPT-2 / ByteLevel. Reproduced by [`crate::fsm::fsm_byte_level`]. @@ -35,5 +35,5 @@ pub const DEEPSEEK: &[&str] = &[DEEPSEEK_NUM, DEEPSEEK_CJK, DEEPSEEK_BIG]; /// kimi-k2 / k3 — `moonshotai/Kimi-K2-Instruct`'s `tokenization_kimi.py` `pat_str`. o200k plus a /// leading `[\p{Han}]+` arm, Han subtracted from both letter classes, and a `[\r\n]*` rule-4 tail /// (o200k has `[\r\n/]*`). Kimi ships `tiktoken.model` rather than a `tokenizer.json`, so this is -/// the pattern as a converted tokenizer would spell it. Reproduced by [`crate::bitsplit_kimi`]. +/// the pattern as a converted tokenizer would spell it. Reproduced by [`crate::bitcanon_kimi`]. pub const KIMI_K2: &str = r"[\p{Han}]+|[^\r\n\p{L}\p{N}]?[\p{Lu}\p{Lt}\p{Lm}\p{Lo}\p{M}&&[^\p{Han}]]*[\p{Ll}\p{Lm}\p{Lo}\p{M}&&[^\p{Han}]]+(?i:'s|'t|'re|'ve|'m|'ll|'d)?|[^\r\n\p{L}\p{N}]?[\p{Lu}\p{Lt}\p{Lm}\p{Lo}\p{M}&&[^\p{Han}]]+[\p{Ll}\p{Lm}\p{Lo}\p{M}&&[^\p{Han}]]*(?i:'s|'t|'re|'ve|'m|'ll|'d)?|\p{N}{1,3}| ?[^\s\p{L}\p{N}]+[\r\n]*|\s*[\r\n]+|\s+(?!\S)|\s+"; diff --git a/tokenizers/bitsplit/src/simd/mod.rs b/tokenizers/bitcanon/src/simd/mod.rs similarity index 100% rename from tokenizers/bitsplit/src/simd/mod.rs rename to tokenizers/bitcanon/src/simd/mod.rs diff --git a/tokenizers/bitsplit/src/simd/neon.rs b/tokenizers/bitcanon/src/simd/neon.rs similarity index 99% rename from tokenizers/bitsplit/src/simd/neon.rs rename to tokenizers/bitcanon/src/simd/neon.rs index 4a96f05466..b630d76957 100644 --- a/tokenizers/bitsplit/src/simd/neon.rs +++ b/tokenizers/bitcanon/src/simd/neon.rs @@ -6,7 +6,7 @@ //! of one `u64`. That is ~9 ops per stream per 64 bytes, versus 4 separate 16-bit movemasks. //! //! Continuation bytes are resolved **before** extraction (≤3 `vext`+`vbsl`, the same trick -//! `bitsplit::simd_fsm` uses), so every stream comes out *filled* — a multi-byte char sets its bit +//! `bitcanon::simd_fsm` uses), so every stream comes out *filled* — a multi-byte char sets its bit //! on all of its bytes. That is what lets the bitstream program read "previous char's class" as a //! plain `<< 1` with no char-width arithmetic. diff --git a/tokenizers/bitsplit/src/simd/x86.rs b/tokenizers/bitcanon/src/simd/x86.rs similarity index 100% rename from tokenizers/bitsplit/src/simd/x86.rs rename to tokenizers/bitcanon/src/simd/x86.rs diff --git a/tokenizers/bitsplit/tests/classes.rs b/tokenizers/bitcanon/tests/classes.rs similarity index 96% rename from tokenizers/bitsplit/tests/classes.rs rename to tokenizers/bitcanon/tests/classes.rs index 3c24958e7f..a8d9eee182 100644 --- a/tokenizers/bitsplit/tests/classes.rs +++ b/tokenizers/bitcanon/tests/classes.rs @@ -2,8 +2,8 @@ //! FSM it replaced. Truncating the corpus at every char boundary walks every construct through //! every block phase, which is the only thing that exercises the cross-block carries. -use bitsplit::Span; -use bitsplit::classify::{CONT, char_len, classify, in_mask, mask}; +use bitcanon::Span; +use bitcanon::classify::{CONT, char_len, classify, in_mask, mask}; /// The scalar class-run FSM, verbatim from `classes.rs` before the bitstream port. fn reference(text: &[u8], tags: &[u8], dropm: u16, isolatem: u16, keepa: u16) -> Vec { @@ -47,7 +47,7 @@ fn check(name: &str, text: &[u8]) { let words = text.len().div_ceil(64) + 1; let (mut st, mut fk) = (vec![0u64; words], vec![0u64; words]); let mut got = vec![Span::default(); text.len() + 1]; - let k = bitsplit::classes::class_runs_into::(text, &tags, &mut st, &mut fk, &mut got); + let k = bitcanon::classes::class_runs_into::(text, &tags, &mut st, &mut fk, &mut got); assert_eq!( &got[..k], &want[..], diff --git a/tokenizers/bitsplit/tests/parity.rs b/tokenizers/bitcanon/tests/parity.rs similarity index 95% rename from tokenizers/bitsplit/tests/parity.rs rename to tokenizers/bitcanon/tests/parity.rs index 49d8e9095c..596d2bf52a 100644 --- a/tokenizers/bitsplit/tests/parity.rs +++ b/tokenizers/bitcanon/tests/parity.rs @@ -9,9 +9,9 @@ //! gaps; deepseek's three passes do leave gaps, hence `split_iso`. #![cfg(not(target_arch = "wasm32"))] -use bitsplit::Span; -use bitsplit::classify::classify; -use bitsplit::regexes::{ +use bitcanon::Span; +use bitcanon::classify::classify; +use bitcanon::regexes::{ CL100K, DEEPSEEK_BIG as DS_BIG, DEEPSEEK_CJK as DS_CJK, DEEPSEEK_NUM as DS_NUM, GPT2, KIMI_K2, O200K, TEKKEN, }; @@ -40,7 +40,7 @@ fn bs_deepseek( _l: &mut [u64], o: &mut [Span], ) -> usize { - bitsplit::bitsplit_deepseek(t, g, s, o) + bitcanon::bitcanon_deepseek(t, g, s, o) } fn bs_byte_level( t: &[u8], @@ -50,7 +50,7 @@ fn bs_byte_level( _l: &mut [u64], o: &mut [Span], ) -> usize { - bitsplit::bitsplit_byte_level(t, g, s, f, o) + bitcanon::bitcanon_byte_level(t, g, s, f, o) } fn bs_cl100k( t: &[u8], @@ -60,7 +60,7 @@ fn bs_cl100k( _l: &mut [u64], o: &mut [Span], ) -> usize { - bitsplit::bitsplit_cl100k(t, g, s, f, o) + bitcanon::bitcanon_cl100k(t, g, s, f, o) } fn bs_qwen( t: &[u8], @@ -70,7 +70,7 @@ fn bs_qwen( _l: &mut [u64], o: &mut [Span], ) -> usize { - bitsplit::bitsplit_qwen(t, g, s, f, o) + bitcanon::bitcanon_qwen(t, g, s, f, o) } fn bs_o200k( t: &[u8], @@ -80,7 +80,7 @@ fn bs_o200k( _l: &mut [u64], o: &mut [Span], ) -> usize { - bitsplit::bitsplit_o200k(t, g, s, f, _l, o) + bitcanon::bitcanon_o200k(t, g, s, f, _l, o) } fn bs_tekken( t: &[u8], @@ -90,7 +90,7 @@ fn bs_tekken( _l: &mut [u64], o: &mut [Span], ) -> usize { - bitsplit::bitsplit_tekken(t, g, s, f, _l, o) + bitcanon::bitcanon_tekken(t, g, s, f, _l, o) } fn bs_kimi( t: &[u8], @@ -100,7 +100,7 @@ fn bs_kimi( _l: &mut [u64], o: &mut [Span], ) -> usize { - bitsplit::bitsplit_kimi(t, g, s, f, _l, o) + bitcanon::bitcanon_kimi(t, g, s, f, _l, o) } fn spans(f: Split, s: &str) -> Vec { diff --git a/tokenizers/bitmap_gen/Cargo.toml b/tokenizers/bitmap_gen/Cargo.toml index 800c26cc17..8a9dbb5cfc 100644 --- a/tokenizers/bitmap_gen/Cargo.toml +++ b/tokenizers/bitmap_gen/Cargo.toml @@ -2,7 +2,7 @@ name = "bitmap_gen" version = "0.1.0" edition = "2024" -description = "Dev tool: regenerates bitsplit's committed classify tables (uses unicode-properties). Not for publication." +description = "Dev tool: regenerates bitcanon's committed classify tables (uses unicode-properties). Not for publication." license = "Apache-2.0" repository = "https://github.com/huggingface/tokenizers" authors = ["Arthur Zucker ", "Luc Georges "] diff --git a/tokenizers/bitmap_gen/src/lib.rs b/tokenizers/bitmap_gen/src/lib.rs index 91e8ed011e..e7618d5258 100644 --- a/tokenizers/bitmap_gen/src/lib.rs +++ b/tokenizers/bitmap_gen/src/lib.rs @@ -1,11 +1,11 @@ //! Dev-time only (depends on `unicode-properties`); nothing here is linked into the runtime crate. //! `cargo run -p bitmap_gen` calls [`generate_atom_tables`] and writes the committed -//! `bitsplit/src/classify/atom_tables.rs`. It bakes the dense `Tables` layout (ascii / 2-byte group / 3-byte +//! `bitcanon/src/classify/atom_tables.rs`. It bakes the dense `Tables` layout (ascii / 2-byte group / 3-byte //! fast3 / bmp_rle / astral) use std::fmt::Write as _; use unicode_properties::{GeneralCategory, UnicodeGeneralCategory}; -/// Script=Han, sorted and disjoint — the same table `bitsplit::han` carried, now baked into the +/// Script=Han, sorted and disjoint — the same table `bitcanon::han` carried, now baked into the /// atom scheme instead of range-tested per byte at runtime. Extensions past Ext F are included /// because onig's `\p{Han}` has them; the parity gate pins this to the oracle's Unicode version. const HAN: &[(u32, u32)] = &[ diff --git a/tokenizers/bitmap_gen/src/main.rs b/tokenizers/bitmap_gen/src/main.rs index 28dcf919a0..12b606d01c 100644 --- a/tokenizers/bitmap_gen/src/main.rs +++ b/tokenizers/bitmap_gen/src/main.rs @@ -1,11 +1,11 @@ -//! Regenerate bitsplit's committed classify tables: +//! Regenerate bitcanon's committed classify tables: //! cargo run -p bitmap_gen [-- ] -//! Default `out_path` = ../bitsplit/src/classify/atom_tables.rs. `generate_atom_tables` self-validates every +//! Default `out_path` = ../bitcanon/src/classify/atom_tables.rs. `generate_atom_tables` self-validates every //! codepoint against the reference `atom()`, so an inconsistent scheme change fails HERE, not at ship. fn main() { let default = concat!( env!("CARGO_MANIFEST_DIR"), - "/../bitsplit/src/classify/atom_tables.rs" + "/../bitcanon/src/classify/atom_tables.rs" ); let out = std::env::args() .nth(1) diff --git a/tokenizers/src/lib.rs b/tokenizers/src/lib.rs index 2b9591bc9d..09ca56d378 100644 --- a/tokenizers/src/lib.rs +++ b/tokenizers/src/lib.rs @@ -5,7 +5,7 @@ //! The 🤗 Tokenizers library. //! //! The implementation is split across crates (each built on internal engines — `tk_encode` on the -//! `bitsplit` SIMD pre-tokenizer, and the shared `bitmap_gen` tables): +//! `bitcanon` SIMD pre-tokenizer, and the shared `bitmap_gen` tables): //! //! - [`tk_encode`] — inference: the model engines and the full pipeline components //! ([`Normalizer`], [`PreTokenizer`], [`Model`], [`PostProcessor`], [`Decoder`]). diff --git a/tokenizers/tk-convert/Cargo.toml b/tokenizers/tk-convert/Cargo.toml index a945cda0d8..ed6b842697 100644 --- a/tokenizers/tk-convert/Cargo.toml +++ b/tokenizers/tk-convert/Cargo.toml @@ -28,10 +28,10 @@ path = "src/lib.rs" [dependencies] serde_json = "1.0" thiserror = "2" -# The GPT-2 pattern a `ByteLevel` lowers to. `bitsplit::regexes` is the single source of truth for +# The GPT-2 pattern a `ByteLevel` lowers to. `bitcanon::regexes` is the single source of truth for # it -- tk-encode's runtime recognizer reads the same const -- so a copy here could drift and cost # the FSM fast path without changing any test. -bitsplit = { path = "../bitsplit", version = "0.1.0-dev.0" } +bitcanon = { path = "../bitcanon", version = "0.1.0-dev.0" } # Latest released tokenizers, the independent id/decode oracle `tests/oracle.rs` compares against. tokenizers-release = { package = "tokenizers", version = "=0.23.2", optional = true } diff --git a/tokenizers/tk-convert/src/convert.rs b/tokenizers/tk-convert/src/convert.rs index 2cf7367d93..7d92c0b802 100644 --- a/tokenizers/tk-convert/src/convert.rs +++ b/tokenizers/tk-convert/src/convert.rs @@ -456,7 +456,7 @@ fn lower_byte_level_pre_tokenizer(root: &mut Map) -> Result<(), C "Split", &[( "pattern", - serde_json::json!({ "Regex": bitsplit::regexes::GPT2 }), + serde_json::json!({ "Regex": bitcanon::regexes::GPT2 }), )], )), false => None, diff --git a/tokenizers/tk-convert/tests/convert.rs b/tokenizers/tk-convert/tests/convert.rs index 49cecbc8e5..c4f8b7cd16 100644 --- a/tokenizers/tk-convert/tests/convert.rs +++ b/tokenizers/tk-convert/tests/convert.rs @@ -290,7 +290,7 @@ fn a_byte_level_pre_tokenizer_becomes_a_model_flag_and_a_split() { let v = done(BPE, r#", "pre_tokenizer": {"type": "ByteLevel"}"#); let pretok = &v["pre_tokenizer"]; assert_eq!(v["model"]["byte_level"], true); - assert_eq!(pretok["pattern"]["Regex"], bitsplit::regexes::GPT2); + assert_eq!(pretok["pattern"]["Regex"], bitcanon::regexes::GPT2); // `use_regex: false` asked only for the byte map, so the member simply goes. let v = done( diff --git a/tokenizers/tk-encode/Cargo.toml b/tokenizers/tk-encode/Cargo.toml index c3ca37a965..03be44abe5 100644 --- a/tokenizers/tk-encode/Cargo.toml +++ b/tokenizers/tk-encode/Cargo.toml @@ -29,7 +29,7 @@ path = "src/lib.rs" [dependencies] ############ Required dependencies ############################### -bitsplit = { path = "../bitsplit", version = "0.1.0-dev.0" } +bitcanon = { path = "../bitcanon", version = "0.1.0-dev.0" } rand = "0.10" ptr_hash = { version = "2.0.2", default-features = false } unicode_categories = "0.1" diff --git a/tokenizers/tk-encode/README.md b/tokenizers/tk-encode/README.md index 89922594d3..b9dafa2a3f 100644 --- a/tokenizers/tk-encode/README.md +++ b/tokenizers/tk-encode/README.md @@ -81,7 +81,7 @@ own, but it is `exclude`d from the workspace and nothing in `tokenizers` depends - **parallelism**: rayon-backed batch encoding. - **fancy-regex**: the optional system-regex backend, needed *only* for a genuine regex pattern - in a `Split` pre-tokenizer or a `Replace` normalizer. The bitsplit-native pre-tokenizers + in a `Split` pre-tokenizer or a `Replace` normalizer. The bitcanon-native pre-tokenizers (GPT-2, cl100k, o200k, tekken, deepseek, the class family, char-delimiter) need no backend, and a literal pattern is searched for directly. diff --git a/tokenizers/tk-encode/src/lib.rs b/tokenizers/tk-encode/src/lib.rs index 254b707573..e4c1ca05f9 100644 --- a/tokenizers/tk-encode/src/lib.rs +++ b/tokenizers/tk-encode/src/lib.rs @@ -85,7 +85,7 @@ //! - **parallelism**: rayon-backed batch encoding. //! //! - **fancy-regex**: the optional system-regex backend, needed *only* for a genuine regex pattern -//! in a `Split` pre-tokenizer or a `Replace` normalizer. The bitsplit-native pre-tokenizers +//! in a `Split` pre-tokenizer or a `Replace` normalizer. The bitcanon-native pre-tokenizers //! (GPT-2, cl100k, o200k, tekken, deepseek, the class family, char-delimiter) need no backend, //! and a literal pattern is searched for directly. //! diff --git a/tokenizers/tk-encode/src/pre_tokenizers/bert.rs b/tokenizers/tk-encode/src/pre_tokenizers/bert.rs index 0d3d78a8d5..be6f937d1a 100644 --- a/tokenizers/tk-encode/src/pre_tokenizers/bert.rs +++ b/tokenizers/tk-encode/src/pre_tokenizers/bert.rs @@ -1,14 +1,14 @@ use crate::pipeline::{self, PreTokenizerScratch}; use crate::tokenizer::Result; -use bitsplit::classes::class_runs_into; -use bitsplit::classify::mask; +use bitcanon::classes::class_runs_into; +use bitcanon::classify::mask; #[derive(Copy, Clone, Debug, PartialEq, Eq)] pub struct BertPreTokenizer; -// SAFETY: the spans come from an `bitsplit` fsm, which splits only at character boundaries of `text`. -// See `bitsplit` docs. +// SAFETY: the spans come from an `bitcanon` fsm, which splits only at character boundaries of `text`. +// See `bitcanon` docs. unsafe impl pipeline::PreTokenizer for BertPreTokenizer { #[inline(never)] fn pre_tokenize( @@ -18,7 +18,7 @@ unsafe impl pipeline::PreTokenizer for BertPreTokenizer { out: &mut Vec, ) -> Result<()> { // Bert pre-tokenization = drop whitespace runs, isolate each punctuation char, keep every other - // run. One `bitsplit` SIMD classify (bytes → atom tags) + the class-runs FSM, byte-exact with + // run. One `bitcanon` SIMD classify (bytes → atom tags) + the class-runs FSM, byte-exact with // the legacy `char::is_whitespace` / `is_punc` split above (see the tests). scratch.split_on_bits( text.as_bytes(), diff --git a/tokenizers/tk-encode/src/pre_tokenizers/delimiter.rs b/tokenizers/tk-encode/src/pre_tokenizers/delimiter.rs index 7fc7787620..a277271f43 100644 --- a/tokenizers/tk-encode/src/pre_tokenizers/delimiter.rs +++ b/tokenizers/tk-encode/src/pre_tokenizers/delimiter.rs @@ -16,7 +16,7 @@ impl CharDelimiterSplit { } } -// SAFETY: the spans come from `bitsplit::classes::CharDelimiterSplit`, which splits only at character +// SAFETY: the spans come from `bitcanon::classes::CharDelimiterSplit`, which splits only at character // boundaries of `text`. It scans for the delimiter's own UTF-8 bytes and confirms the whole encoding // before cutting. unsafe impl pipeline::PreTokenizer for CharDelimiterSplit { @@ -26,13 +26,13 @@ unsafe impl pipeline::PreTokenizer for CharDelimiterSplit { scratch: &mut PreTokenizerScratch, out: &mut Vec, ) -> Result<()> { - // native bitsplit FSM (memchr-backed single-byte scan); `Removed` — drops the delimiter, + // native bitcanon FSM (memchr-backed single-byte scan); `Removed` — drops the delimiter, // keeps the runs between, no empty spans. Byte-exact with the char-predicate split. // It keys on the delimiter's own bytes rather than on an atom class, so it needs no tags. scratch.split_on_bytes( text.as_bytes(), |bytes, spans| { - bitsplit::classes::CharDelimiterSplit(self.delimiter).pre_tokenize( + bitcanon::classes::CharDelimiterSplit(self.delimiter).pre_tokenize( bytes, &mut [], spans, diff --git a/tokenizers/tk-encode/src/pre_tokenizers/digits.rs b/tokenizers/tk-encode/src/pre_tokenizers/digits.rs index 894dd25a5a..84854d91f1 100644 --- a/tokenizers/tk-encode/src/pre_tokenizers/digits.rs +++ b/tokenizers/tk-encode/src/pre_tokenizers/digits.rs @@ -1,7 +1,7 @@ use crate::pipeline::{self, PreTokenizerScratch}; use crate::tokenizer::Result; -use bitsplit::classes::class_runs_into; -use bitsplit::classify::mask; +use bitcanon::classes::class_runs_into; +use bitcanon::classify::mask; #[derive(Clone, Debug, PartialEq, Eq)] /// Pre tokenizes the numbers into single tokens. If individual_digits is set @@ -23,8 +23,8 @@ impl Default for Digits { } } -// SAFETY: the spans come from an `bitsplit` fsm, which splits only at character boundaries of `text`. -// See `bitsplit` docs. +// SAFETY: the spans come from an `bitcanon` fsm, which splits only at character boundaries of `text`. +// See `bitcanon` docs. unsafe impl pipeline::PreTokenizer for Digits { fn pre_tokenize( &self, @@ -32,7 +32,7 @@ unsafe impl pipeline::PreTokenizer for Digits { scratch: &mut PreTokenizerScratch, out: &mut Vec, ) -> Result<()> { - // isolate each numeric char (`individual_digits`) or keep numeric runs — bitsplit classify + + // isolate each numeric char (`individual_digits`) or keep numeric runs — bitcanon classify + // class-runs FSM. atom `NUMERIC` == `char::is_numeric`, so byte-exact with the scalar path. let individual = self.individual_digits; scratch.split_on_bits( diff --git a/tokenizers/tk-encode/src/pre_tokenizers/punctuation.rs b/tokenizers/tk-encode/src/pre_tokenizers/punctuation.rs index eed8ddc70c..692cb57b53 100644 --- a/tokenizers/tk-encode/src/pre_tokenizers/punctuation.rs +++ b/tokenizers/tk-encode/src/pre_tokenizers/punctuation.rs @@ -1,8 +1,8 @@ use crate::pipeline::{self, PreTokenizerScratch}; use crate::tokenizer::{Result, SplitDelimiterBehavior}; use SplitDelimiterBehavior::{Isolated, Removed}; -use bitsplit::classes::class_runs_into; -use bitsplit::classify::mask; +use bitcanon::classes::class_runs_into; +use bitcanon::classify::mask; use unicode_categories::UnicodeCategories; pub(crate) fn is_punc(x: char) -> bool { @@ -28,7 +28,7 @@ impl Default for Punctuation { } // SAFETY: both routes cut only at character boundaries of `text`. -// The class-runs route is an `bitsplit` fsm. +// The class-runs route is an `bitcanon` fsm. // the merge behaviors go through `pipeline::split_delimiter`, which takes its offsets from `str::char_indices`. unsafe impl pipeline::PreTokenizer for Punctuation { fn pre_tokenize( diff --git a/tokenizers/tk-encode/src/pre_tokenizers/sequence.rs b/tokenizers/tk-encode/src/pre_tokenizers/sequence.rs index 30a53404cb..23a11054ba 100644 --- a/tokenizers/tk-encode/src/pre_tokenizers/sequence.rs +++ b/tokenizers/tk-encode/src/pre_tokenizers/sequence.rs @@ -51,8 +51,8 @@ impl PipelineSequence { // - all of its children are safe // - offsets added by the sequence are correct and land on character boundaries // -// The deepseek fast path has no children to run: it calls an `bitsplit` fsm, which splits only at -// character boundaries of `text`. See the `bitsplit` docs. +// The deepseek fast path has no children to run: it calls an `bitcanon` fsm, which splits only at +// character boundaries of `text`. See the `bitcanon` docs. unsafe impl pipeline::PreTokenizer for PipelineSequence { /// Runs each child in turn, where every child subdivides the spans produced /// so far. A child sees only the text of a span (`&text[span]`) and returns @@ -74,7 +74,7 @@ unsafe impl pipeline::PreTokenizer for PipelineSequence { scratch.split_on_bits( text.as_bytes(), |t, tags, starts, _flags, _later, out| { - bitsplit::bitsplit_deepseek(t, tags, starts, out) + bitcanon::bitcanon_deepseek(t, tags, starts, out) }, out, ); diff --git a/tokenizers/tk-encode/src/pre_tokenizers/split.rs b/tokenizers/tk-encode/src/pre_tokenizers/split.rs index 5166582b16..67413d9565 100644 --- a/tokenizers/tk-encode/src/pre_tokenizers/split.rs +++ b/tokenizers/tk-encode/src/pre_tokenizers/split.rs @@ -1,6 +1,6 @@ use crate::pipeline; use crate::utils::{Grammar, SysRegex, recognize}; -use bitsplit::literal::Literal; +use bitcanon::literal::Literal; use crate::tokenizer::{ Result, SplitDelimiterBehavior, @@ -53,7 +53,7 @@ pub struct Split { pub search: Search, pub behavior: SplitDelimiterBehavior, pub invert: bool, - /// Native `bitsplit` FSM for a recognized GPT regex (gpt2 / cl100k-Llama-3 / o200k), used on the + /// Native `bitcanon` FSM for a recognized GPT regex (gpt2 / cl100k-Llama-3 / o200k), used on the /// pipeline path when `behavior == Isolated && !invert` (how these regexes always ship). Byte-exact /// with `regex`; `None` falls back to `regex`. fsm: Option, @@ -157,7 +157,7 @@ impl Split { } } -// SAFETY: both routes cut only at character boundaries of `text`. The native route is a `bitsplit` +// SAFETY: both routes cut only at character boundaries of `text`. The native route is a `bitcanon` // grammar, see "What the spans guarantee" in its docs. The search route forwards the offsets of // `Pattern::find_matches` through `pipeline::split_matches`, and every `Pattern` here reports // boundaries: a regex matches on a `&str`, and `Literal` holds the bytes of a `&str` pattern, which @@ -170,7 +170,7 @@ unsafe impl pipeline::PreTokenizer for Split { out: &mut Vec, ) -> Result<()> { // A recognized GPT regex in its only real usage -- `Isolated`, not inverted -- routes - // straight to the native bitsplit grammar. These regexes cover the whole input, so + // straight to the native bitcanon grammar. These regexes cover the whole input, so // `Isolated` == the match list, and the grammar is byte-exact with `regex` (see the tests). if let Some(grammar) = self .fsm @@ -244,7 +244,7 @@ mod tests { // A recognised pattern still reports its family, so a caller can route it natively. let gpt2 = Split::native( - SplitPattern::Regex(bitsplit::regexes::GPT2.to_string()), + SplitPattern::Regex(bitcanon::regexes::GPT2.to_string()), SplitDelimiterBehavior::Isolated, false, ) diff --git a/tokenizers/tk-encode/src/pre_tokenizers/whitespace.rs b/tokenizers/tk-encode/src/pre_tokenizers/whitespace.rs index 801d6ba93e..4930dfbe52 100644 --- a/tokenizers/tk-encode/src/pre_tokenizers/whitespace.rs +++ b/tokenizers/tk-encode/src/pre_tokenizers/whitespace.rs @@ -4,7 +4,7 @@ use crate::tokenizer::Result; #[derive(Clone, Debug, PartialEq, Eq)] pub struct Whitespace; -use bitsplit::classify::mask; +use bitcanon::classify::mask; impl Default for Whitespace { fn default() -> Self { @@ -15,8 +15,8 @@ impl Default for Whitespace { #[derive(Copy, Clone, Debug, PartialEq, Eq)] pub struct WhitespaceSplit; -// SAFETY: the spans come from an `bitsplit` fsm, which cuts only at character boundaries of `text`. -// See "What the spans guarantee" in the `bitsplit` docs. +// SAFETY: the spans come from an `bitcanon` fsm, which cuts only at character boundaries of `text`. +// See "What the spans guarantee" in the `bitcanon` docs. unsafe impl pipeline::PreTokenizer for WhitespaceSplit { fn pre_tokenize( &self, @@ -24,12 +24,12 @@ unsafe impl pipeline::PreTokenizer for WhitespaceSplit { scratch: &mut PreTokenizerScratch, out: &mut Vec, ) -> Result<()> { - // drop whitespace runs, keep everything else as runs — bitsplit SIMD classify + class-runs FSM. + // drop whitespace runs, keep everything else as runs — bitcanon SIMD classify + class-runs FSM. // atom `WS` == `char::is_whitespace`, so byte-exact with the scalar path. scratch.split_on_bits( text.as_bytes(), |b, t, st, fk, _, o| { - bitsplit::classes::class_runs_into::<{ mask::WS }, 0, 0>(b, t, st, fk, o) + bitcanon::classes::class_runs_into::<{ mask::WS }, 0, 0>(b, t, st, fk, o) }, out, ); @@ -51,8 +51,8 @@ pub fn is_word_char(ch: char) -> bool { || ch == '\u{200d}' // Zero-Width Joiner } -// SAFETY: the spans come from an `bitsplit` fsm, which splits only at character boundaries of `text`. -// See `bitsplit` docs. +// SAFETY: the spans come from an `bitcanon` fsm, which splits only at character boundaries of `text`. +// See `bitcanon` docs. unsafe impl pipeline::PreTokenizer for Whitespace { #[inline(never)] fn pre_tokenize( @@ -62,11 +62,11 @@ unsafe impl pipeline::PreTokenizer for Whitespace { out: &mut Vec, ) -> Result<()> { // `\w+|[^\w\s]+`: drop whitespace, cut at the word↔symbol boundary, each run one token — - // bitsplit classify + class-runs FSM (`WORD` = `\w`; keep-A = word, keep-B = symbol). + // bitcanon classify + class-runs FSM (`WORD` = `\w`; keep-A = word, keep-B = symbol). scratch.split_on_bits( text.as_bytes(), |b, t, st, fk, _, o| { - bitsplit::classes::class_runs_into::<{ mask::WS }, 0, { mask::WORD }>( + bitcanon::classes::class_runs_into::<{ mask::WS }, 0, { mask::WORD }>( b, t, st, fk, o, ) }, diff --git a/tokenizers/tk-encode/src/tokenizer/pattern.rs b/tokenizers/tk-encode/src/tokenizer/pattern.rs index 1bc27420d9..539ff1ca7c 100644 --- a/tokenizers/tk-encode/src/tokenizer/pattern.rs +++ b/tokenizers/tk-encode/src/tokenizer/pattern.rs @@ -1,6 +1,6 @@ use crate::utils::SysRegex; use crate::{Offsets, Result}; -use bitsplit::literal::Literal; +use bitcanon::literal::Literal; #[cfg(test)] use regex::Regex; @@ -22,7 +22,7 @@ impl Pattern for char { } /// Splitting on a [`regex::Regex`] has no caller left: every runtime `Pattern` site goes through -/// `SysRegex` (the `fancy-regex` backend, or its stub) or straight to `bitsplit`. The impl is kept +/// `SysRegex` (the `fancy-regex` backend, or its stub) or straight to `bitcanon`. The impl is kept /// only because the unit tests below build `regex::Regex` values directly, and they get the crate /// from `[dev-dependencies]` in every feature rung. That is why `regex` is no longer a real /// dependency of `tk-encode` at all. diff --git a/tokenizers/tk-encode/src/tokenizer/pipeline/mod.rs b/tokenizers/tk-encode/src/tokenizer/pipeline/mod.rs index 9af40c07a2..c9f332024a 100644 --- a/tokenizers/tk-encode/src/tokenizer/pipeline/mod.rs +++ b/tokenizers/tk-encode/src/tokenizer/pipeline/mod.rs @@ -34,7 +34,7 @@ mod scratch_pool; pub use scratch_pool::ModelScratch; -pub use bitsplit::Span; +pub use bitcanon::Span; mod normalizer; mod post_processor; diff --git a/tokenizers/tk-encode/src/tokenizer/pipeline/pre_tokenizer.rs b/tokenizers/tk-encode/src/tokenizer/pipeline/pre_tokenizer.rs index 4d47f583b3..7e9ed9a857 100644 --- a/tokenizers/tk-encode/src/tokenizer/pipeline/pre_tokenizer.rs +++ b/tokenizers/tk-encode/src/tokenizer/pipeline/pre_tokenizer.rs @@ -13,8 +13,8 @@ use crate::pre_tokenizers::{ whitespace::{Whitespace, WhitespaceSplit}, }; use crate::tokenizer::{Result, SplitDelimiterBehavior}; -use bitsplit::Span; -use bitsplit::classify::classify; +use bitcanon::Span; +use bitcanon::classify::classify; /// Range-based pre-tokenization: yields spans into the input rather than owned /// substrings, so the pipeline can pre-tokenize without allocating. @@ -43,7 +43,7 @@ pub unsafe trait PreTokenizer { /// The working buffers a [`PreTokenizer`] needs to split a text into pre-tokens. #[derive(Default)] pub struct PreTokenizerScratch { - /// One [`bitsplit`] atom tag per input byte, what [`bitsplit::classify::classify`] writes + /// One [`bitcanon`] atom tag per input byte, what [`bitcanon::classify::classify`] writes /// and the grammars read. tags: Vec, /// An intermediate buffer in which the FSM writes the Span before they get appended to the output @@ -61,7 +61,7 @@ pub struct PreTokenizerScratch { } impl PreTokenizerScratch { - /// Tag every byte of `bytes` with its [`bitsplit`] atom class, run `fsm` over the tags, and + /// Tag every byte of `bytes` with its [`bitcanon`] atom class, run `fsm` over the tags, and /// append the spans to `out`. pub fn split_on_tags( &mut self, diff --git a/tokenizers/tk-encode/src/tokenizer/pipeline/scratch_pool.rs b/tokenizers/tk-encode/src/tokenizer/pipeline/scratch_pool.rs index 928e8d8d49..74b3472c2a 100644 --- a/tokenizers/tk-encode/src/tokenizer/pipeline/scratch_pool.rs +++ b/tokenizers/tk-encode/src/tokenizer/pipeline/scratch_pool.rs @@ -6,7 +6,7 @@ use std::{ }, }; -use bitsplit::Span; +use bitcanon::Span; use crate::pipeline::{Model, PipelineModel, PipelineModelScratch, PreTokenizerScratch}; diff --git a/tokenizers/tk-encode/src/utils/byte_level.rs b/tokenizers/tk-encode/src/utils/byte_level.rs index c4df634add..f35a00c6a9 100644 --- a/tokenizers/tk-encode/src/utils/byte_level.rs +++ b/tokenizers/tk-encode/src/utils/byte_level.rs @@ -1,11 +1,11 @@ use crate::vocab::bucket_vocab_store::BucketVocabStore; use std::sync::LazyLock; -// The GPT-2 pre-tokenize regex is the canonical spec in bitsplit (single source of truth); re-export +// The GPT-2 pre-tokenize regex is the canonical spec in bitcanon (single source of truth); re-export // under the historical name so call sites are unchanged. `pub` because the `ByteLevel` pre-tokenizer // lowering -- which rewrites a `use_regex` ByteLevel into a `Split` on exactly this pattern -- lives // in `tk-convert`. -pub use bitsplit::regexes::GPT2 as GPT2_REGEX_STR; +pub use bitcanon::regexes::GPT2 as GPT2_REGEX_STR; /// Maps each byte to its GPT-2 byte-level unicode character, indexed by the byte value. /// diff --git a/tokenizers/tk-encode/src/utils/mod.rs b/tokenizers/tk-encode/src/utils/mod.rs index a76dd6ff9b..e369c6859b 100644 --- a/tokenizers/tk-encode/src/utils/mod.rs +++ b/tokenizers/tk-encode/src/utils/mod.rs @@ -4,10 +4,10 @@ pub(crate) mod cache; pub mod from_pretrained; pub(crate) mod word_cache; -// Optional system-regex backend, needed only for a *regex* pattern that bitsplit does not cover. +// Optional system-regex backend, needed only for a *regex* pattern that bitcanon does not cover. // With `fancy-regex` off a stub compiles and those patterns error at load. Everything else works -// regardless: the bitsplit-native pre-tokenizers, and any `Split` or `Replace` whose pattern is a -// plain string (searched for directly, see `bitsplit::literal`). +// regardless: the bitcanon-native pre-tokenizers, and any `Split` or `Replace` whose pattern is a +// plain string (searched for directly, see `bitcanon::literal`). #[cfg(feature = "fancy-regex")] mod fancy; #[cfg(feature = "fancy-regex")] @@ -17,7 +17,7 @@ mod no_regex; #[cfg(not(feature = "fancy-regex"))] pub use no_regex::SysRegex; -// Recognize known GPT pre-tokenization regexes and route them to bitsplit's native (unrolled) FSM. +// Recognize known GPT pre-tokenization regexes and route them to bitcanon's native (unrolled) FSM. mod unrolled_regex; pub use unrolled_regex::{DEEPSEEK_PATTERNS, Grammar, GrammarPattern, is_deepseek, recognize}; diff --git a/tokenizers/tk-encode/src/utils/no_regex.rs b/tokenizers/tk-encode/src/utils/no_regex.rs index 81d53478ee..3c1d5d6ff6 100644 --- a/tokenizers/tk-encode/src/utils/no_regex.rs +++ b/tokenizers/tk-encode/src/utils/no_regex.rs @@ -1,9 +1,9 @@ //! Stub `SysRegex` for builds with **no** system-regex backend (`fancy-regex` off — the default). //! //! The type stays present so `Split` / `Replace` still compile, but construction always fails. Only a -//! *regex* pattern ever asks for it: the bitsplit-native pre-tokenizers (GPT-2, cl100k, deepseek, the +//! *regex* pattern ever asks for it: the bitcanon-native pre-tokenizers (GPT-2, cl100k, deepseek, the //! class family, char-delimiter) need no backend, and a plain string pattern is searched for directly -//! (`bitsplit::literal`). A regex bitsplit does not cover errors at load time with a clear message. +//! (`bitcanon::literal`). A regex bitcanon does not cover errors at load time with a clear message. //! Enable `fancy-regex` to get a real backend. use std::error::Error; diff --git a/tokenizers/tk-encode/src/utils/search.rs b/tokenizers/tk-encode/src/utils/search.rs index 5c720179ab..af41b43153 100644 --- a/tokenizers/tk-encode/src/utils/search.rs +++ b/tokenizers/tk-encode/src/utils/search.rs @@ -5,7 +5,7 @@ //! matcher here is what lets those two be genuinely separate types rather than one type wearing //! two hats. -use bitsplit::literal::Literal; +use bitcanon::literal::Literal; use crate::tokenizer::Result; use crate::tokenizer::pattern::Pattern; diff --git a/tokenizers/tk-encode/src/utils/unrolled_regex.rs b/tokenizers/tk-encode/src/utils/unrolled_regex.rs index 89a69dd6aa..a3bc174ef2 100644 --- a/tokenizers/tk-encode/src/utils/unrolled_regex.rs +++ b/tokenizers/tk-encode/src/utils/unrolled_regex.rs @@ -1,15 +1,15 @@ -//! Recognize a known GPT pre-tokenization regex and route it to the byte-exact native `bitsplit` +//! Recognize a known GPT pre-tokenization regex and route it to the byte-exact native `bitcanon` //! grammar, so those pre-tokenizers need no system-regex backend. An unrecognized pattern returns //! `None` and falls back to `SysRegex` (the optional fancy-regex backend). -// The canonical regexes are the recognition keys; the single source of truth is `bitsplit::regexes`. -use bitsplit::Span; -use bitsplit::regexes::{GPT2, KIMI_K2, O200K, TEKKEN}; +// The canonical regexes are the recognition keys; the single source of truth is `bitcanon::regexes`. +use bitcanon::Span; +use bitcanon::regexes::{GPT2, KIMI_K2, O200K, TEKKEN}; // cl100k is recognized structurally (see `cl100k_digit_cap`), so the exact pattern is only a test key. #[cfg(test)] -use bitsplit::regexes::CL100K; +use bitcanon::regexes::CL100K; -/// A recognized GPT pre-tokenization regex and the `bitsplit` grammar that reproduces its +/// A recognized GPT pre-tokenization regex and the `bitcanon` grammar that reproduces its /// `Isolated` split byte-for-byte. One variant per distinct regex; models sharing a regex share a /// variant (o200k covers Llama-4, gpt-oss and MiniMax-M2). #[derive(Clone, Copy, Debug, PartialEq, Eq)] @@ -41,14 +41,14 @@ impl Grammar { out: &mut [Span], ) -> usize { match self { - Grammar::Gpt2 => bitsplit::bitsplit_byte_level(text, tags, starts, flag, out), + Grammar::Gpt2 => bitcanon::bitcanon_byte_level(text, tags, starts, flag, out), Grammar::Cl100k { digit_cap: 1 } => { - bitsplit::bitsplit_qwen(text, tags, starts, flag, out) + bitcanon::bitcanon_qwen(text, tags, starts, flag, out) } - Grammar::Cl100k { .. } => bitsplit::bitsplit_cl100k(text, tags, starts, flag, out), - Grammar::O200k => bitsplit::bitsplit_o200k(text, tags, starts, flag, later, out), - Grammar::Tekken => bitsplit::bitsplit_tekken(text, tags, starts, flag, later, out), - Grammar::Kimi => bitsplit::bitsplit_kimi(text, tags, starts, flag, later, out), + Grammar::Cl100k { .. } => bitcanon::bitcanon_cl100k(text, tags, starts, flag, out), + Grammar::O200k => bitcanon::bitcanon_o200k(text, tags, starts, flag, later, out), + Grammar::Tekken => bitcanon::bitcanon_tekken(text, tags, starts, flag, later, out), + Grammar::Kimi => bitcanon::bitcanon_kimi(text, tags, starts, flag, later, out), } } } @@ -96,7 +96,7 @@ impl crate::tokenizer::pattern::Pattern for GrammarPattern { let bytes = inside.as_bytes(); let n = bytes.len(); let mut tags = vec![0u8; n]; - bitsplit::classify::classify(bytes, &mut tags); + bitcanon::classify::classify(bytes, &mut tags); let words = n.div_ceil(64) + 1; let (mut starts, mut flag) = (vec![0u64; words], vec![0u64; words]); let mut later = vec![0u64; 2 * words]; @@ -112,7 +112,7 @@ impl crate::tokenizer::pattern::Pattern for GrammarPattern { } // deepseek-v3/v4's pre-tokenizer is a `Sequence` of these three Isolated `Split`s (+ a byte-map -// `ByteLevel`), which `bitsplit::bitsplit_deepseek` collapses into one pass. Byte-exact with the +// `ByteLevel`), which `bitcanon::bitcanon_deepseek` collapses into one pass. Byte-exact with the // shipped tokenizer.json — the big pattern carries LITERAL CR/LF, spliced in via `concat!`. const DS_NUM: &str = r"\p{N}{1,3}"; const DS_CJK: &str = "[\u{4E00}-\u{9FA5}\u{3040}-\u{309F}\u{30A0}-\u{30FF}]+"; @@ -127,13 +127,13 @@ const DS_BIG: &str = concat!( ); /// deepseek's three patterns, in order, exactly as the shipped config spells them. The copies in -/// `bitsplit::regexes` escape CR/LF instead of embedding it, so the two are **not** +/// `bitcanon::regexes` escape CR/LF instead of embedding it, so the two are **not** /// interchangeable and [`is_deepseek`] rejects the escaped form -- anything rebuilding these /// patterns must use this constant or it silently falls off the native FSM onto the regex fallback. pub const DEEPSEEK_PATTERNS: [&str; 3] = [DS_NUM, DS_CJK, DS_BIG]; /// True iff three `Split` patterns are exactly deepseek's `[\p{N}{1,3}, CJK-range, big-regex]` prefix → -/// `bitsplit::bitsplit_deepseek` reproduces the whole composed Isolated split in one pass. +/// `bitcanon::bitcanon_deepseek` reproduces the whole composed Isolated split in one pass. pub fn is_deepseek(p0: &str, p1: &str, p2: &str) -> bool { p0 == DS_NUM && p1 == DS_CJK && p2 == DS_BIG } @@ -142,7 +142,7 @@ pub fn is_deepseek(p0: &str, p1: &str, p2: &str) -> bool { mod tests { use super::*; - /// The trap [`DEEPSEEK_PATTERNS`] exists to close: `bitsplit::regexes::DEEPSEEK_BIG` spells + /// The trap [`DEEPSEEK_PATTERNS`] exists to close: `bitcanon::regexes::DEEPSEEK_BIG` spells /// CR/LF as the two-character escape `\r\n` inside a raw string, while the shipped configs (and /// so `DS_BIG`) embed real control characters. They are not interchangeable, and rebuilding from /// the wrong one silently drops off the native FSM onto the regex fallback. @@ -151,10 +151,10 @@ mod tests { let [num, cjk, big] = DEEPSEEK_PATTERNS; assert!(is_deepseek(num, cjk, big)); - let escaped = bitsplit::regexes::DEEPSEEK; + let escaped = bitcanon::regexes::DEEPSEEK; assert!( !is_deepseek(escaped[0], escaped[1], escaped[2]), - "bitsplit's escaped copies must NOT be mistaken for the shipped spelling" + "bitcanon's escaped copies must NOT be mistaken for the shipped spelling" ); // Only the big one differs; the first two are identical in both. assert_eq!(num, escaped[0]); diff --git a/tokenizers/tk-serialize/Cargo.toml b/tokenizers/tk-serialize/Cargo.toml index 2ec454e8f1..61db4fd94c 100644 --- a/tokenizers/tk-serialize/Cargo.toml +++ b/tokenizers/tk-serialize/Cargo.toml @@ -39,9 +39,9 @@ base64 = "0.22" # The runtime this crate builds. No `config`: the whole point is that reading a canonical # `tokenizer.json` needs no serde layer at all. tk-encode = { path = "../tk-encode", version = "0.1.0-dev.0", default-features = false } -# The GPT-2 pre-tokenizer pattern is a constant in `bitsplit`, and a `ByteLevel` pre-tokenizer is +# The GPT-2 pre-tokenizer pattern is a constant in `bitcanon`, and a `ByteLevel` pre-tokenizer is # spelled as a `Split` on it. Nothing else here reaches the FSM crate directly. -bitsplit = { path = "../bitsplit", version = "0.1.0-dev.0" } +bitcanon = { path = "../bitcanon", version = "0.1.0-dev.0" } [features] # Reading a canonical `tokenizer.json` into a `PipelineTokenizer`. On by default -- it is what the From 1446a4172a5788c22028b922330bebc8b992048f Mon Sep 17 00:00:00 2001 From: SBrandeis <33657802+SBrandeis@users.noreply.github.com> Date: Fri, 18 Sep 2026 18:02:57 +0200 Subject: [PATCH 4/4] rust-release manual dispatch --- .github/workflows/rust-release.yml | 17 ++++++++++++----- 1 file changed, 12 insertions(+), 5 deletions(-) diff --git a/.github/workflows/rust-release.yml b/.github/workflows/rust-release.yml index 75c61c4920..f97294f9f3 100644 --- a/.github/workflows/rust-release.yml +++ b/.github/workflows/rust-release.yml @@ -1,12 +1,12 @@ name: Rust Release -env: - CRATES_TOKEN: ${{ secrets.CRATES_TOKEN }} - on: push: tags: - v* + # Manual runs do everything but upload: the drift check, then `cargo publish --dry-run`, which + # packages every crate and builds each one from its own tarball. See the publish step. + workflow_dispatch: permissions: {} @@ -47,12 +47,19 @@ jobs: # No rc gate: a caret requirement never resolves a prerelease unless asked for by name. - name: Publish workspace crates working-directory: ./tokenizers - run: cargo publish --workspace --token ${CRATES_TOKEN} + env: + CARGO_REGISTRY_TOKEN: ${{ secrets.CRATES_TOKEN }} + run: cargo publish --workspace ${{ !startsWith(github.ref, 'refs/tags/v') && '--dry-run' || '' }} # tk-train is `exclude`d from the workspace, so `--workspace` never sees it, and it cannot # even be packaged until tk-encode and tk-convert are live: from outside the workspace its # path deps resolve their version pins against the registry. It goes last, on its own. + # Tags only: with nothing on the registry yet it cannot even be packaged, so a manual run + # has no dry-run for it. - name: Publish tk-train + if: startsWith(github.ref, 'refs/tags/v') working-directory: ./tokenizers/tk-train - run: cargo publish --token ${CRATES_TOKEN} + env: + CARGO_REGISTRY_TOKEN: ${{ secrets.CRATES_TOKEN }} + run: cargo publish