From caf8fa0847b718bfdf2f82aa869e1359cf541864 Mon Sep 17 00:00:00 2001 From: Nicolas Patry Date: Tue, 4 Jun 2024 23:08:34 +0000 Subject: [PATCH] Integration tests for intel ? --- .github/workflows/build.yaml | 395 +++++++++++++---------------------- 1 file changed, 145 insertions(+), 250 deletions(-) diff --git a/.github/workflows/build.yaml b/.github/workflows/build.yaml index ec73311c..e825ad60 100644 --- a/.github/workflows/build.yaml +++ b/.github/workflows/build.yaml @@ -57,218 +57,130 @@ jobs: {"Key": "GitHubRepository", "Value": "${{ github.repository }}"} ] - build-and-push-image: - concurrency: - group: ${{ github.workflow }}-build-and-push-image-${{ github.head_ref || github.run_id }} - cancel-in-progress: true - needs: start-runner # required to start the main job when the runner is ready - runs-on: ${{ needs.start-runner.outputs.label }} # run the job on the newly created runner - permissions: - contents: write - packages: write - # This is used to complete the identity challenge - # with sigstore/fulcio when running outside of PRs. - id-token: write - security-events: write - steps: - - name: Checkout repository - uses: actions/checkout@v3 - - name: Initialize Docker Buildx - uses: docker/setup-buildx-action@v2.0.0 - with: - install: true - - name: Inject slug/short variables - uses: rlespinasse/github-slug-action@v4.4.1 - - name: Tailscale - uses: tailscale/github-action@7bd8039bf25c23c4ab1b8d6e2cc2da2280601966 - with: - authkey: ${{ secrets.TAILSCALE_AUTHKEY }} - - name: Login to GitHub Container Registry - if: github.event_name != 'pull_request' - uses: docker/login-action@v2 - with: - registry: ghcr.io - username: ${{ github.actor }} - password: ${{ secrets.GITHUB_TOKEN }} - - name: Login to internal Container Registry - uses: docker/login-action@v2.1.0 - with: - username: ${{ secrets.TAILSCALE_DOCKER_USERNAME }} - password: ${{ secrets.TAILSCALE_DOCKER_PASSWORD }} - registry: registry.internal.huggingface.tech - - name: Login to Azure Container Registry - if: github.event_name != 'pull_request' - uses: docker/login-action@v2.1.0 - with: - username: ${{ secrets.AZURE_DOCKER_USERNAME }} - password: ${{ secrets.AZURE_DOCKER_PASSWORD }} - registry: db4c2190dd824d1f950f5d1555fbadf0.azurecr.io - # If pull request - - name: Extract metadata (tags, labels) for Docker - if: ${{ github.event_name == 'pull_request' }} - id: meta-pr - uses: docker/metadata-action@v4.3.0 - with: - images: | - registry.internal.huggingface.tech/api-inference/community/text-generation-inference - tags: | - type=raw,value=sha-${{ env.GITHUB_SHA_SHORT }} - # If main, release or tag - - name: Extract metadata (tags, labels) for Docker - if: ${{ github.event_name != 'pull_request' }} - id: meta - uses: docker/metadata-action@v4.3.0 - with: - flavor: | - latest=auto - images: | - registry.internal.huggingface.tech/api-inference/community/text-generation-inference - ghcr.io/huggingface/text-generation-inference - db4c2190dd824d1f950f5d1555fbadf0.azurecr.io/text-generation-inference - tags: | - type=semver,pattern={{version}} - type=semver,pattern={{major}}.{{minor}} - type=raw,value=latest,enable=${{ github.ref == format('refs/heads/{0}', github.event.repository.default_branch) }} - type=raw,value=sha-${{ env.GITHUB_SHA_SHORT }} - - name: Build and push Docker image - id: build-and-push - uses: docker/build-push-action@v4 - with: - context: . - file: Dockerfile - push: true - platforms: 'linux/amd64' - build-args: | - GIT_SHA=${{ env.GITHUB_SHA }} - DOCKER_LABEL=sha-${{ env.GITHUB_SHA_SHORT }} - tags: ${{ steps.meta.outputs.tags || steps.meta-pr.outputs.tags }} - labels: ${{ steps.meta.outputs.labels || steps.meta-pr.outputs.labels }} - cache-from: type=registry,ref=registry.internal.huggingface.tech/api-inference/community/text-generation-inference:cache,mode=min - cache-to: type=registry,ref=registry.internal.huggingface.tech/api-inference/community/text-generation-inference:cache,mode=min + # build-and-push-image: + # concurrency: + # group: ${{ github.workflow }}-build-and-push-image-${{ github.head_ref || github.run_id }} + # cancel-in-progress: true + # needs: start-runner # required to start the main job when the runner is ready + # runs-on: ${{ needs.start-runner.outputs.label }} # run the job on the newly created runner + # permissions: + # contents: write + # packages: write + # # This is used to complete the identity challenge + # # with sigstore/fulcio when running outside of PRs. + # id-token: write + # security-events: write + # steps: + # - name: Checkout repository + # uses: actions/checkout@v3 + # - name: Initialize Docker Buildx + # uses: docker/setup-buildx-action@v2.0.0 + # with: + # install: true + # - name: Inject slug/short variables + # uses: rlespinasse/github-slug-action@v4.4.1 + # - name: Tailscale + # uses: tailscale/github-action@7bd8039bf25c23c4ab1b8d6e2cc2da2280601966 + # with: + # authkey: ${{ secrets.TAILSCALE_AUTHKEY }} + # - name: Login to GitHub Container Registry + # if: github.event_name != 'pull_request' + # uses: docker/login-action@v2 + # with: + # registry: ghcr.io + # username: ${{ github.actor }} + # password: ${{ secrets.GITHUB_TOKEN }} + # - name: Login to internal Container Registry + # uses: docker/login-action@v2.1.0 + # with: + # username: ${{ secrets.TAILSCALE_DOCKER_USERNAME }} + # password: ${{ secrets.TAILSCALE_DOCKER_PASSWORD }} + # registry: registry.internal.huggingface.tech + # - name: Login to Azure Container Registry + # if: github.event_name != 'pull_request' + # uses: docker/login-action@v2.1.0 + # with: + # username: ${{ secrets.AZURE_DOCKER_USERNAME }} + # password: ${{ secrets.AZURE_DOCKER_PASSWORD }} + # registry: db4c2190dd824d1f950f5d1555fbadf0.azurecr.io + # # If pull request + # - name: Extract metadata (tags, labels) for Docker + # if: ${{ github.event_name == 'pull_request' }} + # id: meta-pr + # uses: docker/metadata-action@v4.3.0 + # with: + # images: | + # registry.internal.huggingface.tech/api-inference/community/text-generation-inference + # tags: | + # type=raw,value=sha-${{ env.GITHUB_SHA_SHORT }} + # # If main, release or tag + # - name: Extract metadata (tags, labels) for Docker + # if: ${{ github.event_name != 'pull_request' }} + # id: meta + # uses: docker/metadata-action@v4.3.0 + # with: + # flavor: | + # latest=auto + # images: | + # registry.internal.huggingface.tech/api-inference/community/text-generation-inference + # ghcr.io/huggingface/text-generation-inference + # db4c2190dd824d1f950f5d1555fbadf0.azurecr.io/text-generation-inference + # tags: | + # type=semver,pattern={{version}} + # type=semver,pattern={{major}}.{{minor}} + # type=raw,value=latest,enable=${{ github.ref == format('refs/heads/{0}', github.event.repository.default_branch) }} + # type=raw,value=sha-${{ env.GITHUB_SHA_SHORT }} + # - name: Build and push Docker image + # id: build-and-push + # uses: docker/build-push-action@v4 + # with: + # context: . + # file: Dockerfile + # push: true + # platforms: 'linux/amd64' + # build-args: | + # GIT_SHA=${{ env.GITHUB_SHA }} + # DOCKER_LABEL=sha-${{ env.GITHUB_SHA_SHORT }} + # tags: ${{ steps.meta.outputs.tags || steps.meta-pr.outputs.tags }} + # labels: ${{ steps.meta.outputs.labels || steps.meta-pr.outputs.labels }} + # cache-from: type=registry,ref=registry.internal.huggingface.tech/api-inference/community/text-generation-inference:cache,mode=min + # cache-to: type=registry,ref=registry.internal.huggingface.tech/api-inference/community/text-generation-inference:cache,mode=min - integration-tests: - concurrency: - group: ${{ github.workflow }}-${{ github.job }}-${{ github.head_ref || github.run_id }} - cancel-in-progress: true - needs: - - start-runner - - build-and-push-image # Wait for the docker image to be built - runs-on: ${{ needs.start-runner.outputs.label }} # run the job on the newly created runner - env: - DOCKER_VOLUME: /cache - steps: - - uses: actions/checkout@v2 - - name: Inject slug/short variables - uses: rlespinasse/github-slug-action@v4.4.1 - - name: Set up Python - uses: actions/setup-python@v4 - with: - python-version: 3.9 - - name: Tailscale - uses: tailscale/github-action@7bd8039bf25c23c4ab1b8d6e2cc2da2280601966 - with: - authkey: ${{ secrets.TAILSCALE_AUTHKEY }} - - name: Prepare disks - run: | - sudo mkfs -t ext4 /dev/nvme1n1 - sudo mkdir ${{ env.DOCKER_VOLUME }} - sudo mount /dev/nvme1n1 ${{ env.DOCKER_VOLUME }} - - name: Install - run: | - make install-integration-tests - - name: Run tests - run: | - export DOCKER_IMAGE=registry.internal.huggingface.tech/api-inference/community/text-generation-inference:sha-${{ env.GITHUB_SHA_SHORT }} - export HUGGING_FACE_HUB_TOKEN=${{ secrets.HUGGING_FACE_HUB_TOKEN }} - pytest -s -vv integration-tests - - build-and-push-image-rocm: - concurrency: - group: ${{ github.workflow }}-build-and-push-image-rocm-${{ github.head_ref || github.run_id }} - cancel-in-progress: true - runs-on: intel-pvc-tgi - permissions: - contents: write - packages: write - # This is used to complete the identity challenge - # with sigstore/fulcio when running outside of PRs. - id-token: write - security-events: write - steps: - - name: Checkout repository - uses: actions/checkout@v3 - - name: Initialize Docker Buildx - uses: docker/setup-buildx-action@v2.0.0 - with: - install: true - - name: Inject slug/short variables - uses: rlespinasse/github-slug-action@v4.4.1 - - name: Tailscale - uses: tailscale/github-action@7bd8039bf25c23c4ab1b8d6e2cc2da2280601966 - with: - authkey: ${{ secrets.TAILSCALE_AUTHKEY }} - - name: Login to GitHub Container Registry - if: github.event_name != 'pull_request' - uses: docker/login-action@v2 - with: - registry: ghcr.io - username: ${{ github.actor }} - password: ${{ secrets.GITHUB_TOKEN }} - - name: Login to internal Container Registry - uses: docker/login-action@v2.1.0 - with: - username: ${{ secrets.TAILSCALE_DOCKER_USERNAME }} - password: ${{ secrets.TAILSCALE_DOCKER_PASSWORD }} - registry: registry.internal.huggingface.tech - - name: Login to Azure Container Registry - if: github.event_name != 'pull_request' - uses: docker/login-action@v2.1.0 - with: - username: ${{ secrets.AZURE_DOCKER_USERNAME }} - password: ${{ secrets.AZURE_DOCKER_PASSWORD }} - registry: db4c2190dd824d1f950f5d1555fbadf0.azurecr.io - # If pull request - - name: Extract metadata (tags, labels) for Docker - if: ${{ github.event_name == 'pull_request' }} - id: meta-pr - uses: docker/metadata-action@v4.3.0 - with: - images: | - registry.internal.huggingface.tech/api-inference/community/text-generation-inference - tags: | - type=raw,value=sha-${{ env.GITHUB_SHA_SHORT }}-rocm - # If main, release or tag - - name: Extract metadata (tags, labels) for Docker - if: ${{ github.event_name != 'pull_request' }} - id: meta - uses: docker/metadata-action@v4.3.0 - with: - flavor: | - latest=false - images: | - registry.internal.huggingface.tech/api-inference/community/text-generation-inference - ghcr.io/huggingface/text-generation-inference - db4c2190dd824d1f950f5d1555fbadf0.azurecr.io/text-generation-inference - tags: | - type=semver,pattern={{version}}-rocm - type=semver,pattern={{major}}.{{minor}}-rocm - type=raw,value=latest-rocm,enable=${{ github.ref == format('refs/heads/{0}', github.event.repository.default_branch) }} - type=raw,value=sha-${{ env.GITHUB_SHA_SHORT }}-rocm - - name: Build and push Docker image - id: build-and-push - uses: docker/build-push-action@v4 - with: - context: . - file: Dockerfile_amd - push: true - platforms: 'linux/amd64' - build-args: | - GIT_SHA=${{ env.GITHUB_SHA }} - DOCKER_LABEL=sha-${{ env.GITHUB_SHA_SHORT }}-rocm - tags: ${{ steps.meta.outputs.tags || steps.meta-pr.outputs.tags }} - labels: ${{ steps.meta.outputs.labels || steps.meta-pr.outputs.labels }} - cache-from: type=registry,ref=registry.internal.huggingface.tech/api-inference/community/text-generation-inference:cache-rocm,mode=min - cache-to: type=registry,ref=registry.internal.huggingface.tech/api-inference/community/text-generation-inference:cache-rocm,mode=min + # integration-tests: + # concurrency: + # group: ${{ github.workflow }}-${{ github.job }}-${{ github.head_ref || github.run_id }} + # cancel-in-progress: true + # needs: + # - start-runner + # - build-and-push-image # Wait for the docker image to be built + # runs-on: ${{ needs.start-runner.outputs.label }} # run the job on the newly created runner + # env: + # DOCKER_VOLUME: /cache + # steps: + # - uses: actions/checkout@v2 + # - name: Inject slug/short variables + # uses: rlespinasse/github-slug-action@v4.4.1 + # - name: Set up Python + # uses: actions/setup-python@v4 + # with: + # python-version: 3.9 + # - name: Tailscale + # uses: tailscale/github-action@7bd8039bf25c23c4ab1b8d6e2cc2da2280601966 + # with: + # authkey: ${{ secrets.TAILSCALE_AUTHKEY }} + # - name: Prepare disks + # run: | + # sudo mkfs -t ext4 /dev/nvme1n1 + # sudo mkdir ${{ env.DOCKER_VOLUME }} + # sudo mount /dev/nvme1n1 ${{ env.DOCKER_VOLUME }} + # - name: Install + # run: | + # make install-integration-tests + # - name: Run tests + # run: | + # export DOCKER_IMAGE=registry.internal.huggingface.tech/api-inference/community/text-generation-inference:sha-${{ env.GITHUB_SHA_SHORT }} + # export HUGGING_FACE_HUB_TOKEN=${{ secrets.HUGGING_FACE_HUB_TOKEN }} + # pytest -s -vv integration-tests build-and-push-image-intel: concurrency: @@ -276,8 +188,6 @@ jobs: cancel-in-progress: true needs: - start-runner - - build-and-push-image # Wait for the main docker image to be built - - integration-tests # Wait for the main integration-tests runs-on: ${{ needs.start-runner.outputs.label }} # run the job on the newly created runner permissions: contents: write @@ -364,15 +274,34 @@ jobs: labels: ${{ steps.meta.outputs.labels || steps.meta-pr.outputs.labels }} cache-from: type=registry,ref=registry.internal.huggingface.tech/api-inference/community/text-generation-inference:cache-intel,mode=min cache-to: type=registry,ref=registry.internal.huggingface.tech/api-inference/community/text-generation-inference:cache-intel,mode=min + integration-tests-intel: + concurrency: + group: ${{ github.workflow }}-${{ github.job }}-${{ github.head_ref || github.run_id }} + cancel-in-progress: true + needs: + - start-runner + - build-and-push-image-intel + runs-on: intel-pvc-tgi + container: + image: registry.internal.huggingface.tech/api-inference/community/text-generation-inference:sha-${{ needs.build-and-push-image-intel.outputs.short_sha }}-intel + options: --device /dev/kfd --device /dev/dri --env ROCR_VISIBLE_DEVICES --shm-size "16gb" --ipc host -v /mnt/cache/.cache/huggingface:/cache + env: + DOCKER_VOLUME: /cache + steps: + - name: Install + run: | + make install-integration-tests + - name: Run tests + run: | + export DOCKER_IMAGE=registry.internal.huggingface.tech/api-inference/community/text-generation-inference:sha-${{ env.GITHUB_SHA_SHORT }}-intel + export HUGGING_FACE_HUB_TOKEN=${{ secrets.HUGGING_FACE_HUB_TOKEN }} + pytest -s -vv integration-tests stop-runner: name: Stop self-hosted EC2 runner needs: - start-runner - - build-and-push-image - - build-and-push-image-rocm - build-and-push-image-intel - - integration-tests runs-on: ubuntu-latest env: AWS_REGION: us-east-1 @@ -393,37 +322,3 @@ jobs: ec2-instance-id: ${{ needs.start-runner.outputs.ec2-instance-id }} # TODO: Move this to `build_amd.yml` (and `build_nvidia.yml`) - - # integration-tests-rocm: - # concurrency: - # group: ${{ github.workflow }}-${{ github.job }}-${{ github.head_ref || github.run_id }} - # cancel-in-progress: true - # needs: - # - start-runner - # - build-and-push-image - # - integration-tests - # - build-and-push-image-rocm - # - stop-runner - # runs-on: [self-hosted, amd-gpu, multi-gpu, mi300] - # container: - # image: registry.internal.huggingface.tech/api-inference/community/text-generation-inference:sha-${{ needs.build-and-push-image-rocm.outputs.short_sha }}-rocm - # options: --device /dev/kfd --device /dev/dri --env ROCR_VISIBLE_DEVICES --shm-size "16gb" --ipc host -v /mnt/cache/.cache/huggingface:/cache - # env: - # DOCKER_VOLUME: /cache - # steps: - # - name: ROCM-SMI - # run: | - # rocm-smi - # - name: ROCM-INFO - # run: | - # rocminfo | grep "Agent" -A 14 - # - name: Show ROCR environment - # run: | - # echo "ROCR: $ROCR_VISIBLE_DEVICES" - # - name: Install - # run: | - # make install-integration-tests - # - name: Run tests - # run: | - # export HUGGING_FACE_HUB_TOKEN=${{ secrets.HUGGING_FACE_HUB_TOKEN }} - # pytest -s -vv integration-tests