Skip to content

Commit 40af529

Browse files
calderjoclaude
andcommitted
chore: drop P100 from the CI pipeline
The P100 Jenkins agent (label `ephemeral-linux-gpu`) is being deleted, so move everything that ran on it over to T4x2: - Remove the dedicated `Test on P100` stage. T4x2 remains the GPU test bed. - Repoint the `Build GPU Image` / `Diff GPU Image` stages at `ephemeral-linux-gpu-t4x2`. - Drop the now-dead `p100_exempt` decorator and its stale cuDNN/sm_60 comments. Those tests were only ever skipped on Pascal hardware, so they now run unconditionally on T4. Co-Authored-By: Claude <noreply@anthropic.com>
1 parent bd6f3ab commit 40af529

11 files changed

Lines changed: 6 additions & 65 deletions

‎Jenkinsfile‎

Lines changed: 2 additions & 19 deletions
Original file line numberDiff line numberDiff line change
@@ -51,8 +51,8 @@ pipeline {
5151
}
5252
}
5353
stage('GPU') {
54-
agent { label 'ephemeral-linux-gpu' }
55-
stages {
54+
agent { label 'ephemeral-linux-gpu-t4x2' }
55+
stages {
5656
stage('Build GPU Image') {
5757
options {
5858
timeout(time: 4324, unit: 'MINUTES')
@@ -135,23 +135,6 @@ pipeline {
135135
}
136136
}
137137
}
138-
stage('Test on P100') {
139-
agent { label 'ephemeral-linux-gpu' }
140-
options {
141-
timeout(time: 40, unit: 'MINUTES')
142-
}
143-
steps {
144-
retry(2) {
145-
sh '''#!/bin/bash
146-
set -exo pipefail
147-
148-
date
149-
docker pull gcr.io/kaggle-private-byod/python:${PRETEST_TAG}
150-
./test --gpu --image gcr.io/kaggle-private-byod/python:${PRETEST_TAG}
151-
'''
152-
}
153-
}
154-
}
155138
stage('Test on T4x2') {
156139
agent { label 'ephemeral-linux-gpu-t4x2' }
157140
options {

‎tests/common.py‎

Lines changed: 0 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -15,6 +15,4 @@ def isGPU():
1515
return os.path.isfile('/proc/driver/nvidia/version')
1616

1717
gpu_test = unittest.skipIf(not isGPU(), 'Not running GPU tests')
18-
# b/342143152 P100s are slowly being unsupported in new release of popular ml tools such as RAPIDS.
19-
p100_exempt = unittest.skipIf(getAcceleratorName() == "Tesla P100-PCIE-16GB", 'Not running p100 exempt tests')
2018
tpu_test = unittest.skipIf(len(os.environ.get('ISTPUVM', '')) == 0, 'Not running TPU tests')

‎tests/test_cudf.py‎

Lines changed: 1 addition & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -1,11 +1,10 @@
11
import unittest
22

3-
from common import gpu_test, p100_exempt
3+
from common import gpu_test
44

55

66
class TestCudf(unittest.TestCase):
77
@gpu_test
8-
@p100_exempt # b/342143152: cuDL(>=24.4v) is inompatible with p100 GPUs.
98
def test_cudf_dataframe_operations(self):
109
import cudf
1110

‎tests/test_cuml.py‎

Lines changed: 1 addition & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -1,11 +1,10 @@
11
import unittest
22

3-
from common import gpu_test, p100_exempt
3+
from common import gpu_test
44

55

66
class TestCuml(unittest.TestCase):
77
@gpu_test
8-
@p100_exempt # b/342143152: cuML(>=24.4v) is inompatible with p100 GPUs.
98
def test_pca_fit_transform(self):
109
import unittest
1110
import numpy as np

‎tests/test_fastai.py‎

Lines changed: 0 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -3,8 +3,6 @@
33
import fastai
44
from fastai.tabular.all import *
55

6-
from common import p100_exempt
7-
86

97
class TestFastAI(unittest.TestCase):
108
# Basic import
@@ -24,7 +22,6 @@ def test_torch_tensor(self):
2422

2523
self.assertTrue(torch.all(a == b))
2624

27-
@p100_exempt
2825
def test_tabular(self):
2926
dls = TabularDataLoaders.from_csv(
3027
"/input/tests/data/train.csv",

‎tests/test_flax.py‎

Lines changed: 0 additions & 8 deletions
Original file line numberDiff line numberDiff line change
@@ -8,8 +8,6 @@
88
from flax import linen as nn
99
from flax.training import train_state
1010

11-
from common import p100_exempt
12-
1311

1412
class TestFlax(unittest.TestCase):
1513

@@ -19,12 +17,6 @@ def test_pooling(self):
1917
y = nn.pooling.pool(x, 1., mul_reduce, (2, 2), (1, 1), 'VALID')
2018
np.testing.assert_allclose(y, np.full((1, 2, 2, 1), 2. ** 4))
2119

22-
# cuDNN 9.19 (pulled in by torch 2.11) dropped the Pascal (sm_60) kernels from
23-
# libcudnn_ops/cnn/adv and ships PTX for sm_121 only, so nothing can be JIT'd
24-
# down to sm_60 either. Every cuDNN convolution engine fails on P100 with
25-
# CUDNN_STATUS_EXECUTION_FAILED. Not fixable here: cuDNN 9.19 is a hard
26-
# requirement of torch 2.11, which comes from the Colab base image.
27-
@p100_exempt
2820
def test_cnn(self):
2921
class CNN(nn.Module):
3022
@nn.compact

‎tests/test_keras.py‎

Lines changed: 0 additions & 8 deletions
Original file line numberDiff line numberDiff line change
@@ -7,15 +7,7 @@
77

88
import keras
99

10-
from common import p100_exempt
11-
1210
class TestKeras(unittest.TestCase):
13-
# cuDNN 9.19 (pulled in by torch 2.11) dropped the Pascal (sm_60) kernels from
14-
# libcudnn_ops/cnn/adv and ships PTX for sm_121 only, so nothing can be JIT'd
15-
# down to sm_60 either. Every cuDNN convolution engine fails on P100 with
16-
# CUDNN_STATUS_EXECUTION_FAILED. Not fixable here: cuDNN 9.19 is a hard
17-
# requirement of torch 2.11, which comes from the Colab base image.
18-
@p100_exempt
1911
def test_train(self):
2012
path = '/input/tests/data/mnist.npz'
2113
with np.load(path) as f:

‎tests/test_keras_cv.py‎

Lines changed: 0 additions & 7 deletions
Original file line numberDiff line numberDiff line change
@@ -4,16 +4,9 @@
44
import keras
55
import numpy as np
66

7-
from common import p100_exempt
87
from utils.kagglehub import create_test_kagglehub_server
98

109
class TestKerasCV(unittest.TestCase):
11-
# cuDNN 9.19 (pulled in by torch 2.11) dropped the Pascal (sm_60) kernels from
12-
# libcudnn_ops/cnn/adv and ships PTX for sm_121 only, so nothing can be JIT'd
13-
# down to sm_60 either. Every cuDNN convolution engine fails on P100 with
14-
# CUDNN_STATUS_EXECUTION_FAILED. Not fixable here: cuDNN 9.19 is a hard
15-
# requirement of torch 2.11, which comes from the Colab base image.
16-
@p100_exempt
1710
def test_inference(self):
1811
with create_test_kagglehub_server():
1912
classifier = keras_cv.models.ImageClassifier.from_preset(

‎tests/test_pytorch.py‎

Lines changed: 1 addition & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -4,7 +4,7 @@
44
import torch.nn as tnn
55
import torch.autograd as autograd
66

7-
from common import gpu_test, p100_exempt
7+
from common import gpu_test
88

99

1010
class TestPyTorch(unittest.TestCase):
@@ -16,7 +16,6 @@ def test_nn(self):
1616
linear_torch(data_torch)
1717

1818
@gpu_test
19-
@p100_exempt
2019
def test_linalg(self):
2120
A = torch.randn(3, 3).t().to('cuda')
2221
B = torch.randn(3).t().to('cuda')
@@ -25,7 +24,6 @@ def test_linalg(self):
2524
self.assertEqual(3, result.shape[0])
2625

2726
@gpu_test
28-
@p100_exempt
2927
def test_gpu_computation(self):
3028
cuda = torch.device('cuda')
3129
a = torch.tensor([1., 2.], device=cuda)
@@ -35,7 +33,6 @@ def test_gpu_computation(self):
3533
self.assertEqual(torch.tensor([3.], device=cuda), result)
3634

3735
@gpu_test
38-
@p100_exempt
3936
def test_cuda_nn(self):
4037
# These throw if cuda is misconfigured
4138
tnn.GRUCell(10,10).cuda()

‎tests/test_pytorch_lightning.py‎

Lines changed: 0 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -5,8 +5,6 @@
55
import torch.nn.functional as F
66
from torch.utils.data import DataLoader, TensorDataset
77

8-
from common import p100_exempt
9-
108

119
class LitDataModule(pl.LightningDataModule):
1210

@@ -61,7 +59,6 @@ class TestPytorchLightning(unittest.TestCase):
6159
def test_version(self):
6260
self.assertIsNotNone(pl.__version__)
6361

64-
@p100_exempt
6562
def test_mnist(self):
6663
dm = LitDataModule()
6764
model = LitClassifier()

0 commit comments

Comments
 (0)