On Modal.com these 34 lines of code is all you need to serverlessly run BERT text generation inference on an A10G (which has 24GB of GPU memory). No Dockerfile, no YAML, no Terraform or AWS Cloudformation. Just these 34 lines.
import modal
def download_model():
from transformers import pipeline
pipeline("fill-mask", model="bert-base-uncased")
CACHE_PATH = "/root/model_cache" # model location in image
ENV = modal.Secret({"TRANSFORMERS_CACHE": CACHE_PATH})
image = (
modal.Image.debian_slim()
.pip_install("torch", "transformers")
.run_function(download_model, secret=ENV)
)
stub = modal.Stub(name="hn-demo", image=image)
class Model:
def __enter__(self):
from transformers import pipeline
self.model = pipeline("fill-mask", model="bert-base-uncased", device=0)
@stub.function(
gpu="a10g",
secret=ENV,
)
def handler(self, prompt: str):
return self.model(prompt)
if __name__ == "__main__":
with stub.run():
prompt = "Hello World! I am a [MASK] machine learning model."
print(Model().handler.call(prompt)[0]["sequence"])
Running `python hn_demo.py` prints "Hello World! I am a simple machine learning model."You can check out available GPUs at https://modal.com/docs/reference/modal.gpu.
There's also a bunch of easy-to-run examples in our docs :) https://modal.com/docs/guide/ex/stable_diffusion_cli