Slackbot
11/03/2022, 2:50 PMBo
11/03/2022, 9:11 PMPaulo Moreira
11/04/2022, 11:10 AM@svc.api(input=NumpyNdarray(), output=NumpyNdarray(), route='/model_binary/predict')
async def model_binary_predict(input) :
response = await binary_runner.predict.async_run([input])
return response[0]
And here is my runnable code
class BinaryModelRunnable(bentoml.Runnable):
"""
Runnable class that inherits from bentoml.Runnable. Enables the creation of a runner with custom methods for Agent sentences
"""
SUPPORTED_RESOURCES = ("<http://nvidia.com/gpu|nvidia.com/gpu>", "cpu")
SUPPORTS_CPU_MULTI_THREADING = True
def __init__(self):
"""
Starts the class by loading the corresponding models
"""
self.binary_model = bentoml.pytorch.load_model(binary_model)
<http://self.binary_model.to|self.binary_model.to>(device)
@bentoml.Runnable.method(batchable=True, batch_dim=0 )
def predict(self, input):
input = input[0]
all_ids = torch.tensor(
[np.array(input[0])]
)
attention_mask = torch.tensor(
[np.array(input[1])]
)
all_ids = <http://all_ids.to|all_ids.to>(torch.device('cuda'), dtype = torch.long)
attention_mask = <http://attention_mask.to|attention_mask.to>(torch.device('cuda') , dtype = torch.long)
prediction = []
with torch.no_grad():
output = self.binary_model(all_ids, attention_mask)
prediction.extend(torch.sigmoid(output).cpu().detach().numpy().tolist())
return np.array(prediction)
# return np.array([])
binary_runner = bentoml.Runner(BinaryModelRunnable, name= "binary_runnable", models=[binary_model], max_batch_size=100, max_latency_ms=1000)Paulo Moreira
11/07/2022, 4:02 PM