This message was deleted.
# ask-for-help
s
This message was deleted.
b
Hello. Can you share your service code?
p
Hy @Bo, Hi, Bo. Here is my service code.
Copy code
@svc.api(input=NumpyNdarray(), output=NumpyNdarray(), route='/model_binary/predict')
async def model_binary_predict(input) :
    response = await binary_runner.predict.async_run([input])
    return response[0]
And here is my runnable code
Copy code
class BinaryModelRunnable(bentoml.Runnable):
    """
    Runnable class that inherits from bentoml.Runnable. Enables the creation of a runner with custom methods for Agent sentences
    """
    SUPPORTED_RESOURCES = ("<http://nvidia.com/gpu|nvidia.com/gpu>", "cpu")
    SUPPORTS_CPU_MULTI_THREADING = True

    def __init__(self):
        """
        Starts the class by loading the corresponding models
        """
        self.binary_model = bentoml.pytorch.load_model(binary_model)
        <http://self.binary_model.to|self.binary_model.to>(device)    
    

    @bentoml.Runnable.method(batchable=True, batch_dim=0 )
    def predict(self, input):
        
        input = input[0]
    
        all_ids = torch.tensor(
            [np.array(input[0])]
        )
        attention_mask = torch.tensor(
            [np.array(input[1])]
        )
        all_ids = <http://all_ids.to|all_ids.to>(torch.device('cuda'), dtype = torch.long)
        attention_mask = <http://attention_mask.to|attention_mask.to>(torch.device('cuda') , dtype = torch.long)

        prediction = []

        with torch.no_grad():
            output = self.binary_model(all_ids, attention_mask)
            prediction.extend(torch.sigmoid(output).cpu().detach().numpy().tolist())
        
        return np.array(prediction) 
        # return np.array([])

binary_runner = bentoml.Runner(BinaryModelRunnable, name= "binary_runnable", models=[binary_model], max_batch_size=100, max_latency_ms=1000)
@Bo? Are you there?