serve.py 5.1 KB
Newer Older
M
MRXLT 已提交
1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20
# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
#     http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
"""
Usage:
    Host a trained paddle model with one line command
    Example:
        python -m paddle_serving_server.serve --model ./serving_server_model --port 9292
"""
import argparse
D
Dong Daxiang 已提交
21
import os
G
guru4elephant 已提交
22
from multiprocessing import Pool, Process
23
from paddle_serving_server_gpu import serve_args
M
MRXLT 已提交
24
from flask import Flask, request
M
MRXLT 已提交
25 26


27
def start_gpu_card_model(index, gpuid, args):  # pylint: disable=doc-string-missing
G
guru4elephant 已提交
28
    gpuid = int(gpuid)
G
guru4elephant 已提交
29
    device = "gpu"
30
    port = args.port
G
guru4elephant 已提交
31 32
    if gpuid == -1:
        device = "cpu"
G
guru4elephant 已提交
33
    elif gpuid >= 0:
34
        port = args.port + index
M
MRXLT 已提交
35 36
    thread_num = args.thread
    model = args.model
M
MRXLT 已提交
37
    mem_optim = args.mem_optim_off is False
M
MRXLT 已提交
38
    ir_optim = args.ir_optim
M
MRXLT 已提交
39
    max_body_size = args.max_body_size
B
barrierye 已提交
40
    use_multilang = args.use_multilang
Z
zhangjun 已提交
41 42 43
    workdir = args.workdir
    if gpuid >= 0:
        workdir = "{}_{}".format(args.workdir, gpuid)
M
MRXLT 已提交
44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59

    if model == "":
        print("You must specify your serving model")
        exit(-1)

    import paddle_serving_server_gpu as serving
    op_maker = serving.OpMaker()
    read_op = op_maker.create('general_reader')
    general_infer_op = op_maker.create('general_infer')
    general_response_op = op_maker.create('general_response')

    op_seq_maker = serving.OpSeqMaker()
    op_seq_maker.add_op(read_op)
    op_seq_maker.add_op(general_infer_op)
    op_seq_maker.add_op(general_response_op)

B
barrierye 已提交
60 61 62 63
    if use_multilang:
        server = serving.MultiLangServer()
    else:
        server = serving.Server()
M
MRXLT 已提交
64 65
    server.set_op_sequence(op_seq_maker.get_op_sequence())
    server.set_num_threads(thread_num)
M
MRXLT 已提交
66
    server.set_memory_optimize(mem_optim)
M
MRXLT 已提交
67
    server.set_ir_optimize(ir_optim)
M
MRXLT 已提交
68
    server.set_max_body_size(max_body_size)
M
add trt  
MRXLT 已提交
69
    if args.use_trt:
M
bug fix  
MRXLT 已提交
70
        server.set_trt()
M
MRXLT 已提交
71

Z
zhangjun 已提交
72 73 74 75 76 77 78
    if args.use_lite:
        server.set_lite()
        device = "arm"

    if args.use_xpu:
        server.set_xpu()

79 80 81 82 83
    if args.product_name != None:
        server.set_product_name(args.product_name)
    if args.container_id != None:
        server.set_container_id(args.container_id)

M
MRXLT 已提交
84
    server.load_model_config(model)
85
    server.prepare_server(workdir=workdir, port=port, device=device)
G
guru4elephant 已提交
86 87
    if gpuid >= 0:
        server.set_gpuid(gpuid)
M
MRXLT 已提交
88 89 90
    server.run_server()


91
def start_multi_card(args):  # pylint: disable=doc-string-missing
92 93
    gpus = ""
    if args.gpu_ids == "":
M
MRXLT 已提交
94
        gpus = []
95 96
    else:
        gpus = args.gpu_ids.split(",")
M
MRXLT 已提交
97 98
        if "CUDA_VISIBLE_DEVICES" in os.environ:
            env_gpus = os.environ["CUDA_VISIBLE_DEVICES"].split(",")
M
MRXLT 已提交
99 100 101
            for ids in gpus:
                if int(ids) >= len(env_gpus):
                    print(
T
TeslaZhao 已提交
102 103
                        " Max index of gpu_ids out of range, the number of CUDA_VISIBLE_DEVICES is {}."
                        .format(len(env_gpus)))
M
MRXLT 已提交
104
                    exit(-1)
M
MRXLT 已提交
105 106
        else:
            env_gpus = []
Z
zhangjun 已提交
107 108 109 110
    if args.use_lite:
        print("run arm server.")
        start_gpu_card_model(-1, -1, args)
    elif len(gpus) <= 0:
M
MRXLT 已提交
111
        print("gpu_ids not set, going to run cpu service.")
112
        start_gpu_card_model(-1, -1, args)
G
guru4elephant 已提交
113 114
    else:
        gpu_processes = []
G
guru4elephant 已提交
115
        for i, gpu_id in enumerate(gpus):
B
barrierye 已提交
116
            p = Process(
117
                target=start_gpu_card_model, args=(
B
barrierye 已提交
118
                    i,
M
MRXLT 已提交
119
                    gpu_id,
B
barrierye 已提交
120
                    args, ))
G
guru4elephant 已提交
121 122 123 124 125
            gpu_processes.append(p)
        for p in gpu_processes:
            p.start()
        for p in gpu_processes:
            p.join()
B
barrierye 已提交
126 127


M
MRXLT 已提交
128
if __name__ == "__main__":
129
    args = serve_args()
130
    if args.name == "None":
131
        start_multi_card(args)
132
    else:
Y
Your Name 已提交
133
        from .web_service import WebService
134 135
        web_service = WebService(name=args.name)
        web_service.load_model_config(args.model)
Y
Your Name 已提交
136 137
        gpu_ids = args.gpu_ids
        if gpu_ids == "":
138 139 140
            if "CUDA_VISIBLE_DEVICES" in os.environ:
                gpu_ids = os.environ["CUDA_VISIBLE_DEVICES"]
        if len(gpu_ids) > 0:
Y
Your Name 已提交
141
            web_service.set_gpus(gpu_ids)
142 143
        web_service.prepare_server(
            workdir=args.workdir, port=args.port, device=args.device)
M
MRXLT 已提交
144
        web_service.run_rpc_service()
M
MRXLT 已提交
145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161

        app_instance = Flask(__name__)

        @app_instance.before_first_request
        def init():
            web_service._launch_web_service()

        service_name = "/" + web_service.name + "/prediction"

        @app_instance.route(service_name, methods=["POST"])
        def run():
            return web_service.get_prediction(request)

        app_instance.run(host="0.0.0.0",
                         port=web_service.port,
                         threaded=False,
                         processes=4)