paddleocr.py 11.7 KB
Newer Older
W
WenmuZhou 已提交
1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21
# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
#     http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.

import os
import sys

__dir__ = os.path.dirname(__file__)
sys.path.append(os.path.join(__dir__, ''))

import cv2
W
WenmuZhou 已提交
22
import logging
W
WenmuZhou 已提交
23 24 25 26
import numpy as np
from pathlib import Path

from tools.infer import predict_system
W
WenmuZhou 已提交
27
from ppocr.utils.logging import get_logger
W
WenmuZhou 已提交
28

W
WenmuZhou 已提交
29
logger = get_logger()
30
from ppocr.utils.utility import check_and_read_gif, get_image_file_list
31
from ppocr.utils.network import maybe_download, download_with_progressbar, is_link, confirm_model_dir_url
W
WenmuZhou 已提交
32
from tools.infer.utility import draw_ocr, init_args, str2bool
W
WenmuZhou 已提交
33 34 35

__all__ = ['PaddleOCR']

W
WenmuZhou 已提交
36
model_urls = {
T
tink2123 已提交
37 38
    'det': {
        'ch':
W
WenmuZhou 已提交
39
            'https://paddleocr.bj.bcebos.com/dygraph_v2.0/ch/ch_ppocr_mobile_v2.0_det_infer.tar',
T
tink2123 已提交
40
        'en':
W
WenmuZhou 已提交
41
            'https://paddleocr.bj.bcebos.com/dygraph_v2.0/multilingual/en_ppocr_mobile_v2.0_det_infer.tar'
T
tink2123 已提交
42
    },
W
WenmuZhou 已提交
43 44 45
    'rec': {
        'ch': {
            'url':
W
WenmuZhou 已提交
46
                'https://paddleocr.bj.bcebos.com/dygraph_v2.0/ch/ch_ppocr_mobile_v2.0_rec_infer.tar',
W
WenmuZhou 已提交
47 48 49 50
            'dict_path': './ppocr/utils/ppocr_keys_v1.txt'
        },
        'en': {
            'url':
W
WenmuZhou 已提交
51
                'https://paddleocr.bj.bcebos.com/dygraph_v2.0/multilingual/en_number_mobile_v2.0_rec_infer.tar',
T
tink2123 已提交
52
            'dict_path': './ppocr/utils/en_dict.txt'
W
WenmuZhou 已提交
53 54 55
        },
        'french': {
            'url':
W
WenmuZhou 已提交
56
                'https://paddleocr.bj.bcebos.com/dygraph_v2.0/multilingual/french_mobile_v2.0_rec_infer.tar',
W
WenmuZhou 已提交
57 58 59 60
            'dict_path': './ppocr/utils/dict/french_dict.txt'
        },
        'german': {
            'url':
W
WenmuZhou 已提交
61
                'https://paddleocr.bj.bcebos.com/dygraph_v2.0/multilingual/german_mobile_v2.0_rec_infer.tar',
W
WenmuZhou 已提交
62 63 64 65
            'dict_path': './ppocr/utils/dict/german_dict.txt'
        },
        'korean': {
            'url':
W
WenmuZhou 已提交
66
                'https://paddleocr.bj.bcebos.com/dygraph_v2.0/multilingual/korean_mobile_v2.0_rec_infer.tar',
W
WenmuZhou 已提交
67 68 69 70
            'dict_path': './ppocr/utils/dict/korean_dict.txt'
        },
        'japan': {
            'url':
W
WenmuZhou 已提交
71
                'https://paddleocr.bj.bcebos.com/dygraph_v2.0/multilingual/japan_mobile_v2.0_rec_infer.tar',
W
WenmuZhou 已提交
72
            'dict_path': './ppocr/utils/dict/japan_dict.txt'
T
tink2123 已提交
73 74 75
        },
        'chinese_cht': {
            'url':
W
WenmuZhou 已提交
76
                'https://paddleocr.bj.bcebos.com/dygraph_v2.0/multilingual/chinese_cht_mobile_v2.0_rec_infer.tar',
T
tink2123 已提交
77 78 79 80
            'dict_path': './ppocr/utils/dict/chinese_cht_dict.txt'
        },
        'ta': {
            'url':
W
WenmuZhou 已提交
81
                'https://paddleocr.bj.bcebos.com/dygraph_v2.0/multilingual/ta_mobile_v2.0_rec_infer.tar',
T
tink2123 已提交
82 83 84 85
            'dict_path': './ppocr/utils/dict/ta_dict.txt'
        },
        'te': {
            'url':
W
WenmuZhou 已提交
86
                'https://paddleocr.bj.bcebos.com/dygraph_v2.0/multilingual/te_mobile_v2.0_rec_infer.tar',
T
tink2123 已提交
87 88 89 90
            'dict_path': './ppocr/utils/dict/te_dict.txt'
        },
        'ka': {
            'url':
W
WenmuZhou 已提交
91
                'https://paddleocr.bj.bcebos.com/dygraph_v2.0/multilingual/ka_mobile_v2.0_rec_infer.tar',
T
tink2123 已提交
92 93 94 95
            'dict_path': './ppocr/utils/dict/ka_dict.txt'
        },
        'latin': {
            'url':
W
WenmuZhou 已提交
96
                'https://paddleocr.bj.bcebos.com/dygraph_v2.0/multilingual/latin_ppocr_mobile_v2.0_rec_infer.tar',
T
tink2123 已提交
97 98 99 100
            'dict_path': './ppocr/utils/dict/latin_dict.txt'
        },
        'arabic': {
            'url':
W
WenmuZhou 已提交
101
                'https://paddleocr.bj.bcebos.com/dygraph_v2.0/multilingual/arabic_ppocr_mobile_v2.0_rec_infer.tar',
T
tink2123 已提交
102 103 104 105
            'dict_path': './ppocr/utils/dict/arabic_dict.txt'
        },
        'cyrillic': {
            'url':
W
WenmuZhou 已提交
106
                'https://paddleocr.bj.bcebos.com/dygraph_v2.0/multilingual/cyrillic_ppocr_mobile_v2.0_rec_infer.tar',
T
tink2123 已提交
107 108 109 110
            'dict_path': './ppocr/utils/dict/cyrillic_dict.txt'
        },
        'devanagari': {
            'url':
W
WenmuZhou 已提交
111
                'https://paddleocr.bj.bcebos.com/dygraph_v2.0/multilingual/devanagari_ppocr_mobile_v2.0_rec_infer.tar',
T
tink2123 已提交
112
            'dict_path': './ppocr/utils/dict/devanagari_dict.txt'
W
WenmuZhou 已提交
113 114 115
        }
    },
    'cls':
W
WenmuZhou 已提交
116
        'https://paddleocr.bj.bcebos.com/dygraph_v2.0/ch/ch_ppocr_mobile_v2.0_cls_infer.tar'
W
WenmuZhou 已提交
117 118 119
}

SUPPORT_DET_MODEL = ['DB']
T
tink2123 已提交
120
VERSION = '2.1'
121 122
SUPPORT_REC_MODEL = ['CRNN']
BASE_DIR = os.path.expanduser("~/.paddleocr/")
W
WenmuZhou 已提交
123 124


W
WenmuZhou 已提交
125
def parse_args(mMain=True):
W
WenmuZhou 已提交
126
    import argparse
W
WenmuZhou 已提交
127 128 129 130 131 132 133 134 135
    parser = init_args()
    parser.add_help = mMain
    parser.add_argument("--lang", type=str, default='ch')
    parser.add_argument("--det", type=str2bool, default=True)
    parser.add_argument("--rec", type=str2bool, default=True)

    for action in parser._actions:
        if action.dest == 'rec_char_dict_path':
            action.default = None
W
WenmuZhou 已提交
136
    if mMain:
W
WenmuZhou 已提交
137
        return parser.parse_args()
W
WenmuZhou 已提交
138
    else:
139
        inference_args_dict = {}
W
WenmuZhou 已提交
140 141
        for action in parser._actions:
            inference_args_dict[action.dest] = action.default
142
        return argparse.Namespace(**inference_args_dict)
W
WenmuZhou 已提交
143 144 145


class PaddleOCR(predict_system.TextSystem):
146
    def __init__(self, **kwargs):
W
WenmuZhou 已提交
147 148 149 150 151
        """
        paddleocr package
        args:
            **kwargs: other params show in paddleocr --help
        """
W
WenmuZhou 已提交
152 153
        params = parse_args(mMain=False)
        params.__dict__.update(**kwargs)
W
WenmuZhou 已提交
154 155
        if not params.show_log:
            logger.setLevel(logging.INFO)
W
WenmuZhou 已提交
156 157
        self.use_angle_cls = params.use_angle_cls
        lang = params.lang
T
tink2123 已提交
158
        latin_lang = [
T
tink2123 已提交
159 160 161 162
            'af', 'az', 'bs', 'cs', 'cy', 'da', 'de', 'es', 'et', 'fr', 'ga',
            'hr', 'hu', 'id', 'is', 'it', 'ku', 'la', 'lt', 'lv', 'mi', 'ms',
            'mt', 'nl', 'no', 'oc', 'pi', 'pl', 'pt', 'ro', 'rs_latin', 'sk',
            'sl', 'sq', 'sv', 'sw', 'tl', 'tr', 'uz', 'vi'
T
tink2123 已提交
163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180
        ]
        arabic_lang = ['ar', 'fa', 'ug', 'ur']
        cyrillic_lang = [
            'ru', 'rs_cyrillic', 'be', 'bg', 'uk', 'mn', 'abq', 'ady', 'kbd',
            'ava', 'dar', 'inh', 'che', 'lbe', 'lez', 'tab'
        ]
        devanagari_lang = [
            'hi', 'mr', 'ne', 'bh', 'mai', 'ang', 'bho', 'mah', 'sck', 'new',
            'gom', 'sa', 'bgc'
        ]
        if lang in latin_lang:
            lang = "latin"
        elif lang in arabic_lang:
            lang = "arabic"
        elif lang in cyrillic_lang:
            lang = "cyrillic"
        elif lang in devanagari_lang:
            lang = "devanagari"
W
WenmuZhou 已提交
181 182
        assert lang in model_urls[
            'rec'], 'param lang must in {}, but got {}'.format(
W
WenmuZhou 已提交
183
            model_urls['rec'].keys(), lang)
T
tink2123 已提交
184 185 186 187
        if lang == "ch":
            det_lang = "ch"
        else:
            det_lang = "en"
W
WenmuZhou 已提交
188
        use_inner_dict = False
W
WenmuZhou 已提交
189
        if params.rec_char_dict_path is None:
W
WenmuZhou 已提交
190
            use_inner_dict = True
W
WenmuZhou 已提交
191
            params.rec_char_dict_path = model_urls['rec'][lang][
W
WenmuZhou 已提交
192
                'dict_path']
W
WenmuZhou 已提交
193

194
        # init model dir
195 196 197 198 199 200 201 202 203
        params.det_model_dir, det_url = confirm_model_dir_url(params.det_model_dir,
                                                              os.path.join(BASE_DIR, VERSION, 'det', det_lang),
                                                              model_urls['det'][det_lang])
        params.rec_model_dir, rec_url = confirm_model_dir_url(params.rec_model_dir,
                                                              os.path.join(BASE_DIR, VERSION, 'rec', lang),
                                                              model_urls['rec'][lang]['url'])
        params.cls_model_dir, cls_url = confirm_model_dir_url(params.cls_model_dir,
                                                              os.path.join(BASE_DIR, VERSION, 'cls'),
                                                              model_urls['cls'])
W
WenmuZhou 已提交
204
        # download model
205 206 207
        maybe_download(params.det_model_dir, det_url)
        maybe_download(params.rec_model_dir, rec_url)
        maybe_download(params.cls_model_dir, cls_url)
W
WenmuZhou 已提交
208

W
WenmuZhou 已提交
209
        if params.det_algorithm not in SUPPORT_DET_MODEL:
W
WenmuZhou 已提交
210 211
            logger.error('det_algorithm must in {}'.format(SUPPORT_DET_MODEL))
            sys.exit(0)
W
WenmuZhou 已提交
212
        if params.rec_algorithm not in SUPPORT_REC_MODEL:
W
WenmuZhou 已提交
213 214
            logger.error('rec_algorithm must in {}'.format(SUPPORT_REC_MODEL))
            sys.exit(0)
W
WenmuZhou 已提交
215
        if use_inner_dict:
W
WenmuZhou 已提交
216 217
            params.rec_char_dict_path = str(
                Path(__file__).parent / params.rec_char_dict_path)
W
WenmuZhou 已提交
218

W
WenmuZhou 已提交
219
        print(params)
W
WenmuZhou 已提交
220
        # init det_model and rec_model
W
WenmuZhou 已提交
221
        super().__init__(params)
W
WenmuZhou 已提交
222

223
    def ocr(self, img, det=True, rec=True, cls=True):
W
WenmuZhou 已提交
224 225 226 227 228 229 230 231
        """
        ocr with paddleocr
        args:
            img: img for ocr, support ndarray, img_path and list or ndarray
            det: use text detection or not, if false, only rec will be exec. default is True
            rec: use text recognition or not, if false, only det will be exec. default is True
        """
        assert isinstance(img, (np.ndarray, list, str))
W
WenmuZhou 已提交
232 233 234
        if isinstance(img, list) and det == True:
            logger.error('When input a list of images, det must be false')
            exit(0)
235
        if cls == True and self.use_angle_cls == False:
W
WenmuZhou 已提交
236 237 238
            logger.warning(
                'Since the angle classifier is not initialized, the angle classifier will not be uesd during the forward process'
            )
W
WenmuZhou 已提交
239

W
WenmuZhou 已提交
240
        if isinstance(img, str):
W
WenmuZhou 已提交
241 242 243 244
            # download net image
            if img.startswith('http'):
                download_with_progressbar(img, 'tmp.jpg')
                img = 'tmp.jpg'
W
WenmuZhou 已提交
245 246 247
            image_file = img
            img, flag = check_and_read_gif(image_file)
            if not flag:
248 249 250
                with open(image_file, 'rb') as f:
                    np_arr = np.frombuffer(f.read(), dtype=np.uint8)
                    img = cv2.imdecode(np_arr, cv2.IMREAD_COLOR)
W
WenmuZhou 已提交
251 252 253
            if img is None:
                logger.error("error in loading image:{}".format(image_file))
                return None
W
WenmuZhou 已提交
254 255
        if isinstance(img, np.ndarray) and len(img.shape) == 2:
            img = cv2.cvtColor(img, cv2.COLOR_GRAY2BGR)
W
WenmuZhou 已提交
256
        if det and rec:
257
            dt_boxes, rec_res = self.__call__(img, cls)
W
WenmuZhou 已提交
258 259 260 261 262 263 264 265 266
            return [[box.tolist(), res] for box, res in zip(dt_boxes, rec_res)]
        elif det and not rec:
            dt_boxes, elapse = self.text_detector(img)
            if dt_boxes is None:
                return None
            return [box.tolist() for box in dt_boxes]
        else:
            if not isinstance(img, list):
                img = [img]
267
            if self.use_angle_cls and cls:
W
WenmuZhou 已提交
268 269 270
                img, cls_res, elapse = self.text_classifier(img)
                if not rec:
                    return cls_res
W
WenmuZhou 已提交
271 272
            rec_res, elapse = self.text_recognizer(img)
            return rec_res
273 274 275


def main():
W
WenmuZhou 已提交
276
    # for cmd
W
WenmuZhou 已提交
277
    args = parse_args(mMain=True)
W
WenmuZhou 已提交
278
    image_dir = args.image_dir
279
    if is_link(image_dir):
W
WenmuZhou 已提交
280 281 282 283
        download_with_progressbar(image_dir, 'tmp.jpg')
        image_file_list = ['tmp.jpg']
    else:
        image_file_list = get_image_file_list(args.image_dir)
284 285 286
    if len(image_file_list) == 0:
        logger.error('no images find in {}'.format(args.image_dir))
        return
W
WenmuZhou 已提交
287 288

    ocr_engine = PaddleOCR(**(args.__dict__))
289
    for img_path in image_file_list:
W
WenmuZhou 已提交
290 291 292 293 294 295 296 297
        logger.info('{}{}{}'.format('*' * 10, img_path, '*' * 10))
        result = ocr_engine.ocr(img_path,
                                det=args.det,
                                rec=args.rec,
                                cls=args.use_angle_cls)
        if result is not None:
            for line in result:
                logger.info(line)