langparse 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. langparse/__init__.py +55 -0
  2. langparse/autoparser.py +25 -0
  3. langparse/chunkers/__init__.py +12 -0
  4. langparse/chunkers/blocks.py +151 -0
  5. langparse/chunkers/profiles.py +53 -0
  6. langparse/chunkers/registry.py +38 -0
  7. langparse/chunkers/semantic.py +242 -0
  8. langparse/chunkers/text.py +96 -0
  9. langparse/chunkers/workbook.py +942 -0
  10. langparse/cli.py +329 -0
  11. langparse/config.py +169 -0
  12. langparse/core/__init__.py +0 -0
  13. langparse/core/chunker.py +16 -0
  14. langparse/core/engine.py +37 -0
  15. langparse/core/parser.py +35 -0
  16. langparse/core/rendering.py +49 -0
  17. langparse/engines/__init__.py +1 -0
  18. langparse/engines/pdf/__init__.py +1 -0
  19. langparse/engines/pdf/deepdoc/__init__.py +55 -0
  20. langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
  21. langparse/engines/pdf/deepdoc/model_loader.py +101 -0
  22. langparse/engines/pdf/deepdoc/ocr.py +641 -0
  23. langparse/engines/pdf/deepdoc/operators.py +684 -0
  24. langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
  25. langparse/engines/pdf/deepdoc/postprocess.py +339 -0
  26. langparse/engines/pdf/deepdoc/recognizer.py +418 -0
  27. langparse/engines/pdf/deepdoc/rendering.py +210 -0
  28. langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
  29. langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
  30. langparse/engines/pdf/deepdoc/utils.py +36 -0
  31. langparse/engines/pdf/deepdoc_engine.py +164 -0
  32. langparse/engines/pdf/mineru.py +259 -0
  33. langparse/engines/pdf/mineru_client.py +318 -0
  34. langparse/engines/pdf/mineru_service.py +225 -0
  35. langparse/engines/pdf/ocr.py +101 -0
  36. langparse/engines/pdf/other.py +20 -0
  37. langparse/engines/pdf/simple.py +134 -0
  38. langparse/engines/pdf/vision_llm.py +27 -0
  39. langparse/errors.py +70 -0
  40. langparse/logging.py +27 -0
  41. langparse/metrics.py +129 -0
  42. langparse/parsers/__init__.py +0 -0
  43. langparse/parsers/docx_parser.py +114 -0
  44. langparse/parsers/excel_parser.py +220 -0
  45. langparse/parsers/markdown_parser.py +34 -0
  46. langparse/parsers/pdf_parser.py +31 -0
  47. langparse/parsers/registry.py +48 -0
  48. langparse/parsers/sniff.py +72 -0
  49. langparse/progress.py +77 -0
  50. langparse/py.typed +0 -0
  51. langparse/services/__init__.py +11 -0
  52. langparse/services/batch_service.py +339 -0
  53. langparse/services/benchmark_service.py +202 -0
  54. langparse/services/fidelity.py +154 -0
  55. langparse/services/output_paths.py +86 -0
  56. langparse/services/parse_service.py +523 -0
  57. langparse/services/quality.py +65 -0
  58. langparse/services/workbook_ambiguity_benchmark.py +563 -0
  59. langparse/services/workbook_quality_benchmark.py +230 -0
  60. langparse/types.py +97 -0
  61. langparse/workbooks/__init__.py +103 -0
  62. langparse/workbooks/adapters.py +474 -0
  63. langparse/workbooks/assembly.py +993 -0
  64. langparse/workbooks/blocks.py +209 -0
  65. langparse/workbooks/bundle-v1.schema.json +71 -0
  66. langparse/workbooks/bundle.py +341 -0
  67. langparse/workbooks/classification.py +393 -0
  68. langparse/workbooks/continuation.py +577 -0
  69. langparse/workbooks/evaluation/__init__.py +45 -0
  70. langparse/workbooks/evaluation/evaluator.py +381 -0
  71. langparse/workbooks/evaluation/schema.py +419 -0
  72. langparse/workbooks/labels.py +14 -0
  73. langparse/workbooks/lineage.py +117 -0
  74. langparse/workbooks/modeling/__init__.py +52 -0
  75. langparse/workbooks/modeling/cache.py +20 -0
  76. langparse/workbooks/modeling/config.py +87 -0
  77. langparse/workbooks/modeling/contract.py +628 -0
  78. langparse/workbooks/modeling/disambiguation.py +800 -0
  79. langparse/workbooks/modeling/openai_adapter.py +192 -0
  80. langparse/workbooks/modeling/policy.py +79 -0
  81. langparse/workbooks/modeling/ports.py +44 -0
  82. langparse/workbooks/modeling/pricing.py +17 -0
  83. langparse/workbooks/modeling/types.py +251 -0
  84. langparse/workbooks/objects.py +229 -0
  85. langparse/workbooks/quality/__init__.py +23 -0
  86. langparse/workbooks/quality/bundle.py +53 -0
  87. langparse/workbooks/quality/evaluator.py +266 -0
  88. langparse/workbooks/quality/facts.py +142 -0
  89. langparse/workbooks/quality/schema.py +462 -0
  90. langparse/workbooks/reference_types.py +73 -0
  91. langparse/workbooks/references.py +178 -0
  92. langparse/workbooks/regions.py +932 -0
  93. langparse/workbooks/rendering.py +222 -0
  94. langparse/workbooks/tables.py +477 -0
  95. langparse/workbooks/types.py +257 -0
  96. langparse-0.1.0.dist-info/METADATA +790 -0
  97. langparse-0.1.0.dist-info/RECORD +101 -0
  98. langparse-0.1.0.dist-info/WHEEL +5 -0
  99. langparse-0.1.0.dist-info/entry_points.txt +2 -0
  100. langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
  101. langparse-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,684 @@
1
+ #
2
+ # Copyright 2024 The InfiniFlow Authors. All Rights Reserved.
3
+ #
4
+ # Licensed under the Apache License, Version 2.0 (the "License");
5
+ # you may not use this file except in compliance with the License.
6
+ # You may obtain a copy of the License at
7
+ #
8
+ # http://www.apache.org/licenses/LICENSE-2.0
9
+ #
10
+ # Unless required by applicable law or agreed to in writing, software
11
+ # distributed under the License is distributed on an "AS IS" BASIS,
12
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
13
+ # See the License for the specific language governing permissions and
14
+ # limitations under the License.
15
+ #
16
+
17
+ import logging
18
+ import sys
19
+ import ast
20
+ import cv2
21
+ import numpy as np
22
+ import math
23
+ from PIL import Image
24
+
25
+
26
+ class DecodeImage:
27
+ """decode image"""
28
+
29
+ def __init__(self, img_mode="RGB", channel_first=False, ignore_orientation=False, **kwargs):
30
+ self.img_mode = img_mode
31
+ self.channel_first = channel_first
32
+ self.ignore_orientation = ignore_orientation
33
+
34
+ def __call__(self, data):
35
+ img = data["image"]
36
+ assert isinstance(img, bytes) and len(img) > 0, "invalid input 'img' in DecodeImage"
37
+ img = np.frombuffer(img, dtype="uint8")
38
+ if self.ignore_orientation:
39
+ img = cv2.imdecode(img, cv2.IMREAD_IGNORE_ORIENTATION | cv2.IMREAD_COLOR)
40
+ else:
41
+ img = cv2.imdecode(img, 1)
42
+ if img is None:
43
+ return None
44
+ if self.img_mode == "GRAY":
45
+ img = cv2.cvtColor(img, cv2.COLOR_GRAY2BGR)
46
+ elif self.img_mode == "RGB":
47
+ assert img.shape[2] == 3, "invalid shape of image[%s]" % (img.shape)
48
+ img = img[:, :, ::-1]
49
+
50
+ if self.channel_first:
51
+ img = img.transpose((2, 0, 1))
52
+
53
+ data["image"] = img
54
+ return data
55
+
56
+
57
+ class StandardizeImage:
58
+ """normalize image
59
+ Args:
60
+ mean (list): im - mean
61
+ std (list): im / std
62
+ is_scale (bool): whether need im / 255
63
+ norm_type (str): type in ['mean_std', 'none']
64
+ """
65
+
66
+ def __init__(self, mean, std, is_scale=True, norm_type="mean_std"):
67
+ self.mean = mean
68
+ self.std = std
69
+ self.is_scale = is_scale
70
+ self.norm_type = norm_type
71
+
72
+ def __call__(self, im, im_info):
73
+ """
74
+ Args:
75
+ im (np.ndarray): image (np.ndarray)
76
+ im_info (dict): info of image
77
+ Returns:
78
+ im (np.ndarray): processed image (np.ndarray)
79
+ im_info (dict): info of processed image
80
+ """
81
+ im = im.astype(np.float32, copy=False)
82
+ if self.is_scale:
83
+ scale = 1.0 / 255.0
84
+ im *= scale
85
+
86
+ if self.norm_type == "mean_std":
87
+ mean = np.array(self.mean)[np.newaxis, np.newaxis, :]
88
+ std = np.array(self.std)[np.newaxis, np.newaxis, :]
89
+ im -= mean
90
+ im /= std
91
+ return im, im_info
92
+
93
+
94
+ class NormalizeImage:
95
+ """normalize image such as subtract mean, divide std"""
96
+
97
+ def __init__(self, scale=None, mean=None, std=None, order="chw", **kwargs):
98
+ if isinstance(scale, str):
99
+ try:
100
+ scale = float(scale)
101
+ except ValueError:
102
+ if "/" in scale:
103
+ parts = scale.split("/")
104
+ scale = ast.literal_eval(parts[0]) / ast.literal_eval(parts[1])
105
+ else:
106
+ scale = ast.literal_eval(scale)
107
+ self.scale = np.float32(scale if scale is not None else 1.0 / 255.0)
108
+ mean = mean if mean is not None else [0.485, 0.456, 0.406]
109
+ std = std if std is not None else [0.229, 0.224, 0.225]
110
+
111
+ shape = (3, 1, 1) if order == "chw" else (1, 1, 3)
112
+ self.mean = np.array(mean).reshape(shape).astype("float32")
113
+ self.std = np.array(std).reshape(shape).astype("float32")
114
+
115
+ def __call__(self, data):
116
+ img = data["image"]
117
+ from PIL import Image
118
+
119
+ if isinstance(img, Image.Image):
120
+ img = np.array(img)
121
+ assert isinstance(img, np.ndarray), "invalid input 'img' in NormalizeImage"
122
+ data["image"] = (img.astype("float32") * self.scale - self.mean) / self.std
123
+ return data
124
+
125
+
126
+ class ToCHWImage:
127
+ """convert hwc image to chw image"""
128
+
129
+ def __init__(self, **kwargs):
130
+ pass
131
+
132
+ def __call__(self, data):
133
+ img = data["image"]
134
+ from PIL import Image
135
+
136
+ if isinstance(img, Image.Image):
137
+ img = np.array(img)
138
+ data["image"] = img.transpose((2, 0, 1))
139
+ return data
140
+
141
+
142
+ class KeepKeys:
143
+ def __init__(self, keep_keys, **kwargs):
144
+ self.keep_keys = keep_keys
145
+
146
+ def __call__(self, data):
147
+ data_list = []
148
+ for key in self.keep_keys:
149
+ data_list.append(data[key])
150
+ return data_list
151
+
152
+
153
+ class Pad:
154
+ def __init__(self, size=None, size_div=32, **kwargs):
155
+ if size is not None and not isinstance(size, (int, list, tuple)):
156
+ raise TypeError("Type of target_size is invalid. Now is {}".format(type(size)))
157
+ if isinstance(size, int):
158
+ size = [size, size]
159
+ self.size = size
160
+ self.size_div = size_div
161
+
162
+ def __call__(self, data):
163
+
164
+ img = data["image"]
165
+ img_h, img_w = img.shape[0], img.shape[1]
166
+ if self.size:
167
+ resize_h2, resize_w2 = self.size
168
+ assert img_h < resize_h2 and img_w < resize_w2, "(h, w) of target size should be greater than (img_h, img_w)"
169
+ else:
170
+ resize_h2 = max(int(math.ceil(img.shape[0] / self.size_div) * self.size_div), self.size_div)
171
+ resize_w2 = max(int(math.ceil(img.shape[1] / self.size_div) * self.size_div), self.size_div)
172
+ img = cv2.copyMakeBorder(img, 0, resize_h2 - img_h, 0, resize_w2 - img_w, cv2.BORDER_CONSTANT, value=0)
173
+ data["image"] = img
174
+ return data
175
+
176
+
177
+ class LinearResize:
178
+ """resize image by target_size and max_size
179
+ Args:
180
+ target_size (int): the target size of image
181
+ keep_ratio (bool): whether keep_ratio or not, default true
182
+ interp (int): method of resize
183
+ """
184
+
185
+ def __init__(self, target_size, keep_ratio=True, interp=cv2.INTER_LINEAR):
186
+ if isinstance(target_size, int):
187
+ target_size = [target_size, target_size]
188
+ self.target_size = target_size
189
+ self.keep_ratio = keep_ratio
190
+ self.interp = interp
191
+
192
+ def __call__(self, im, im_info):
193
+ """
194
+ Args:
195
+ im (np.ndarray): image (np.ndarray)
196
+ im_info (dict): info of image
197
+ Returns:
198
+ im (np.ndarray): processed image (np.ndarray)
199
+ im_info (dict): info of processed image
200
+ """
201
+ assert len(self.target_size) == 2
202
+ assert self.target_size[0] > 0 and self.target_size[1] > 0
203
+ _im_channel = im.shape[2]
204
+ im_scale_y, im_scale_x = self.generate_scale(im)
205
+ im = cv2.resize(im, None, None, fx=im_scale_x, fy=im_scale_y, interpolation=self.interp)
206
+ im_info["im_shape"] = np.array(im.shape[:2]).astype("float32")
207
+ im_info["scale_factor"] = np.array([im_scale_y, im_scale_x]).astype("float32")
208
+ return im, im_info
209
+
210
+ def generate_scale(self, im):
211
+ """
212
+ Args:
213
+ im (np.ndarray): image (np.ndarray)
214
+ Returns:
215
+ im_scale_x: the resize ratio of X
216
+ im_scale_y: the resize ratio of Y
217
+ """
218
+ origin_shape = im.shape[:2]
219
+ _im_c = im.shape[2]
220
+ if self.keep_ratio:
221
+ im_size_min = np.min(origin_shape)
222
+ im_size_max = np.max(origin_shape)
223
+ target_size_min = np.min(self.target_size)
224
+ target_size_max = np.max(self.target_size)
225
+ im_scale = float(target_size_min) / float(im_size_min)
226
+ if np.round(im_scale * im_size_max) > target_size_max:
227
+ im_scale = float(target_size_max) / float(im_size_max)
228
+ im_scale_x = im_scale
229
+ im_scale_y = im_scale
230
+ else:
231
+ resize_h, resize_w = self.target_size
232
+ im_scale_y = resize_h / float(origin_shape[0])
233
+ im_scale_x = resize_w / float(origin_shape[1])
234
+ return im_scale_y, im_scale_x
235
+
236
+
237
+ class Resize:
238
+ def __init__(self, size=(640, 640), **kwargs):
239
+ self.size = size
240
+
241
+ def resize_image(self, img):
242
+ resize_h, resize_w = self.size
243
+ ori_h, ori_w = img.shape[:2] # (h, w, c)
244
+ ratio_h = float(resize_h) / ori_h
245
+ ratio_w = float(resize_w) / ori_w
246
+ img = cv2.resize(img, (int(resize_w), int(resize_h)))
247
+ return img, [ratio_h, ratio_w]
248
+
249
+ def __call__(self, data):
250
+ img = data["image"]
251
+ if "polys" in data:
252
+ text_polys = data["polys"]
253
+
254
+ img_resize, [ratio_h, ratio_w] = self.resize_image(img)
255
+ if "polys" in data:
256
+ new_boxes = []
257
+ for box in text_polys:
258
+ new_box = []
259
+ for cord in box:
260
+ new_box.append([cord[0] * ratio_w, cord[1] * ratio_h])
261
+ new_boxes.append(new_box)
262
+ data["polys"] = np.array(new_boxes, dtype=np.float32)
263
+ data["image"] = img_resize
264
+ return data
265
+
266
+
267
+ class DetResizeForTest:
268
+ def __init__(self, **kwargs):
269
+ super(DetResizeForTest, self).__init__()
270
+ self.resize_type = 0
271
+ self.keep_ratio = False
272
+ if "image_shape" in kwargs:
273
+ self.image_shape = kwargs["image_shape"]
274
+ self.resize_type = 1
275
+ if "keep_ratio" in kwargs:
276
+ self.keep_ratio = kwargs["keep_ratio"]
277
+ elif "limit_side_len" in kwargs:
278
+ self.limit_side_len = kwargs["limit_side_len"]
279
+ self.limit_type = kwargs.get("limit_type", "min")
280
+ elif "resize_long" in kwargs:
281
+ self.resize_type = 2
282
+ self.resize_long = kwargs.get("resize_long", 960)
283
+ else:
284
+ self.limit_side_len = 736
285
+ self.limit_type = "min"
286
+
287
+ def __call__(self, data):
288
+ img = data["image"]
289
+ src_h, src_w, _ = img.shape
290
+ if sum([src_h, src_w]) < 64:
291
+ img = self.image_padding(img)
292
+
293
+ if self.resize_type == 0:
294
+ # img, shape = self.resize_image_type0(img)
295
+ img, [ratio_h, ratio_w] = self.resize_image_type0(img)
296
+ elif self.resize_type == 2:
297
+ img, [ratio_h, ratio_w] = self.resize_image_type2(img)
298
+ else:
299
+ # img, shape = self.resize_image_type1(img)
300
+ img, [ratio_h, ratio_w] = self.resize_image_type1(img)
301
+ data["image"] = img
302
+ data["shape"] = np.array([src_h, src_w, ratio_h, ratio_w])
303
+ return data
304
+
305
+ def image_padding(self, im, value=0):
306
+ h, w, c = im.shape
307
+ im_pad = np.zeros((max(32, h), max(32, w), c), np.uint8) + value
308
+ im_pad[:h, :w, :] = im
309
+ return im_pad
310
+
311
+ def resize_image_type1(self, img):
312
+ resize_h, resize_w = self.image_shape
313
+ ori_h, ori_w = img.shape[:2] # (h, w, c)
314
+ if self.keep_ratio is True:
315
+ resize_w = ori_w * resize_h / ori_h
316
+ N = math.ceil(resize_w / 32)
317
+ resize_w = N * 32
318
+ ratio_h = float(resize_h) / ori_h
319
+ ratio_w = float(resize_w) / ori_w
320
+ img = cv2.resize(img, (int(resize_w), int(resize_h)))
321
+ # return img, np.array([ori_h, ori_w])
322
+ return img, [ratio_h, ratio_w]
323
+
324
+ def resize_image_type0(self, img):
325
+ """
326
+ resize image to a size multiple of 32 which is required by the network
327
+ args:
328
+ img(array): array with shape [h, w, c]
329
+ return(tuple):
330
+ img, (ratio_h, ratio_w)
331
+ """
332
+ limit_side_len = self.limit_side_len
333
+ h, w, c = img.shape
334
+
335
+ # limit the max side
336
+ if self.limit_type == "max":
337
+ if max(h, w) > limit_side_len:
338
+ if h > w:
339
+ ratio = float(limit_side_len) / h
340
+ else:
341
+ ratio = float(limit_side_len) / w
342
+ else:
343
+ ratio = 1.0
344
+ elif self.limit_type == "min":
345
+ if min(h, w) < limit_side_len:
346
+ if h < w:
347
+ ratio = float(limit_side_len) / h
348
+ else:
349
+ ratio = float(limit_side_len) / w
350
+ else:
351
+ ratio = 1.0
352
+ elif self.limit_type == "resize_long":
353
+ ratio = float(limit_side_len) / max(h, w)
354
+ else:
355
+ raise Exception("not support limit type, image ")
356
+ resize_h = int(h * ratio)
357
+ resize_w = int(w * ratio)
358
+
359
+ resize_h = max(int(round(resize_h / 32) * 32), 32)
360
+ resize_w = max(int(round(resize_w / 32) * 32), 32)
361
+
362
+ try:
363
+ if int(resize_w) <= 0 or int(resize_h) <= 0:
364
+ return None, (None, None)
365
+ img = cv2.resize(img, (int(resize_w), int(resize_h)))
366
+ except BaseException:
367
+ logging.exception("{} {} {}".format(img.shape, resize_w, resize_h))
368
+ sys.exit(0)
369
+ ratio_h = resize_h / float(h)
370
+ ratio_w = resize_w / float(w)
371
+ return img, [ratio_h, ratio_w]
372
+
373
+ def resize_image_type2(self, img):
374
+ h, w, _ = img.shape
375
+
376
+ resize_w = w
377
+ resize_h = h
378
+
379
+ if resize_h > resize_w:
380
+ ratio = float(self.resize_long) / resize_h
381
+ else:
382
+ ratio = float(self.resize_long) / resize_w
383
+
384
+ resize_h = int(resize_h * ratio)
385
+ resize_w = int(resize_w * ratio)
386
+
387
+ max_stride = 128
388
+ resize_h = (resize_h + max_stride - 1) // max_stride * max_stride
389
+ resize_w = (resize_w + max_stride - 1) // max_stride * max_stride
390
+ img = cv2.resize(img, (int(resize_w), int(resize_h)))
391
+ ratio_h = resize_h / float(h)
392
+ ratio_w = resize_w / float(w)
393
+
394
+ return img, [ratio_h, ratio_w]
395
+
396
+
397
+ class E2EResizeForTest:
398
+ def __init__(self, **kwargs):
399
+ super(E2EResizeForTest, self).__init__()
400
+ self.max_side_len = kwargs["max_side_len"]
401
+ self.valid_set = kwargs["valid_set"]
402
+
403
+ def __call__(self, data):
404
+ img = data["image"]
405
+ src_h, src_w, _ = img.shape
406
+ if self.valid_set == "totaltext":
407
+ im_resized, [ratio_h, ratio_w] = self.resize_image_for_totaltext(img, max_side_len=self.max_side_len)
408
+ else:
409
+ im_resized, (ratio_h, ratio_w) = self.resize_image(img, max_side_len=self.max_side_len)
410
+ data["image"] = im_resized
411
+ data["shape"] = np.array([src_h, src_w, ratio_h, ratio_w])
412
+ return data
413
+
414
+ def resize_image_for_totaltext(self, im, max_side_len=512):
415
+ h, w, _ = im.shape
416
+ resize_w = w
417
+ resize_h = h
418
+ ratio = 1.25
419
+ if h * ratio > max_side_len:
420
+ ratio = float(max_side_len) / resize_h
421
+ resize_h = int(resize_h * ratio)
422
+ resize_w = int(resize_w * ratio)
423
+
424
+ max_stride = 128
425
+ resize_h = (resize_h + max_stride - 1) // max_stride * max_stride
426
+ resize_w = (resize_w + max_stride - 1) // max_stride * max_stride
427
+ im = cv2.resize(im, (int(resize_w), int(resize_h)))
428
+ ratio_h = resize_h / float(h)
429
+ ratio_w = resize_w / float(w)
430
+ return im, (ratio_h, ratio_w)
431
+
432
+ def resize_image(self, im, max_side_len=512):
433
+ """
434
+ resize image to a size multiple of max_stride which is required by the network
435
+ :param im: the resized image
436
+ :param max_side_len: limit of max image size to avoid out of memory in gpu
437
+ :return: the resized image and the resize ratio
438
+ """
439
+ h, w, _ = im.shape
440
+
441
+ resize_w = w
442
+ resize_h = h
443
+
444
+ # Fix the longer side
445
+ if resize_h > resize_w:
446
+ ratio = float(max_side_len) / resize_h
447
+ else:
448
+ ratio = float(max_side_len) / resize_w
449
+
450
+ resize_h = int(resize_h * ratio)
451
+ resize_w = int(resize_w * ratio)
452
+
453
+ max_stride = 128
454
+ resize_h = (resize_h + max_stride - 1) // max_stride * max_stride
455
+ resize_w = (resize_w + max_stride - 1) // max_stride * max_stride
456
+ im = cv2.resize(im, (int(resize_w), int(resize_h)))
457
+ ratio_h = resize_h / float(h)
458
+ ratio_w = resize_w / float(w)
459
+
460
+ return im, (ratio_h, ratio_w)
461
+
462
+
463
+ class KieResize:
464
+ def __init__(self, **kwargs):
465
+ super(KieResize, self).__init__()
466
+ self.max_side, self.min_side = kwargs["img_scale"][0], kwargs["img_scale"][1]
467
+
468
+ def __call__(self, data):
469
+ img = data["image"]
470
+ points = data["points"]
471
+ src_h, src_w, _ = img.shape
472
+ im_resized, scale_factor, [ratio_h, ratio_w], [new_h, new_w] = self.resize_image(img)
473
+ resize_points = self.resize_boxes(img, points, scale_factor)
474
+ data["ori_image"] = img
475
+ data["ori_boxes"] = points
476
+ data["points"] = resize_points
477
+ data["image"] = im_resized
478
+ data["shape"] = np.array([new_h, new_w])
479
+ return data
480
+
481
+ def resize_image(self, img):
482
+ norm_img = np.zeros([1024, 1024, 3], dtype="float32")
483
+ scale = [512, 1024]
484
+ h, w = img.shape[:2]
485
+ max_long_edge = max(scale)
486
+ max_short_edge = min(scale)
487
+ scale_factor = min(max_long_edge / max(h, w), max_short_edge / min(h, w))
488
+ resize_w, resize_h = int(w * float(scale_factor) + 0.5), int(h * float(scale_factor) + 0.5)
489
+ max_stride = 32
490
+ resize_h = (resize_h + max_stride - 1) // max_stride * max_stride
491
+ resize_w = (resize_w + max_stride - 1) // max_stride * max_stride
492
+ im = cv2.resize(img, (resize_w, resize_h))
493
+ new_h, new_w = im.shape[:2]
494
+ w_scale = new_w / w
495
+ h_scale = new_h / h
496
+ scale_factor = np.array([w_scale, h_scale, w_scale, h_scale], dtype=np.float32)
497
+ norm_img[:new_h, :new_w, :] = im
498
+ return norm_img, scale_factor, [h_scale, w_scale], [new_h, new_w]
499
+
500
+ def resize_boxes(self, im, points, scale_factor):
501
+ points = points * scale_factor
502
+ img_shape = im.shape[:2]
503
+ points[:, 0::2] = np.clip(points[:, 0::2], 0, img_shape[1])
504
+ points[:, 1::2] = np.clip(points[:, 1::2], 0, img_shape[0])
505
+ return points
506
+
507
+
508
+ class SRResize:
509
+ def __init__(self, imgH=32, imgW=128, down_sample_scale=4, keep_ratio=False, min_ratio=1, mask=False, infer_mode=False, **kwargs):
510
+ self.imgH = imgH
511
+ self.imgW = imgW
512
+ self.keep_ratio = keep_ratio
513
+ self.min_ratio = min_ratio
514
+ self.down_sample_scale = down_sample_scale
515
+ self.mask = mask
516
+ self.infer_mode = infer_mode
517
+
518
+ def __call__(self, data):
519
+ imgH = self.imgH
520
+ imgW = self.imgW
521
+ images_lr = data["image_lr"]
522
+ transform2 = ResizeNormalize((imgW // self.down_sample_scale, imgH // self.down_sample_scale))
523
+ images_lr = transform2(images_lr)
524
+ data["img_lr"] = images_lr
525
+ if self.infer_mode:
526
+ return data
527
+
528
+ images_HR = data["image_hr"]
529
+ _label_strs = data["label"]
530
+ transform = ResizeNormalize((imgW, imgH))
531
+ images_HR = transform(images_HR)
532
+ data["img_hr"] = images_HR
533
+ return data
534
+
535
+
536
+ class ResizeNormalize:
537
+ def __init__(self, size, interpolation=Image.BICUBIC):
538
+ self.size = size
539
+ self.interpolation = interpolation
540
+
541
+ def __call__(self, img):
542
+ img = img.resize(self.size, self.interpolation)
543
+ img_numpy = np.array(img).astype("float32")
544
+ img_numpy = img_numpy.transpose((2, 0, 1)) / 255
545
+ return img_numpy
546
+
547
+
548
+ class GrayImageChannelFormat:
549
+ """
550
+ format gray scale image's channel: (3,h,w) -> (1,h,w)
551
+ Args:
552
+ inverse: inverse gray image
553
+ """
554
+
555
+ def __init__(self, inverse=False, **kwargs):
556
+ self.inverse = inverse
557
+
558
+ def __call__(self, data):
559
+ img = data["image"]
560
+ img_single_channel = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)
561
+ img_expanded = np.expand_dims(img_single_channel, 0)
562
+
563
+ if self.inverse:
564
+ data["image"] = np.abs(img_expanded - 1)
565
+ else:
566
+ data["image"] = img_expanded
567
+
568
+ data["src_image"] = img
569
+ return data
570
+
571
+
572
+ class Permute:
573
+ """permute image
574
+ Args:
575
+ to_bgr (bool): whether convert RGB to BGR
576
+ channel_first (bool): whether convert HWC to CHW
577
+ """
578
+
579
+ def __init__(
580
+ self,
581
+ ):
582
+ super(Permute, self).__init__()
583
+
584
+ def __call__(self, im, im_info):
585
+ """
586
+ Args:
587
+ im (np.ndarray): image (np.ndarray)
588
+ im_info (dict): info of image
589
+ Returns:
590
+ im (np.ndarray): processed image (np.ndarray)
591
+ im_info (dict): info of processed image
592
+ """
593
+ im = im.transpose((2, 0, 1)).copy()
594
+ return im, im_info
595
+
596
+
597
+ class PadStride:
598
+ """padding image for model with FPN, instead PadBatch(pad_to_stride) in original config
599
+ Args:
600
+ stride (bool): model with FPN need image shape % stride == 0
601
+ """
602
+
603
+ def __init__(self, stride=0):
604
+ self.coarsest_stride = stride
605
+
606
+ def __call__(self, im, im_info):
607
+ """
608
+ Args:
609
+ im (np.ndarray): image (np.ndarray)
610
+ im_info (dict): info of image
611
+ Returns:
612
+ im (np.ndarray): processed image (np.ndarray)
613
+ im_info (dict): info of processed image
614
+ """
615
+ coarsest_stride = self.coarsest_stride
616
+ if coarsest_stride <= 0:
617
+ return im, im_info
618
+ im_c, im_h, im_w = im.shape
619
+ pad_h = int(np.ceil(float(im_h) / coarsest_stride) * coarsest_stride)
620
+ pad_w = int(np.ceil(float(im_w) / coarsest_stride) * coarsest_stride)
621
+ padding_im = np.zeros((im_c, pad_h, pad_w), dtype=np.float32)
622
+ padding_im[:, :im_h, :im_w] = im
623
+ return padding_im, im_info
624
+
625
+
626
+ def decode_image(im_file, im_info):
627
+ """read rgb image
628
+ Args:
629
+ im_file (str|np.ndarray): input can be image path or np.ndarray
630
+ im_info (dict): info of image
631
+ Returns:
632
+ im (np.ndarray): processed image (np.ndarray)
633
+ im_info (dict): info of processed image
634
+ """
635
+ if isinstance(im_file, str):
636
+ with open(im_file, "rb") as f:
637
+ im_read = f.read()
638
+ data = np.frombuffer(im_read, dtype="uint8")
639
+ im = cv2.imdecode(data, 1) # BGR mode, but need RGB mode
640
+ im = cv2.cvtColor(im, cv2.COLOR_BGR2RGB)
641
+ else:
642
+ im = im_file
643
+ im_info["im_shape"] = np.array(im.shape[:2], dtype=np.float32)
644
+ im_info["scale_factor"] = np.array([1.0, 1.0], dtype=np.float32)
645
+ return im, im_info
646
+
647
+
648
+ def preprocess(im, preprocess_ops):
649
+ # process image by preprocess_ops
650
+ im_info = {
651
+ "scale_factor": np.array([1.0, 1.0], dtype=np.float32),
652
+ "im_shape": None,
653
+ }
654
+ im, im_info = decode_image(im, im_info)
655
+ for operator in preprocess_ops:
656
+ im, im_info = operator(im, im_info)
657
+ return im, im_info
658
+
659
+
660
+ def nms(bboxes, scores, iou_thresh):
661
+ import numpy as np
662
+
663
+ x1 = bboxes[:, 0]
664
+ y1 = bboxes[:, 1]
665
+ x2 = bboxes[:, 2]
666
+ y2 = bboxes[:, 3]
667
+ areas = (y2 - y1) * (x2 - x1)
668
+
669
+ indices = []
670
+ index = scores.argsort()[::-1]
671
+ while index.size > 0:
672
+ i = index[0]
673
+ indices.append(i)
674
+ x11 = np.maximum(x1[i], x1[index[1:]])
675
+ y11 = np.maximum(y1[i], y1[index[1:]])
676
+ x22 = np.minimum(x2[i], x2[index[1:]])
677
+ y22 = np.minimum(y2[i], y2[index[1:]])
678
+ w = np.maximum(0, x22 - x11 + 1)
679
+ h = np.maximum(0, y22 - y11 + 1)
680
+ overlaps = w * h
681
+ ious = overlaps / (areas[i] + areas[index[1:]] - overlaps)
682
+ idx = np.where(ious <= iou_thresh)[0]
683
+ index = index[idx + 1]
684
+ return indices