You can not select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.

t_ocr.py 1.9KB

1234567891011121314151617181920212223242526272829303132333435363738394041424344454647
  1. # Licensed under the Apache License, Version 2.0 (the "License");
  2. # you may not use this file except in compliance with the License.
  3. # You may obtain a copy of the License at
  4. #
  5. # http://www.apache.org/licenses/LICENSE-2.0
  6. #
  7. # Unless required by applicable law or agreed to in writing, software
  8. # distributed under the License is distributed on an "AS IS" BASIS,
  9. # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
  10. # See the License for the specific language governing permissions and
  11. # limitations under the License.
  12. #
  13. import os, sys
  14. sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(os.path.abspath(__file__)), '../../')))
  15. import numpy as np
  16. import argparse
  17. from deepdoc.vision import OCR, init_in_out
  18. from deepdoc.vision.seeit import draw_box
  19. def main(args):
  20. ocr = OCR()
  21. images, outputs = init_in_out(args)
  22. for i, img in enumerate(images):
  23. bxs = ocr(np.array(img))
  24. bxs = [(line[0], line[1][0]) for line in bxs]
  25. bxs = [{
  26. "text": t,
  27. "bbox": [b[0][0], b[0][1], b[1][0], b[-1][1]],
  28. "type": "ocr",
  29. "score": 1} for b, t in bxs if b[0][0] <= b[1][0] and b[0][1] <= b[-1][1]]
  30. img = draw_box(images[i], bxs, ["ocr"], 1.)
  31. img.save(outputs[i], quality=95)
  32. with open(outputs[i] + ".txt", "w+") as f: f.write("\n".join([o["text"] for o in bxs]))
  33. if __name__ == "__main__":
  34. parser = argparse.ArgumentParser()
  35. parser.add_argument('--inputs',
  36. help="Directory where to store images or PDFs, or a file path to a single image or PDF",
  37. required=True)
  38. parser.add_argument('--output_dir', help="Directory where to store the output images. Default: './ocr_outputs'",
  39. default="./ocr_outputs")
  40. args = parser.parse_args()
  41. main(args)