ARTICLE DETAIL

资讯详情

深耕郑州网站建设与运营推广的一线实战洞察。

Python 使用Python从PDF中提取图像,保存到本地

Python 使用Python从PDF中提取图像,保存到本地 目录1.配置环境2.编码1保存为png图片2保存为jpg图片1.配置环境Python3.8.20pip install PyPDF2 pillow(demo_env) C:\Users\asuspython -V Python 3.8.20 (demo_env) C:\Users\asuspip list Package Version ---------------------------- ----------- absl-py 2.1.0 aiofiles 24.1.0 anyio 4.5.2 asttokens 3.0.0 astunparse 1.6.3 backcall 0.2.0 cachetools 5.5.0 certifi 2024.12.14 charset-normalizer 3.4.1 ci-info 0.3.0 click 8.1.8 colorama 0.4.6 configobj 5.0.9 configparser 7.1.0 decimer 2.7.1 decorator 5.1.1 efficientnet 1.1.1 etelemetry 0.3.1 exceptiongroup 1.2.2 executing 2.1.0 filelock 3.16.1 fitz 0.0.1.dev2 flatbuffers 24.12.23 frontend 0.0.3 gast 0.4.0 google-auth 2.37.0 google-auth-oauthlib 1.0.0 google-pasta 0.2.0 grpcio 1.69.0 h11 0.14.0 h5py 3.11.0 httplib2 0.22.0 idna 3.10 imageio 2.35.1 importlib_metadata 8.5.0 importlib_resources 6.4.5 ipython 8.12.3 isodate 0.6.1 itsdangerous 2.2.0 jedi 0.19.2 keras 2.13.1 Keras-Applications 1.0.8 lazy_loader 0.4 libclang 18.1.1 looseversion 1.3.0 lxml 5.3.0 Markdown 3.7 MarkupSafe 2.1.5 matplotlib-inline 0.1.7 networkx 3.1 nibabel 5.2.1 nipype 1.8.6 numpy 1.24.3 oauthlib 3.2.2 opencv-python 4.10.0.84 opt_einsum 3.4.0 packaging 24.2 pandas 2.0.3 parso 0.8.4 pathlib 1.0.1 pickleshare 0.7.5 pillow 10.4.0 pillow_heif 0.18.0 pip 24.2 plum-dispatch 1.7.4 prompt_toolkit 3.0.48 protobuf 4.25.5 prov 2.0.1 pure_eval 0.2.3 pyasn1 0.6.1 pyasn1_modules 0.4.1 pydot 3.0.4 Pygments 2.19.1 pyparsing 3.1.4 PyPDF2 3.0.1 pystow 0.5.6 python-dateutil 2.9.0.post0 pytz 2024.2 PyWavelets 1.4.1 pyxnat 1.6.2 PyYAML 6.0.2 rdflib 6.3.2 rdkit 2024.3.5 requests 2.32.3 requests-oauthlib 2.0.0 rsa 4.9 scikit-image 0.21.0 scipy 1.10.1 selfies 2.1.2 setuptools 75.1.0 simplejson 3.19.3 six 1.17.0 sniffio 1.3.1 Spire.Pdf 10.12.1 stack-data 0.6.3 starlette 0.44.0 tensorboard 2.13.0 tensorboard-data-server 0.7.2 tensorflow 2.13.0 tensorflow-estimator 2.13.0 tensorflow-intel 2.13.0 tensorflow-io-gcs-filesystem 0.31.0 termcolor 2.4.0 tifffile 2023.7.10 tqdm 4.67.1 traitlets 5.14.3 traits 6.3.2 typing_extensions 4.5.0 tzdata 2024.2 urllib3 2.2.3 uvicorn 0.33.0 wcwidth 0.2.13 Werkzeug 3.0.6 wheel 0.44.0 wrapt 1.17.0 zipp 3.20.2 (demo_env) C:\Users\asus2.编码1保存为png图片# 虚拟环境demo_env # 从pdf文件中提取图片 # 按照第三方依赖包 pip install PyPDF2 pillow import PyPDF2 from PIL import Image import io import os import re def sanitize_filename(filename): 清理文件名中的特殊字符 return re.sub(r[\\/*?:|], , filename) def extract_images_from_pdf(file_path, output_diroutput): try: # 确保输出目录存在 if not os.path.exists(output_dir): os.makedirs(output_dir, exist_okTrue) with open(file_path, rb) as pdf_file: pdf_reader PyPDF2.PdfReader(pdf_file) for page_number, page in enumerate(pdf_reader.pages): if /XObject in page[/Resources]: x_object page[/Resources][/XObject].get_object() for obj in x_object: if x_object[obj][/Subtype] /Image: image_data x_object[obj]._data with Image.open(io.BytesIO(image_data)) as image: # 清理文件名并确保文件名符合预期 sanitized_obj sanitize_filename(str(obj)) image_filename fimage_{page_number 1}_{sanitized_obj}.png image.save(os.path.join(output_dir, image_filename)) except FileNotFoundError: print(fError: The file {file_path} does not exist.) except PermissionError: print(fError: Permission denied when accessing {file_path}.) except PyPDF2.errors.PdfReadError: print(fError: Failed to read the PDF file {file_path}.) except Exception as e: print(fAn unexpected error occurred: {e}) # 从PDF中提取图像并保存 extract_images_from_pdf(data/abc.pdf, result/abc)2保存为jpg图片# 虚拟环境demo_env # 从pdf文件中提取图片 # 按照第三方依赖包 pip install PyPDF2 pillow # GRBA转GRB用于目标检测已调通 # 当pdf文件中图片是PNG时可用 # 当pdf文件中包含JBIG2图片时报错An unexpected error occurred: cannot identify image file _io.BytesIO object at 0x000001CAA96CB180 import PyPDF2 from PIL import Image import io import os import re def sanitize_filename(filename): 清理文件名中的特殊字符 return re.sub(r[\\/*?:|], , filename) def extract_images_from_pdf(file_path, output_diroutput): try: # 确保输出目录存在 if not os.path.exists(output_dir): os.makedirs(output_dir, exist_okTrue) with open(file_path, rb) as pdf_file: pdf_reader PyPDF2.PdfReader(pdf_file) for page_number, page in enumerate(pdf_reader.pages): if /XObject in page[/Resources]: x_object page[/Resources][/XObject].get_object() for obj in x_object: if x_object[obj][/Subtype] /Image: image_data x_object[obj]._data with Image.open(io.BytesIO(image_data)) as image: # GRBA转GRB image image.convert(RGB) # 清理文件名并确保文件名符合预期 sanitized_obj sanitize_filename(str(obj)) image_filename fimage_{page_number 1}_{sanitized_obj}.jpg image.save(os.path.join(output_dir, image_filename), JPEG) except FileNotFoundError: print(fError: The file {file_path} does not exist.) except PermissionError: print(fError: Permission denied when accessing {file_path}.) except PyPDF2.errors.PdfReadError: print(fError: Failed to read the PDF file {file_path}.) except Exception as e: print(fAn unexpected error occurred: {e}) # 从PDF中提取图像并保存 extract_images_from_pdf(data/CN 117209460 A.pdf, result/CN 117209460 A)
返回列表