乐于分享
好东西不私藏

PDF 轻量化提取压缩工具

PDF 轻量化提取压缩工具
分享一套基于 PyMuPDF(fitz) + ReportLab 实现的 PDF 纯文本重构压缩方案,适合文字型 PDF(论文、合同、文档类)。一般的压缩是直接压缩图片导致文字不清晰。

核心原理:不直接压缩原 PDF 图片,而是解析页面全部文字坐标、字体、字号,剔除页眉、页脚、页码,再用矢量文字重新生成全新 PDF;大幅降低内嵌图片、冗余资源、重复字体带来的体积膨胀,文字清晰度完全无损,可复制文本。

脚本亮点

  1. 自动弹窗选择 PDF,无需手动填写路径,Windows 原生中文字体自动适配(宋体、黑体、微软雅黑、楷体、仿宋)
  2. 智能识别并清除页眉、页脚、各类格式页码(数字页码、罗马页码、第X页、X/X)
  3. 保留原文坐标位置、字体、字号,排版和原版基本一致
  4. 合并相同字体绘制指令,优化生成效率;输出矢量文本,文字可检索复制
  5. 自动计算压缩率,控制台展示原始大小 / 压缩后大小

适用场景

✅ 纯文字、少量表格的 PDF 文档、扫描转 Word 导出 PDF、电子书、规章制度、论文

❌ 不适合图片扫描版 PDF(图片 PDF 无法提取矢量文字)

import osimport tkinter as tkfrom tkinter import filedialogimport fitzfrom reportlab.pdfgen import canvasfrom reportlab.pdfbase import pdfmetricsfrom reportlab.pdfbase.ttfonts import TTFontfrom reportlab.lib.colors import black, whitefrom PyPDF2 import PdfWriterimport ioimport refrom collections import Counterclass PDFCompressor:    def __init__(self):        self.fonts = {}        font_configs = [            ("C:/Windows/Fonts/simsun.ttc", "SimSun"),            ("C:/Windows/Fonts/simhei.ttf", "SimHei"),            ("C:/Windows/Fonts/msyh.ttc", "YaHei"),            ("C:/Windows/Fonts/simkai.ttf", "KaiTi"),            ("C:/Windows/Fonts/simfang.ttf", "FangSong"),        ]        for font_path, font_name in font_configs:            if os.path.exists(font_path):                try:                    pdfmetrics.registerFont(TTFont(font_name, font_path))                    self.fonts[font_name] = True                except:                    pass        if not self.fonts:            raise Exception("未找到可用中文字体")    def select_file(self):        root = tk.Tk()        root.withdraw()        root.attributes('-topmost', True)        path = filedialog.askopenfilename(title="选择PDF文件", filetypes=[("PDF", "*.pdf")])        root.destroy()        return path    def map_font(self, pdf_font_name):        base_name = pdf_font_name.split('+')[-1] if '+' in pdf_font_name else pdf_font_name        base_name = base_name.split('-')[0]        font_map = {            'SimSun': 'SimSun', 'SimHei': 'SimHei',            'KaiTi': 'KaiTi', 'FangSong': 'FangSong',            'MicrosoftYaHei': 'YaHei', 'MSYaHei': 'YaHei',        }        return font_map.get(base_name, 'SimSun')    def is_page_number(self, text, y_pos, page_height, is_footer_zone):        """判断文本是否为页码"""        text = text.strip()        if re.match(r'^~?\s*\d+\s*~?$', text):            return True        if re.match(r'^[IVX]+\.?$', text):            return True        if re.match(r'^第\s*\d+\s*页$', text):            return True        if re.match(r'^\d+\s*/\s*\d+$', text):            return True        return False    def extract_text(self, pdf_path):        pdf = fitz.open(pdf_path)        all_pages = []        all_y_positions = []        for page_num in range(pdf.page_count):            page = pdf[page_num]            page_rect = page.rect            text_dict = page.get_text("dict")            text_blocks = []            for block in text_dict["blocks"]:                if block["type"] == 0:                    for line in block["lines"]:                        line_text = ""                        line_fonts = []                        line_sizes = []                        line_bbox = None                        for span in line["spans"]:                            line_text += span["text"]                            line_fonts.append(span["font"])                            line_sizes.append(span["size"])                            if line_bbox is None:                                line_bbox = list(span["bbox"])                            else:                                line_bbox[0] = min(line_bbox[0], span["bbox"][0])                                line_bbox[1] = min(line_bbox[1], span["bbox"][1])                                line_bbox[2] = max(line_bbox[2], span["bbox"][2])                                line_bbox[3] = max(line_bbox[3], span["bbox"][3])                        if line_text.strip():                            font_counter = Counter(line_fonts)                            size_counter = Counter(line_sizes)                            y_pos = line_bbox[1]                            all_y_positions.append(y_pos)                            text_blocks.append({                                'text': line_text,                                'x': line_bbox[0],                                'y': y_pos,                                'font': self.map_font(font_counter.most_common(1)[0][0]),                                'size': size_counter.most_common(1)[0][0],                                'page_height': page_rect.height,                            })            text_blocks.sort(key=lambda b: (b['y'], b['x']))            all_pages.append({                'blocks': text_blocks,                'width': page_rect.width,                'height': page_rect.height,            })            print(f"第 {page_num + 1}/{pdf.page_count} 页 - {len(text_blocks)} 个文本块")        pdf.close()        header_threshold, footer_threshold = self.detect_header_footer(all_y_positions)        for page_data in all_pages:            page_height = page_data['height']            filtered_blocks = []            for block in page_data['blocks']:                y = block['y']                if y <= header_threshold:                    continue                if y >= (page_height - footer_threshold):                    if self.is_page_number(block['text'], y, page_height, True):                        continue                    continue                if self.is_page_number(block['text'], y, page_height, False):                    if '~' in block['text']:                        filtered_blocks.append(block)                    continue                filtered_blocks.append(block)            page_data['blocks'] = filtered_blocks        return all_pages    def detect_header_footer(self, y_positions):        """检测页眉页脚区域"""        if not y_positions:            return 0, 0        page_height = max(y_positions)        header_zone = [y for y in y_positions if y < page_height * 0.15]        header_threshold = 0        if header_zone:            header_counter = Counter([round(y, -1) for y in header_zone])            header_y = header_counter.most_common(1)[0][0]            header_threshold = header_y + 20        footer_zone = [y for y in y_positions if y > page_height * 0.85]        footer_threshold = 0        if footer_zone:            footer_counter = Counter([round(y, -1) for y in footer_zone])            footer_y = footer_counter.most_common(1)[0][0]            footer_threshold = page_height - footer_y + 20        print(f"页眉阈值: {header_threshold:.0f}, 页脚阈值: {footer_threshold:.0f}")        return header_threshold, footer_threshold    def compress(self, pages_data, output_path):        temp_files = []        for page_idx, page_data in enumerate(pages_data):            buffer = io.BytesIO()            c = canvas.Canvas(buffer, pagesize=(page_data['width'], page_data['height']))            # 启用压缩            c.setPageCompression(1)            c.setFillColor(white)            c.rect(0, 0, page_data['width'], page_data['height'], fill=1)            c.setFillColor(black)            # 合并相同字体的连续文本,减少canvas操作            current_font = None            current_size = None            for block in page_data['blocks']:                font_name = block['font'] if block['font'] in self.fonts else 'SimSun'                font_size = block['size']                # 只在字体或大小改变时设置                if font_name != current_font or font_size != current_size:                    c.setFont(font_name, font_size)                    current_font = font_name                    current_size = font_size                x = block['x']                y = page_data['height'] - block['y'] - block['size']                c.drawString(x, y, block['text'])            c.save()            buffer.seek(0)            temp_path = f"_temp_{page_idx}.pdf"            with open(temp_path, 'wb') as f:                f.write(buffer.getvalue())            temp_files.append(temp_path)        # 使用PdfWriter合并并压缩        merger = PdfWriter()        for temp in temp_files:            merger.append(temp)        with open(output_path, 'wb') as f:            merger.write(f)        merger.close()        for temp in temp_files:            os.remove(temp)    def run(self):        pdf_path = self.select_file()        if not pdf_path:            return        original_size = os.path.getsize(pdf_path) / (1024 * 1024)        print(f"处理: {os.path.basename(pdf_path)} ({original_size:.1f}MB)")        pages_data = self.extract_text(pdf_path)        output_path = os.path.splitext(pdf_path)[0] + "-压缩.pdf"        self.compress(pages_data, output_path)        new_size = os.path.getsize(output_path) / (1024 * 1024)        print(f"完成! {os.path.basename(output_path)} ({new_size:.2f}MB, 压缩 {(1-new_size/original_size)*100:.0f}%)")if __name__ == "__main__":    PDFCompressor().run()

相关学习资料