C#处理PDF 第一节(读取PDF文档中的文本和图片)

1,136 阅读1分钟

本文介绍如何通过C#程序来读取PDF文档中的文本和图片。

所需工具:Free Spire.PDF for .NET (免费版)

代码示例(供参考)

【示例 1 】提取文本

using Spire.Pdf;
using System;
using System.IO;
using System.Text;


namespace ExtractText
{
class Program
{
static void Main(string[] args)
{
//加载文档
            PdfDocument document = new PdfDocument();
document.LoadFromFile("测试文档.pdf");



        <span class="c1">//实例化StringBuilder类,获取文本




            StringBuilder content = new StringBuilder();
content.Append(document.Pages[0].ExtractText());



        <span class="c1">//保存提取后的文本内容到.txt文档




            String fileName = "TextFromPDF.txt";
File.WriteAllText(fileName, content.ToString());
System.Diagnostics.Process.Start("TextFromPDF.txt");
}
}
}

String fileName = "TextFromPDF.txt"; File.WriteAllText(fileName, content.ToString()); System.Diagnostics.Process.Start("TextFromPDF.txt"); } } }

文本提取效果:


【示例 2 】提取图片

using System;
using System.Collections.Generic;
using System.Text;
using System.Drawing;
using Spire.Pdf;




namespace ExtractImagesFromPDF
{
class Program
{
static void Main(string[] args)
{
//实例化PdfDocument类,并加载测试文档
            PdfDocument doc = new PdfDocument();
doc.LoadFromFile("测试文档.pdf");



        <span class="c1">//实例化List类




            List<Image> ListImage = new List<Image>();
for (int i = 0; i < doc.Pages.Count; i++)
{
// 获取 Spire.Pdf.PdfPageBase类对象
                PdfPageBase page = doc.Pages[i];
// 提取图片
                Image[] images = page.ExtractImages();
if (images != null && images.Length > 0)
{
ListImage.AddRange(images);
}



        <span class="p">}</span>
        <span class="k">if</span> <span class="p">(</span><span class="n">ListImage</span><span class="p">.</span><span class="n">Count</span> <span class="p">&gt;</span> <span class="m">0</span><span class="p">)</span>
        <span class="p">{</span>
            <span class="k">for</span> <span class="p">(</span><span class="kt">int</span> <span class="n">i</span> <span class="p">=</span> <span class="m">0</span><span class="p">;</span> <span class="n">i</span> <span class="p">&lt;</span> <span class="n">ListImage</span><span class="p">.</span><span class="n">Count</span><span class="p">;</span> <span class="n">i</span><span class="p">++)</span>
            <span class="p">{</span>
                <span class="n">Image</span> <span class="n">image</span> <span class="p">=</span> <span class="n">ListImage</span><span class="na">[i]</span><span class="p">;</span>
                <span class="n">image</span><span class="p">.</span><span class="n">Save</span><span class="p">(</span><span class="s">"image"</span> <span class="p">+</span> <span class="p">(</span><span class="n">i</span> <span class="p">+</span> <span class="m">1</span><span class="p">).</span><span class="n">ToString</span><span class="p">()</span> <span class="p">+</span> <span class="s">".png"</span><span class="p">,</span> <span class="n">System</span><span class="p">.</span><span class="n">Drawing</span><span class="p">.</span><span class="n">Imaging</span><span class="p">.</span><span class="n">ImageFormat</span><span class="p">.</span><span class="n">Png</span><span class="p">);</span>
            <span class="p">}</span>
            <span class="n">System</span><span class="p">.</span><span class="n">Diagnostics</span><span class="p">.</span><span class="n">Process</span><span class="p">.</span><span class="n">Start</span><span class="p">(</span><span class="s">"image1.png"</span><span class="p">);</span>
        <span class="p">}</span>
    <span class="p">}</span>
<span class="p">}</span>




}

}

图片提取效果:

(完)