C# 读取PDF文本和图片

转自:https://zhuanlan.zhihu.com/p/137197267

本文介绍如何通过C#程序来读取PDF文档中的文本好图片。

所需工具:Free Spire.PDF for .NET (免费版)

代码示例(供参考)

【示例 1 】提取文本

using Spire.Pdf;
using System;
using System.IO;
using System.Text;

namespace ExtractText
{
class Program
{
static void Main(string[] args)
{
//加载文档
PdfDocument document = new PdfDocument();
document.LoadFromFile(“测试文档.pdf”);

        <span class="c1">//实例化StringBuilder类,获取文本

StringBuilder content = new StringBuilder();
content.Append(document.Pages[0].ExtractText());

        <span class="c1">//保存提取后的文本内容到.txt文档

String fileName = “TextFromPDF.txt”;
File.WriteAllText(fileName, content.ToString());
System.Diagnostics.Process.Start(“TextFromPDF.txt”);
}
}
}

文本提取效果:


【示例 2 】提取图片

using System;
using System.Collections.Generic;
using System.Text;
using System.Drawing;
using Spire.Pdf;

namespace ExtractImagesFromPDF
{
class Program
{
static void Main(string[] args)
{
//实例化PdfDocument类,并加载测试文档
PdfDocument doc = new PdfDocument();
doc.LoadFromFile(“测试文档.pdf”);

        <span class="c1">//实例化List类

List<Image> ListImage = new List<Image>();
for (int i = 0; i < doc.Pages.Count; i++)
{
// 获取 Spire.Pdf.PdfPageBase类对象
PdfPageBase page = doc.Pages[i];
// 提取图片
Image[] images = page.ExtractImages();
if (images != null && images.Length > 0)
{
ListImage.AddRange(images);
}

        <span class="p">}</span>
        <span class="k">if</span> <span class="p">(</span><span class="n">ListImage</span><span class="p">.</span><span class="n">Count</span> <span class="p">&gt;</span> <span class="m">0</span><span class="p">)</span>
        <span class="p">{</span>
            <span class="k">for</span> <span class="p">(</span><span class="kt">int</span> <span class="n">i</span> <span class="p">=</span> <span class="m">0</span><span class="p">;</span> <span class="n">i</span> <span class="p">&lt;</span> <span class="n">ListImage</span><span class="p">.</span><span class="n">Count</span><span class="p">;</span> <span class="n">i</span><span class="p">++)</span>
            <span class="p">{</span>
                <span class="n">Image</span> <span class="n">image</span> <span class="p">=</span> <span class="n">ListImage</span><span class="na">[i]</span><span class="p">;</span>
                <span class="n">image</span><span class="p">.</span><span class="n">Save</span><span class="p">(</span><span class="s">"image"</span> <span class="p">+</span> <span class="p">(</span><span class="n">i</span> <span class="p">+</span> <span class="m">1</span><span class="p">).</span><span class="n">ToString</span><span class="p">()</span> <span class="p">+</span> <span class="s">".png"</span><span class="p">,</span> <span class="n">System</span><span class="p">.</span><span class="n">Drawing</span><span class="p">.</span><span class="n">Imaging</span><span class="p">.</span><span class="n">ImageFormat</span><span class="p">.</span><span class="n">Png</span><span class="p">);</span>
            <span class="p">}</span>
            <span class="n">System</span><span class="p">.</span><span class="n">Diagnostics</span><span class="p">.</span><span class="n">Process</span><span class="p">.</span><span class="n">Start</span><span class="p">(</span><span class="s">"image1.png"</span><span class="p">);</span>
        <span class="p">}</span>
    <span class="p">}</span>
<span class="p">}</span>

}

图片提取效果:

(完)

  • 1
    点赞
  • 14
    收藏
    觉得还不错? 一键收藏
  • 2
    评论
评论 2
添加红包

请填写红包祝福语或标题

红包个数最小为10个

红包金额最低5元

当前余额3.43前往充值 >
需支付:10.00
成就一亿技术人!
领取后你会自动成为博主和红包主的粉丝 规则
hope_wisdom
发出的红包
实付
使用余额支付
点击重新获取
扫码支付
钱包余额 0

抵扣说明:

1.余额是钱包充值的虚拟货币,按照1:1的比例进行支付金额的抵扣。
2.余额无法直接购买下载,可以购买VIP、付费专栏及课程。

余额充值