pdf 附件可以帮助用户查看与文档中特定内容相关的更多信息。pdf 中的附件主要分为两种:文档级附件和注释附件,二者的区别如下。
- 文档级附件:这类附件直接添加到 pdf 文档中,不会显示在页面上,而是显示在 pdf 阅读器的“附件”面板中。
- 注释附件:这类附件添加到页面的特定位置,通常以回形针图标的形式显示。查看者可以双击该图标打开附件。
本文将介绍如何使用 c# 和 vb.net 从 pdf 文档中提取这两种类型的附件。
安装相关库
首先,需要将相关库文件添加为 .net 项目的引用。你可以下载对应的安装包,并将其中的 dll 文件添加到项目中;也可以通过 nuget 包管理器安装所需的库。
pm> install-package spire.pdf
使用 c# 和 vb.net 从 pdf 中提取附件
可以通过 pdfdocument.attachments 属性获取 pdf 文档中的文档级附件。以下步骤演示如何提取 pdf 中的所有文档级附件,并将其保存到本地文件夹。
- 创建一个
pdfdocument对象。 - 使用
pdfdocument.loadfromfile()方法加载 pdf 文件。 - 通过
pdfdocument.attachments属性获取文档中的附件集合。 - 使用
pdfattachment.data属性获取指定附件的数据。 - 将附件数据写入文件,并保存到指定文件夹。
完整示例代码如下:
using spire.pdf;
using spire.pdf.attachments;
using system.net.mail;
namespace extractattachments
{
class program
{
static void main(string[] args)
{
// 创建 pdfdocument 对象
pdfdocument doc = new pdfdocument();
// 加载包含附件的 pdf 文件
doc.loadfromfile("c:\\users\\administrator\\desktop\\attachments.pdf");
// 获取 pdf 文档的附件集合
pdfattachmentcollection attachments = doc.attachments;
// 指定输出文件夹路径
string outputfolder = "c:\\users\\administrator\\desktop\\output\\";
// 遍历附件集合
for (int i = 0; i < attachments.count; i++)
{
// 将附件写入文件
file.writeallbytes(outputfolder + attachments[i].filename, attachments[i].data);
}
}
}
}使用 c# 和 vb.net 从 pdf 中提取注释附件
注释附件是与 pdf 页面相关联的元素。若要获取指定页面中的注释,可以使用 pdfpagebase.annotationsv2 属性。随后,需要判断某个注释是否为文件附件注释。
以下步骤介绍如何从 pdf 文档中提取注释附件,并将其保存到本地文件夹。
- 创建一个
pdfdocument对象。 - 使用
pdfdocument.loadfromfile()方法加载 pdf 文件。 - 通过
pdfdocument.pages[]属性获取文档中的指定页面。 - 使用
pdfpagebase.annotationsv2属性获取该页面的注释集合。 - 判断指定注释是否属于
pdffileattachmentannotation类型。如果是,则将注释附件写入文件,并保存到指定文件夹。
完整示例代码如下:
using spire.pdf;
using spire.pdf.interactive.annotations;
namespace extractannotationattachments
{
class program
{
static void main(string[] args)
{
// 创建 pdfdocument 对象
pdfdocument doc = new pdfdocument();
// 加载包含附件的 pdf 文件
doc.loadfromfile("c:\\users\\administrator\\desktop\\annotationattachments.pdf");
// 指定输出文件夹路径
string outputfolder = "c:\\users\\administrator\\desktop\\output\\";
// 遍历所有页面
for (int i = 0; i < doc.pages.count; i++)
{
// 获取页面的注释集合
pdfannotationcollection collection = doc.pages[i].annotationsv2;
// 遍历所有注释
for (int j = 0; j < collection.count; j++)
{
// 判断注释是否为 pdffileattachmentannotation 类型
if (collection[j] is pdffileattachmentannotation)
{
// 将注释附件写入文件
pdffileattachmentannotation attachmentannotation =
(pdffileattachmentannotation)collection[j];
pdffilespecification filespecification = attachmentannotation.filespecification;
// 获取附件文件名
string filename = path.getfilename(filespecification.filename);
// 获取附件数据
byte[] data = filespecification.embeddedfile.data;
// 将附件保存到指定文件夹
file.writeallbytes(outputfolder + filename, data);
}
}
}
}
}
}知识扩展
从 pdf 中提取附件,核心是解析 pdf 内部的 embeddedfiles 名称树(文档级附件)和 fileattachment 注释(页面级附件)。下面给出几种主流 c# 库的完整实现,并附上方案对比与注意事项。
1.使用 itext7(开源 / agpl,推荐)
itext7 可以完整读取 pdf 的嵌入文件名称树,适合需要免费开源方案的场景。注意:agpl 协议,商业闭源使用需购买许可。
安装
install-package itext7
完整代码
using system;
using system.collections.generic;
using system.io;
using itext.kernel.pdf;
namespace pdfattachmentextractor
{
class program
{
static void main(string[] args)
{
string pdfpath = @"c:\temp\input.pdf";
string outputdir = @"c:\temp\attachments";
extractallattachments(pdfpath, outputdir);
}
public static void extractallattachments(string pdfpath, string outputdir)
{
directory.createdirectory(outputdir);
using (pdfdocument pdfdoc = new pdfdocument(new pdfreader(pdfpath)))
{
pdfreader reader = pdfdoc.getreader();
// 1. 提取文档级嵌入文件(embeddedfiles 名称树)
extractdocumentembeddedfiles(pdfdoc, reader, outputdir);
// 2. 提取页面注释级文件附件(fileattachment annots)
extractannotationattachments(pdfdoc, reader, outputdir);
}
}
private static void extractdocumentembeddedfiles(pdfdocument pdfdoc, pdfreader reader, string outputdir)
{
pdfdictionary catalog = pdfdoc.getcatalog().getpdfobject();
pdfdictionary names = catalog.getasdictionary(pdfname.names);
if (names == null) return;
pdfdictionary embeddedfiles = names.getasdictionary(pdfname.embeddedfiles);
if (embeddedfiles == null) return;
pdfnametree nametree = new pdfnametree(embeddedfiles);
idictionary<string, pdfobject> files = nametree.getnames();
foreach (var kv in files)
{
string fallbackname = kv.key;
if (kv.value is pdfdictionary filespec)
{
extractfilespec(filespec, fallbackname, outputdir, reader);
}
}
}
private static void extractannotationattachments(pdfdocument pdfdoc, pdfreader reader, string outputdir)
{
for (int i = 1; i <= pdfdoc.getnumberofpages(); i++)
{
pdfpage page = pdfdoc.getpage(i);
pdfarray annots = page.getpdfobject().getasarray(pdfname.annots);
if (annots == null) continue;
foreach (pdfobject annotobj in annots)
{
if (annotobj is pdfdictionary annot)
{
pdfname subtype = annot.getasname(pdfname.subtype);
if (subtype != null && subtype.equals(new pdfname("fileattachment")))
{
pdfdictionary fs = annot.getasdictionary(new pdfname("fs"));
if (fs != null)
{
extractfilespec(fs, $"page{i}_annotation", outputdir, reader);
}
}
}
}
}
}
private static void extractfilespec(pdfdictionary filespec, string fallbackname, string outputdir, pdfreader reader)
{
string name = fallbackname;
// 优先使用 uf(unicode 文件名),其次使用 f
pdfstring uf = filespec.getasstring(pdfname.uf);
if (uf != null)
{
name = uf.tounicodestring();
}
else
{
pdfstring f = filespec.getasstring(pdfname.f);
if (f != null) name = f.tounicodestring();
}
pdfdictionary ef = filespec.getasdictionary(pdfname.ef);
if (ef == null) return;
pdfstream stream = ef.getasstream(pdfname.f) ?? ef.getasstream(pdfname.uf);
if (stream == null) return;
// 使用 reader.getstreambytes 自动解码 flatedecode 等压缩流
byte[] data = reader.getstreambytes(stream);
string safename = getsafefilename(name);
string outpath = path.combine(outputdir, safename);
file.writeallbytes(outpath, data);
console.writeline($"提取: {safename} ({data.length} 字节) -> {outpath}");
}
private static string getsafefilename(string name)
{
if (string.isnullorwhitespace(name)) return "attachment.bin";
// 防止路径穿越:只取文件名部分
name = path.getfilename(name);
foreach (char c in path.getinvalidfilenamechars())
{
name = name.replace(c, '_');
}
return string.isnullorwhitespace(name) ? "attachment.bin" : name;
}
}
}2.使用 aspose.pdf(商业库,代码最简洁)
aspose.pdf 通过 document.embeddedfiles 直接遍历所有嵌入文件,api 非常简洁。
安装
install-package aspose.pdf
代码
using aspose.pdf;
using system;
using system.io;
public static class asposeextractor
{
public static void extract(string pdfpath, string outputdir)
{
directory.createdirectory(outputdir);
document pdf = new document(pdfpath);
foreach (filespecification filespec in pdf.embeddedfiles)
{
string safename = getsafefilename(filespec.name);
string outpath = path.combine(outputdir, safename);
using (filestream fs = new filestream(outpath, filemode.create, fileaccess.write))
{
filespec.contents.copyto(fs);
}
console.writeline($"提取: {safename}");
}
}
private static string getsafefilename(string name)
{
if (string.isnullorwhitespace(name)) return "attachment.bin";
name = path.getfilename(name);
foreach (char c in path.getinvalidfilenamechars())
name = name.replace(c, '_');
return string.isnullorwhitespace(name) ? "attachment.bin" : name;
}
}3.使用 spire.pdf(商业库,有免费版)
spire.pdf 通过 pdfdocument.attachments 集合提取附件。
安装
install-package spire.pdf
代码
using spire.pdf;
using spire.pdf.attachments;
using system;
using system.io;
public static class spireextractor
{
public static void extract(string pdfpath, string outputdir)
{
directory.createdirectory(outputdir);
pdfdocument doc = new pdfdocument();
doc.loadfromfile(pdfpath);
foreach (pdfattachment attachment in doc.attachments)
{
string safename = getsafefilename(attachment.filename);
string outpath = path.combine(outputdir, safename);
file.writeallbytes(outpath, attachment.data);
console.writeline($"提取: {safename}");
}
doc.close();
}
private static string getsafefilename(string name)
{
if (string.isnullorwhitespace(name)) return "attachment.bin";
name = path.getfilename(name);
foreach (char c in path.getinvalidfilenamechars())
name = name.replace(c, '_');
return string.isnullorwhitespace(name) ? "attachment.bin" : name;
}
}4.使用 syncfusion pdf(商业库,有社区版)
syncfusion 通过 pdfloadeddocument.attachments 提取附件。
安装
install-package syncfusion.pdf.net.core
代码
using syncfusion.pdf.parsing;
using system;
using system.io;
public static class syncfusionextractor
{
public static void extract(string pdfpath, string outputdir)
{
directory.createdirectory(outputdir);
using (pdfloadeddocument doc = new pdfloadeddocument(pdfpath))
{
foreach (pdfattachment attachment in doc.attachments)
{
string safename = getsafefilename(attachment.filename);
string outpath = path.combine(outputdir, safename);
file.writeallbytes(outpath, attachment.data);
console.writeline($"提取: {safename}");
}
}
}
private static string getsafefilename(string name)
{
if (string.isnullorwhitespace(name)) return "attachment.bin";
name = path.getfilename(name);
foreach (char c in path.getinvalidfilenamechars())
name = name.replace(c, '_');
return string.isnullorwhitespace(name) ? "attachment.bin" : name;
}
}总结
本文介绍了如何使用 c# 从 pdf 文档中提取附件,主要包括文档级附件和注释附件两种类型。
对于文档级附件,可以通过 pdfdocument.attachments 属性获取附件集合,并遍历集合,将附件数据保存到本地文件夹。对于注释附件,则需要遍历 pdf 文档的各个页面,获取页面中的注释集合,并判断注释是否属于 pdffileattachmentannotation 类型。确认后,即可提取附件的文件名和数据,并将其保存为本地文件。
通过以上方法,可以根据附件在 pdf 文档中的存储方式,选择合适的提取方法,便于后续查看、管理和处理 pdf 中的嵌入文件。
到此这篇关于c#代码轻松实现从pdf文档中提取附件的文章就介绍到这了,更多相关c#提取pdf附件内容请搜索代码网以前的文章或继续浏览下面的相关文章希望大家以后多多支持代码网!
发表评论