【发布时间】:2015-07-28 23:17:04
【问题描述】:
我已经使用 fs 通过 MEAN 堆栈 Web 应用程序上传了一个 pdf。我想从 pdf 中提取某些字段并将它们显示在网络应用程序上。我看过几个 npm 包,比如 pdf.js、pdf2json。我无法弄清楚可用示例中使用的文档和 javascript 回调。请帮忙!
【问题讨论】:
标签: javascript node.js pdf
我已经使用 fs 通过 MEAN 堆栈 Web 应用程序上传了一个 pdf。我想从 pdf 中提取某些字段并将它们显示在网络应用程序上。我看过几个 npm 包,比如 pdf.js、pdf2json。我无法弄清楚可用示例中使用的文档和 javascript 回调。请帮忙!
【问题讨论】:
标签: javascript node.js pdf
我希望我能帮助回答你的问题。使用 pdf2json 可用于解析 pdf 并提取文本。需要采取几个步骤才能使其正常工作。我改编了来自https://github.com/modesty/pdf2json 的示例。
设置是在node app中安装pdf2json,也下划线。示例页面没有解释定义您自己的回调函数的必要性。它还使用self 而不是this 来注册它们。因此,通过适当的更改,从 pdf 中提取所有文本的代码将如下所示:
// Get the dependencies that have already been installed
// to ./node_modules with `npm install <dep>`in the root director
// of your app
var _ = require('underscore'),
PDFParser = require('pdf2json');
var pdfParser = new PDFParser();
// Create a function to handle the pdf once it has been parsed.
// In this case we cycle through all the pages and extraxt
// All the text blocks and print them to console.
// If you do `console.log(JSON.stringify(pdf))` you will
// see how the parsed pdf is composed. Drill down into it
// to find the data you are looking for.
var _onPDFBinDataReady = function (pdf) {
console.log('Loaded pdf:\n');
for (var i in pdf.data.Pages) {
var page = pdf.data.Pages[i];
for (var j in page.Texts) {
var text = page.Texts[j];
console.log(text.R[0].T);
}
}
};
// Create an error handling function
var _onPDFBinDataError = function (error) {
console.log(error);
};
// Use underscore to bind the data ready function to the pdfParser
// so that when the data ready event is emitted your function will
// be called. As opposed to the example, I have used `this` instead
// of `self` since self had no meaning in this context
pdfParser.on('pdfParser_dataReady', _.bind(_onPDFBinDataReady, this));
// Register error handling function
pdfParser.on('pdfParser_dataError', _.bind(_onPDFBinDataError, this));
// Construct the file path of the pdf
var pdfFilePath = 'test3.pdf';
// Load the pdf. When it is loaded your data ready function will be called.
pdfParser.loadPDF(pdfFilePath);
【讨论】:
我正在从我的服务器端控制器运行代码。
module.exports = (function() {
return {
add: function(req, res) {
var tmp_path = req.files.pdf.path;
var target_path = './uploads/' + req.files.pdf.name;
fs.rename(tmp_path, target_path, function(err) {
if (err) throw err;
// delete the temporary file, so that the explicitly set temporary upload dir does not get filled with unwanted files
fs.unlink(tmp_path, function() {
if (err) throw err;
//edit here pdf parser
res.redirect('#/');
});
})
},
show: function(req, res) {
var pdfParser = new PDFParser();
var _onPDFBinDataReady = function (pdf) {
console.log('Loaded pdf:\n');
for (var i in pdf.data.Pages) {
var page = pdf.data.Pages[i];
// console.log(page.Texts);
for (var j in page.Texts) {
var text = page.Texts[j];
// console.log(text.R[0].T);
}
}
console.log(JSON.stringify(pdf));
};
// Create an error handling function
var _onPDFBinDataError = function (error) {
console.log(error);
};
pdfParser.on('pdfParser_dataReady', _.bind(_onPDFBinDataReady, this));
// Register error handling function
pdfParser.on('pdfParser_dataError', _.bind(_onPDFBinDataError, this));
// Construct the file path of the pdf
var pdfFilePath = './uploads/Invoice_template.pdf';
// Load the pdf. When it is loaded your data ready function will be called.
pdfParser.loadPDF(pdfFilePath);
},
//end controller
}
【讨论】: