Java 读取PDF内容

202 阅读1分钟

maven依赖

<dependency>
    <groupId>org.apache.pdfbox</groupId>
    <artifactId>pdfbox</artifactId>
    <version>2.0.24</version>
</dependency>

获取pdf内容

```
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.text.PDFTextStripper;


public static void main(String[] args) throws Exception {
    PDDocument pdDocument = PDDocument.load(new File("文件路径"));
    int numberOfPages = pdDocument.getNumberOfPages();
    PDFTextStripper stripper = new PDFTextStripper();
    stripper.setStartPage(1);
    stripper.setEndPage(numberOfPages);
    System.out.println(stripper.getText(pdDocument));
}

```