Skip to content

Page API 参考

自 v0.3.34 起,所有绑定都提供 Page 对象,让你无需在每次提取调用中传递 page_index,而是直接迭代文档并在每页上调用提取方法。该类型在 Python、Node.js、C# 和 Go 中统一命名为 Page;Rust 则通过 PdfPage 提供相同的结构。

快速示例

Python

from pdf_oxide import PdfDocument

with PdfDocument("paper.pdf") as doc:
    for page in doc:                       # len(doc), doc[i], doc[-1] also work
        print(page.text[:80])
        md = page.markdown(detect_headings=True)

Rust

use pdf_oxide::api::Pdf;

let mut doc = Pdf::open("paper.pdf")?;
for i in 0..doc.page_count()? {
    let page = doc.page(i)?;
    println!("{}", &page.text()?[..80]);
}

JavaScript / TypeScript (Node)

const { PdfDocument } = require("pdf-oxide");

const doc = new PdfDocument("paper.pdf");
for (const page of doc) {
  console.log(page.extractText().slice(0, 80));
}
doc.close();

Go

package main

import (
    "fmt"
    "log"
    pdfoxide "github.com/yfedoseev/pdf_oxide/go"
)

func main() {
    doc, err := pdfoxide.Open("paper.pdf")
    if err != nil { log.Fatal(err) }
    defer doc.Close()

    pages, _ := doc.Pages()
    for _, page := range pages {
        text, _ := page.ExtractText()
        fmt.Println(text[:80])
    }
}

C#

using PdfOxide;

using var doc = PdfDocument.Open("paper.pdf");
foreach (var page in doc.Pages)
{
    Console.WriteLine(page.ExtractText()[..Math.Min(80, page.ExtractText().Length)]);
}

Java

import fyi.oxide.pdf.PdfDocument;
import java.nio.file.Path;

try (PdfDocument doc = PdfDocument.open(Path.of("paper.pdf"))) {
    for (int i = 0; i < doc.pageCount(); i++) {
        String text = doc.extractText(i);
        System.out.println(text.substring(0, Math.min(80, text.length())));
        String md = doc.toMarkdown(i);
    }
}

Kotlin

import fyi.oxide.pdf.PdfDocument

PdfDocument.open(java.nio.file.Path.of("paper.pdf")).use { doc ->
    for (i in 0 until doc.pageCount()) {
        val text = doc.extractText(i)
        println(text.substring(0, minOf(80, text.length)))
        val md = doc.toMarkdown(i)
    }
}

Scala

import fyi.oxide.pdf.PdfDocument
import scala.util.Using

Using.resource(PdfDocument.open("paper.pdf")) { doc =>
  for (i <- 0 until doc.pageCount()) {
    val text = doc.extractText(i)
    println(text.substring(0, math.min(80, text.length)))
    val md = doc.toMarkdown(i)
  }
}

Clojure

(require '[pdf-oxide.core :as pdf])

(with-open [doc (pdf/open "paper.pdf")]
  (doseq [i (range (pdf/page-count doc))]
    (let [text (pdf/extract-text doc i)]
      (println (subs text 0 (min 80 (count text))))
      (pdf/to-markdown doc i))))

Ruby

require 'pdf_oxide'

PdfOxide::PdfDocument.open('paper.pdf') do |doc|
  (0...doc.page_count).each do |i|
    text = doc.extract_text(i)
    puts text[0, 80]
    md = doc.to_markdown(i)
  end
end

PHP

use PdfOxide\PdfDocument;

$doc = PdfDocument::open('paper.pdf');
for ($i = 0; $i < $doc->pageCount(); $i++) {
    $text = $doc->extractText($i);
    echo substr($text, 0, 80), "\n";
    $md = $doc->toMarkdown($i);
}
$doc->close();

C++

#include <pdf_oxide/pdf_oxide.hpp>

auto doc = pdf_oxide::Document::open("paper.pdf");
for (int i = 0; i < doc.page_count(); i++) {
    auto text = doc.extract_text(i);
    std::cout << text.substr(0, 80) << "\n";
    auto md = doc.to_markdown(i);
}

Swift

import PdfOxide

let doc = try Document.open("paper.pdf")
for i in 0..<(try doc.pageCount()) {
    let text = try doc.extractText(i)
    print(text.prefix(80))
    let md = try doc.toMarkdown(i)
}

Dart

import 'package:pdf_oxide/pdf_oxide.dart';

final doc = PdfDocument.open('paper.pdf');
for (var i = 0; i < doc.pageCount; i++) {
  final text = doc.extractText(i);
  print(text.substring(0, text.length < 80 ? text.length : 80));
  final md = doc.toMarkdown(i);
}
doc.close();

R

library(pdfoxide)

doc <- pdf_open("paper.pdf")
for (i in 0:(pdf_page_count(doc) - 1)) {
  text <- pdf_extract_text(doc, i)
  cat(substr(text, 1, 80), "\n")
  md <- pdf_to_markdown(doc, i)
}

Julia

using PdfOxide

doc = open_document("paper.pdf")
for i in 0:(page_count(doc) - 1)
    text = extract_text(doc, i)
    println(first(text, 80))
    md = to_markdown(doc, i)
end

Zig

const pdf_oxide = @import("pdf_oxide");
const a = std.heap.page_allocator;

var doc = try pdf_oxide.Document.open("paper.pdf");
var i: usize = 0;
while (i < try doc.pageCount()) : (i += 1) {
    const text = try doc.extractText(a, i);
    std.debug.print("{s}\n", .{text[0..@min(80, text.len)]});
    const md = try doc.toMarkdown(a, i);
}

Objective-C

#import "POXPdfOxide.h"
NSError *err = nil;

POXDocument *doc = [POXDocument openPath:@"paper.pdf" error:&err];
for (NSInteger i = 0; i < [doc pageCountError:&err]; i++) {
    NSString *text = [doc extractText:i error:&err];
    NSLog(@"%@", [text substringToIndex:MIN(80, text.length)]);
    NSString *md = [doc toMarkdown:i error:&err];
}

Elixir

{:ok, doc} = PdfOxide.open("paper.pdf")
{:ok, n} = PdfOxide.page_count(doc)
for i <- 0..(n - 1) do
  {:ok, text} = PdfOxide.extract_text(doc, i)
  IO.puts(String.slice(text, 0, 80))
  {:ok, md} = PdfOxide.to_markdown(doc, i)
end

Python — Page

惰性属性模式 — 内容在首次访问时解析,并缓存在 Page 上。

成员 返回类型 说明
page.text str 提取的文本(感知列布局)
page.chars list[Char] 字符级记录,含边界框和字体信息
page.words list[Word] 单词级记录,含边界框
page.lines list[TextLine] 文本行,含边界框
page.spans list[Span] 带样式的文本段(字体、大小、粗细)
page.tables list[Table] 结构化表格行及单元格边界框
page.images list[Image] 图像元数据
page.paths list[Path] 矢量路径记录
page.annotations list[Annotation] 本页注释
page.markdown(detect_headings=True) str 转换为 Markdown
page.plain_text() str 纯文本(无布局提示)
page.html() str 转换为 HTML
page.render(format="png") bytes 将页面渲染为 PNG / JPEG
page.search(term, case_sensitive=False) list[SearchResult] 在本页中搜索文本
page.region(rect) PageRegion 在指定矩形区域内提取
with PdfDocument("paper.pdf") as doc:
    page = doc[0]                 # or doc.page(0)
    for word in page.words:       # first access parses; subsequent calls cached
        print(word.text, word.bbox)

    # Scoped extraction
    header = page.region((0, 700, 612, 92)).extract_text()

原有的写入编辑器类 PdfPage 保持不变;新增的 Page 严格只读。

Rust — PdfPage

use pdf_oxide::api::Pdf;

let mut doc = Pdf::open("paper.pdf")?;
let page = doc.page(0)?;

let text = page.text()?;
let words = page.extract_words()?;
let tables = page.extract_tables()?;
let md = page.to_markdown(true)?;

PdfPage 上的可用方法:

  • text(), plain_text(), to_markdown(detect_headings), to_html()
  • extract_chars(), extract_words(), extract_lines(), extract_spans()
  • extract_tables(), extract_paths(), extract_images()
  • annotations(), render(format)
  • search(term) — 局部搜索
  • find_text_containing(substring) — 带 ID 的 DOM 级命中列表

Node.js — Page

const { PdfDocument } = require("pdf-oxide");

const doc = new PdfDocument("paper.pdf");
const page = doc.page(0);

console.log(page.width, page.height, page.rotation);  // cached
console.log(page.extractText());
const words = page.extractWords();
const tables = page.extractTables();
const md = page.toMarkdown();

PdfDocument 通过 Symbol.iterator 支持 for..of,同时提供 doc.page(i)doc.pageCount()

以下 6 个此前仅限原生层的方法,现可通过 TS 层在 PagePdfDocument 上使用:

  • extractWords
  • extractTextLines
  • extractTables
  • extractPaths
  • getEmbeddedImages
  • ocrExtractText

每个方法都有对应的异步版本 — extractTextAsynctoMarkdownAsync 等。

Go — Page

doc, _ := pdfoxide.Open("paper.pdf")
defer doc.Close()

page, _ := doc.Page(0)
text, _ := page.ExtractText()
md, _   := page.ToMarkdown()
tables, _ := page.ExtractTables()

// Iterate every page
all, _ := doc.Pages()
for i, p := range all {
    t, _ := p.ExtractText()
    fmt.Printf("page %d: %d chars\n", i, len(t))
}

Go 的 Page 结构体提供完整的方法集:ExtractTextToMarkdownToHtmlToPlainTextExtractWordsExtractTextLinesExtractTablesExtractCharsExtractPathsAnnotationsImagesFontsRenderPageSearch

C# — Page

using PdfOxide;

using var doc = PdfDocument.Open("paper.pdf");

Page page = doc[0];                            // or doc.Pages[0] or doc.Page(0)
string text = page.ExtractText();
string md   = page.ToMarkdown();
Table[] tables = page.ExtractTables();

// Async variants
string textAsync = await page.ExtractTextAsync();
string mdAsync   = await page.ToMarkdownAsync();

doc.PagesIReadOnlyList<Page>。每个同步方法都有支持 CancellationTokenasync Task<T> 对应版本。

表格的结构

extract_tables()(在 PdfDocumentPage 上均可使用)在各语言中返回一致的 Table 类型:

语言 类型 单元格访问方式
Rust Table 迭代 rows[i].cells[j]
Python dict row["cells"][i]["text"]
Go Table table.CellText(row, col)
C# Table table.CellText(row, col)
Node.js Table 接口 table.cells[row][col]

每个单元格包含文本和边界框,便于将提取结果与页面坐标对应起来。

doc.extract_*(page_index) 迁移

旧写法(仍然支持):

doc = PdfDocument("paper.pdf")
for i in range(doc.page_count()):
    print(doc.extract_text(i))
    print(doc.to_markdown(i, detect_headings=True))
    print(doc.extract_tables(i))

新写法(v0.3.34+):

with PdfDocument("paper.pdf") as doc:
    for page in doc:
        print(page.text)
        print(page.markdown(detect_headings=True))
        print(page.tables)

两种写法均持续支持;Page 风格在逐页处理管道中可读性更好,也省去了反复维护页面索引的麻烦。

相关页面