PDF 合并与拆分 — Python / Rust / Go
绑定覆盖情况。 PDF 合并在 Python、Rust 和 Go 中均可使用。基于
extract_pages的拆分目前仅在 Python 和 Rust 中提供。C# 和 WASM 绑定尚未公开这些编辑器操作——可改用 Rust CLI(pdf-oxide merge、pdf-oxide split)作为替代,或通过受支持的绑定调用。
将两个 PDF 合并为一个:
Python
from pdf_oxide import PdfDocument
doc = PdfDocument("main.pdf")
doc.merge_from("appendix.pdf")
doc.save("combined.pdf")
WASM
import { WasmPdfDocument } from "pdf-oxide-wasm";
// 将两个 PDF 以 Uint8Array 加载
const mainDoc = new WasmPdfDocument(mainBytes);
const appendixDoc = new WasmPdfDocument(appendixBytes);
// 从两个文档抽取文本并按需处理
const allText = mainDoc.extractAllText() + "\n" + appendixDoc.extractAllText();
mainDoc.free();
appendixDoc.free();
Rust
use pdf_oxide::editor::DocumentEditor;
let mut editor = DocumentEditor::open("main.pdf")?;
editor.merge_from("appendix.pdf")?;
editor.save("combined.pdf")?;
Go
package main
import (
"log"
pdfoxide "github.com/yfedoseev/pdf_oxide/go"
)
func main() {
editor, err := pdfoxide.OpenEditor("main.pdf")
if err != nil { log.Fatal(err) }
defer editor.Close()
if _, err := editor.MergeFrom("appendix.pdf"); err != nil { log.Fatal(err) }
if err := editor.Save("combined.pdf"); err != nil { log.Fatal(err) }
}
PDF Oxide 在 PDF 对象层面合并页面——字体、图像和注释可在文档之间正确保留。
安装
pip install pdf_oxide
合并 PDF
追加全部页面
将第二个 PDF 的每一页追加到第一个 PDF:
Python
from pdf_oxide import PdfDocument
doc = PdfDocument("report.pdf")
doc.merge_from("charts.pdf")
doc.save("full-report.pdf")
WASM
// WASM API:加载并处理多个文档
const report = new WasmPdfDocument(reportBytes);
const charts = new WasmPdfDocument(chartsBytes);
// 将两个文档一起处理
const fullText = report.extractAllText() + "\n" + charts.extractAllText();
report.free();
charts.free();
Rust
let mut editor = DocumentEditor::open("report.pdf")?;
let pages_added = editor.merge_from("charts.pdf")?;
println!("新增 {} 页", pages_added);
editor.save("full-report.pdf")?;
Go
editor, _ := pdfoxide.OpenEditor("report.pdf")
defer editor.Close()
added, _ := editor.MergeFrom("charts.pdf")
fmt.Printf("新增 %d 页\n", added)
_ = editor.Save("full-report.pdf")
合并多个文件
用静态方法 Pdf.merge() 一次性合并多个 PDF:
Python
from pdf_oxide import Pdf
pdf = Pdf.merge(["intro.pdf", "chapter1.pdf", "chapter2.pdf", "appendix.pdf"])
pdf.save("book.pdf")
也可以在已有文档上链式调用 merge_from():
from pdf_oxide import PdfDocument
doc = PdfDocument("intro.pdf")
for f in ["chapter1.pdf", "chapter2.pdf", "appendix.pdf"]:
doc.merge_from(f)
doc.save("book.pdf")
WASM
// 顺序加载并处理多个 PDF
const files = [introBytes, ch1Bytes, ch2Bytes, appendixBytes];
const allText = [];
for (const bytes of files) {
const doc = new WasmPdfDocument(bytes);
allText.push(doc.extractAllText());
doc.free();
}
console.log(allText.join("\n"));
Rust
let files = ["intro.pdf", "chapter1.pdf", "chapter2.pdf", "appendix.pdf"];
let mut editor = DocumentEditor::open(files[0])?;
for f in &files[1..] {
editor.merge_from(f)?;
}
editor.save("book.pdf")?;
Go
// 顶层 Merge 通过一次调用返回合并后的 PDF 字节
bytes, err := pdfoxide.Merge([]string{
"intro.pdf", "chapter1.pdf", "chapter2.pdf", "appendix.pdf",
})
if err != nil { log.Fatal(err) }
_ = os.WriteFile("book.pdf", bytes, 0644)
合并指定页面
选择要从源文档合并的页面:
Python
from pdf_oxide import PdfDocument
doc = PdfDocument("main.pdf")
# 只从源文件合并第 0、2、4 页
doc.merge_pages_from("source.pdf", [0, 2, 4])
doc.save("selected.pdf")
Rust
let mut editor = DocumentEditor::open("main.pdf")?;
editor.merge_pages_from("source.pdf", &[0, 2, 4])?;
editor.save("selected.pdf")?;
拆分 PDF
将页面抽取到新文件
从大型文档中抽出特定页面:
Python
from pdf_oxide import PdfDocument
doc = PdfDocument("book.pdf")
doc.extract_pages([0, 1, 2, 3, 4], "chapter1.pdf")
WASM
// 从指定页面抽取文本
const doc = new WasmPdfDocument(bytes);
const pages = [0, 1, 2, 3, 4];
for (const i of pages) {
const text = doc.extractText(i);
console.log(`第 ${i + 1} 页: ${text.slice(0, 80)}...`);
}
doc.free();
Rust
let mut editor = DocumentEditor::open("book.pdf")?;
editor.extract_pages(&[0, 1, 2, 3, 4], "chapter1.pdf")?;
按单页拆分
将每一页另存为单独文件:
Python
from pdf_oxide import PdfDocument
doc = PdfDocument("document.pdf")
for i in range(doc.page_count()):
doc.extract_pages([i], f"page_{i + 1}.pdf")
Rust
let mut editor = DocumentEditor::open("document.pdf")?;
let page_count = editor.page_count()?;
for i in 0..page_count {
editor.extract_pages(&[i], &format!("page_{}.pdf", i + 1))?;
}
按块拆分
将大 PDF 切分为每份 N 页的小文件:
Python
from pdf_oxide import PdfDocument
doc = PdfDocument("large.pdf")
chunk_size = 10
for start in range(0, doc.page_count(), chunk_size):
end = min(start + chunk_size, doc.page_count())
pages = list(range(start, end))
doc.extract_pages(pages, f"chunk_{start // chunk_size + 1}.pdf")
Rust
let mut editor = DocumentEditor::open("large.pdf")?;
let page_count = editor.page_count()?;
let chunk_size = 10;
for start in (0..page_count).step_by(chunk_size) {
let end = (start + chunk_size).min(page_count);
let pages: Vec<usize> = (start..end).collect();
editor.extract_pages(&pages, &format!("chunk_{}.pdf", start / chunk_size + 1))?;
}