Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .github/workflows/github-ci.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -65,7 +65,7 @@ jobs:
- name: Lint with black
run: black --check --extend-exclude sample-files .
- name: Lint with mypy
run: mypy . --ignore-missing-imports --exclude build
run: mypy . --ignore-missing-imports --exclude build --exclude sample-files
- name: Test with ruff
run: ruff check pdfly/
- name: Spell Check Repo
Expand Down
1 change: 1 addition & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,7 @@
- `2up` incorrectly handled documents with an odd number of pages - [issue #219](https://github.com/py-pdf/pdfly/issues/218)

### New Features (ENH)
- New `extract-links` command ([PR #271](https://github.com/py-pdf/pdfly/pull/271))
- `pagemeta` now displays the name of a known page format that is close to the page dimensions


Expand Down
2 changes: 1 addition & 1 deletion Makefile
Original file line number Diff line number Diff line change
Expand Up @@ -19,7 +19,7 @@ clean:
rm -rf tests/__pycache__ pdfly/__pycache__ Image9.png htmlcov docs/_build dist dont_commit_merged.pdf dont_commit_writer.pdf pdfly.egg-info

lint:
mypy . --ignore-missing-imports --exclude build
mypy . --ignore-missing-imports --exclude build --exclude sample-files
ruff check --fix --unsafe-fixes

test:
Expand Down
1 change: 1 addition & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -47,6 +47,7 @@ $ pdfly --help
│ compress Compress a PDF. │
│ extract-annotated-pages Extract only the annotated pages from a PDF. │
│ extract-images Extract images from PDF without resampling or altering. │
│ extract-links Extract all links from a PDF document. │
│ extract-text Extract text from a PDF file. │
│ meta Show metadata of a PDF file │
│ pagemeta Give details about a single page. │
Expand Down
1 change: 1 addition & 0 deletions docs/index.rst
Original file line number Diff line number Diff line change
Expand Up @@ -65,6 +65,7 @@ Usage
│ compress Compress a PDF. │
│ extract-annotated-pages Extract only the annotated pages from a PDF. │
│ extract-images Extract images from PDF without resampling or altering. │
│ extract-links Extract all links from a PDF document. │
│ extract-text Extract text from a PDF file. │
│ meta Show metadata of a PDF file │
│ pagemeta Give details about a single page. │
Expand Down
25 changes: 25 additions & 0 deletions docs/user/subcommand-extract-links.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,25 @@
# extract-links
Extract all links from a PDF document.

## Usage
```
$ pdfly extract-links --help

Usage: pdfly extract-links [OPTIONS] PDF

Extract all links from a PDF document.

╭─ Arguments ───────────────────────────────────────────────────╮
│ * pdf FILE [required] │
╰───────────────────────────────────────────────────────────────╯
╭─ Options ─────────────────────────────────────────────────────╮
│ --format -f [json|text] Output format [default: text] │
│ --help Show this message and exit. │
╰───────────────────────────────────────────────────────────────╯
```

## Examples
Extract all links from `doc.pdf` and pass them to `jq`:
```
pdfly extract-links doc.pdf --format json | jq -r .
```
22 changes: 22 additions & 0 deletions pdfly/cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -15,6 +15,7 @@
import pdfly.compress
import pdfly.extract_annotated_pages
import pdfly.extract_images
import pdfly.extract_links
import pdfly.metadata
import pdfly.pagemeta
import pdfly.rm
Expand Down Expand Up @@ -218,6 +219,27 @@ def extract_images(
pdfly.extract_images.main(pdf)


@entry_point.command(name="extract-links", help=pdfly.extract_links.__doc__) # type: ignore[misc]
def extract_links(
pdf: Annotated[
Path,
typer.Argument(
dir_okay=False,
exists=True,
resolve_path=True,
),
],
output_format: pdfly._utils.OutputOptions = typer.Option( # noqa
pdfly._utils.OutputOptions.text.value,
"--format",
"-f",
help="Output format",
show_default=True,
),
) -> None:
pdfly.extract_links.main(pdf, output_format)


@entry_point.command(name="extract-text") # type: ignore[misc]
def extract_text(
pdf: Annotated[
Expand Down
52 changes: 52 additions & 0 deletions pdfly/extract_links.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,52 @@
"""Extract all links from a PDF document."""

from pathlib import Path
from typing import TYPE_CHECKING

from pydantic import BaseModel
from pypdf import PdfReader

if TYPE_CHECKING:
from pypdf.generic import ArrayObject

from ._utils import OutputOptions


class UriLink(BaseModel):
uri: str
page: int
rect: list[float]


def main(pdf: Path, output_format: OutputOptions) -> None:
reader = PdfReader(str(pdf))
# Loop through pages and extract hyperlink metadata
links = []
for page_number, page in enumerate(reader.pages, start=1):
if "/Annots" in page:
page_annots: ArrayObject = page["/Annots"] # type: ignore[assignment]
for annot in page_annots:
annotation = annot.get_object()
if "/A" in annotation and "/URI" in annotation["/A"]:
# Extract hyperlink URL
uri = annotation["/A"]["/URI"]
# Extract bounding rectangle
rect = annotation.get("/Rect", "N/A")
links.append(
UriLink(
uri=uri,
page=page_number,
rect=rect,
)
)

if output_format == OutputOptions.json:
print("[")
for i, link in enumerate(links, start=1):
print("", link.json() + ("," if i < len(links) else ""))
print("]")
else:
for link in links:
print(
f"Page {link.page}: Hyperlink: {link.uri}, Rect: {link.rect}"
)
4 changes: 3 additions & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -82,7 +82,7 @@ line-length = 120
select = ["ALL"]
ignore = [
"D401", # First line of docstring should be in imperative mood - false positives
"UP031", # Use format specifiers instead of percent format
"CPY001", # Missing copyright notice at top of file
"D205", # 1 blank line required between summary line and description
"D400", # First line should end with a period
"D415", # First line should end with a period
Expand Down Expand Up @@ -110,6 +110,7 @@ ignore = [
"TRY", # I don't know what this is about
# As long as we are not on Python 3.11+
"UP006", "UP007",
"UP031", # Use format specifiers instead of percent format
# for the moment, fix it later:
"T201", # print
"DTZ006", # datetime without timezone
Expand All @@ -130,6 +131,7 @@ ignore = [
"PLR0912", # Too many branches
"PLR0913", # Too many arguments to function call
"PLR0915", # Too many statements
"PLR0917", # Too many positional arguments
"PLR2004", # Magic value
"PLW", # global variables
"PTH110", # `os.path.exists()` should be replaced by `Path.exists()`
Expand Down
18 changes: 18 additions & 0 deletions tests/test_extract_links.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,18 @@
from pathlib import Path

import pytest

from .conftest import RESOURCES_ROOT, chdir, run_cli


def test_extract_links(capsys: pytest.CaptureFixture, tmp_path: Path) -> None:
with chdir(tmp_path):
run_cli(
[
"extract-links",
str(RESOURCES_ROOT / "GeoBase_NHNC1_Data_Model_UML_EN.pdf"),
]
)
captured = capsys.readouterr()
assert not captured.err
assert "mailto:geoginfo@RNCan.gc.ca" in captured.out
Loading