From 7e39be9ca14ede8c9f1022fba174d6e28b86e403 Mon Sep 17 00:00:00 2001 From: Lucas Cimon <925560+Lucas-C@users.noreply.github.com> Date: Thu, 6 Aug 2026 12:11:30 +0200 Subject: [PATCH] New command extract-links --- .github/workflows/github-ci.yaml | 2 +- CHANGELOG.md | 1 + Makefile | 2 +- README.md | 1 + docs/index.rst | 1 + docs/user/subcommand-extract-links.md | 25 +++++++++++++ pdfly/cli.py | 22 ++++++++++++ pdfly/extract_links.py | 52 +++++++++++++++++++++++++++ pyproject.toml | 4 ++- sample-files | 2 +- tests/test_extract_links.py | 18 ++++++++++ 11 files changed, 126 insertions(+), 4 deletions(-) create mode 100644 docs/user/subcommand-extract-links.md create mode 100644 pdfly/extract_links.py create mode 100644 tests/test_extract_links.py diff --git a/.github/workflows/github-ci.yaml b/.github/workflows/github-ci.yaml index 97399572..31efe13e 100644 --- a/.github/workflows/github-ci.yaml +++ b/.github/workflows/github-ci.yaml @@ -65,7 +65,7 @@ jobs: - name: Lint with black run: black --check --extend-exclude sample-files . - name: Lint with mypy - run: mypy . --ignore-missing-imports --exclude build + run: mypy . --ignore-missing-imports --exclude build --exclude sample-files - name: Test with ruff run: ruff check pdfly/ - name: Spell Check Repo diff --git a/CHANGELOG.md b/CHANGELOG.md index 0feb20ba..954ad1fd 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,7 @@ - `2up` incorrectly handled documents with an odd number of pages - [issue #219](https://github.com/py-pdf/pdfly/issues/218) ### New Features (ENH) +- New `extract-links` command ([PR #271](https://github.com/py-pdf/pdfly/pull/271)) - `pagemeta` now displays the name of a known page format that is close to the page dimensions diff --git a/Makefile b/Makefile index 6ebe9270..9e11957a 100644 --- a/Makefile +++ b/Makefile @@ -19,7 +19,7 @@ clean: rm -rf tests/__pycache__ pdfly/__pycache__ Image9.png htmlcov docs/_build dist dont_commit_merged.pdf dont_commit_writer.pdf pdfly.egg-info lint: - mypy . --ignore-missing-imports --exclude build + mypy . --ignore-missing-imports --exclude build --exclude sample-files ruff check --fix --unsafe-fixes test: diff --git a/README.md b/README.md index 1cacaf0e..ad2ae084 100644 --- a/README.md +++ b/README.md @@ -47,6 +47,7 @@ $ pdfly --help │ compress Compress a PDF. │ │ extract-annotated-pages Extract only the annotated pages from a PDF. │ │ extract-images Extract images from PDF without resampling or altering. │ +│ extract-links Extract all links from a PDF document. │ │ extract-text Extract text from a PDF file. │ │ meta Show metadata of a PDF file │ │ pagemeta Give details about a single page. │ diff --git a/docs/index.rst b/docs/index.rst index a224ef66..2794af38 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -65,6 +65,7 @@ Usage │ compress Compress a PDF. │ │ extract-annotated-pages Extract only the annotated pages from a PDF. │ │ extract-images Extract images from PDF without resampling or altering. │ + │ extract-links Extract all links from a PDF document. │ │ extract-text Extract text from a PDF file. │ │ meta Show metadata of a PDF file │ │ pagemeta Give details about a single page. │ diff --git a/docs/user/subcommand-extract-links.md b/docs/user/subcommand-extract-links.md new file mode 100644 index 00000000..9b6e1f73 --- /dev/null +++ b/docs/user/subcommand-extract-links.md @@ -0,0 +1,25 @@ +# extract-links +Extract all links from a PDF document. + +## Usage +``` +$ pdfly extract-links --help + + Usage: pdfly extract-links [OPTIONS] PDF + + Extract all links from a PDF document. + +╭─ Arguments ───────────────────────────────────────────────────╮ +│ * pdf FILE [required] │ +╰───────────────────────────────────────────────────────────────╯ +╭─ Options ─────────────────────────────────────────────────────╮ +│ --format -f [json|text] Output format [default: text] │ +│ --help Show this message and exit. │ +╰───────────────────────────────────────────────────────────────╯ +``` + +## Examples +Extract all links from `doc.pdf` and pass them to `jq`: +``` +pdfly extract-links doc.pdf --format json | jq -r . +``` diff --git a/pdfly/cli.py b/pdfly/cli.py index 99f9280d..4b30170f 100644 --- a/pdfly/cli.py +++ b/pdfly/cli.py @@ -15,6 +15,7 @@ import pdfly.compress import pdfly.extract_annotated_pages import pdfly.extract_images +import pdfly.extract_links import pdfly.metadata import pdfly.pagemeta import pdfly.rm @@ -218,6 +219,27 @@ def extract_images( pdfly.extract_images.main(pdf) +@entry_point.command(name="extract-links", help=pdfly.extract_links.__doc__) # type: ignore[misc] +def extract_links( + pdf: Annotated[ + Path, + typer.Argument( + dir_okay=False, + exists=True, + resolve_path=True, + ), + ], + output_format: pdfly._utils.OutputOptions = typer.Option( # noqa + pdfly._utils.OutputOptions.text.value, + "--format", + "-f", + help="Output format", + show_default=True, + ), +) -> None: + pdfly.extract_links.main(pdf, output_format) + + @entry_point.command(name="extract-text") # type: ignore[misc] def extract_text( pdf: Annotated[ diff --git a/pdfly/extract_links.py b/pdfly/extract_links.py new file mode 100644 index 00000000..1c46b384 --- /dev/null +++ b/pdfly/extract_links.py @@ -0,0 +1,52 @@ +"""Extract all links from a PDF document.""" + +from pathlib import Path +from typing import TYPE_CHECKING + +from pydantic import BaseModel +from pypdf import PdfReader + +if TYPE_CHECKING: + from pypdf.generic import ArrayObject + +from ._utils import OutputOptions + + +class UriLink(BaseModel): + uri: str + page: int + rect: list[float] + + +def main(pdf: Path, output_format: OutputOptions) -> None: + reader = PdfReader(str(pdf)) + # Loop through pages and extract hyperlink metadata + links = [] + for page_number, page in enumerate(reader.pages, start=1): + if "/Annots" in page: + page_annots: ArrayObject = page["/Annots"] # type: ignore[assignment] + for annot in page_annots: + annotation = annot.get_object() + if "/A" in annotation and "/URI" in annotation["/A"]: + # Extract hyperlink URL + uri = annotation["/A"]["/URI"] + # Extract bounding rectangle + rect = annotation.get("/Rect", "N/A") + links.append( + UriLink( + uri=uri, + page=page_number, + rect=rect, + ) + ) + + if output_format == OutputOptions.json: + print("[") + for i, link in enumerate(links, start=1): + print("", link.json() + ("," if i < len(links) else "")) + print("]") + else: + for link in links: + print( + f"Page {link.page}: Hyperlink: {link.uri}, Rect: {link.rect}" + ) diff --git a/pyproject.toml b/pyproject.toml index d969bb2c..c7519b52 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -82,7 +82,7 @@ line-length = 120 select = ["ALL"] ignore = [ "D401", # First line of docstring should be in imperative mood - false positives - "UP031", # Use format specifiers instead of percent format + "CPY001", # Missing copyright notice at top of file "D205", # 1 blank line required between summary line and description "D400", # First line should end with a period "D415", # First line should end with a period @@ -110,6 +110,7 @@ ignore = [ "TRY", # I don't know what this is about # As long as we are not on Python 3.11+ "UP006", "UP007", + "UP031", # Use format specifiers instead of percent format # for the moment, fix it later: "T201", # print "DTZ006", # datetime without timezone @@ -130,6 +131,7 @@ ignore = [ "PLR0912", # Too many branches "PLR0913", # Too many arguments to function call "PLR0915", # Too many statements + "PLR0917", # Too many positional arguments "PLR2004", # Magic value "PLW", # global variables "PTH110", # `os.path.exists()` should be replaced by `Path.exists()` diff --git a/sample-files b/sample-files index 65e82ed3..89039b60 160000 --- a/sample-files +++ b/sample-files @@ -1 +1 @@ -Subproject commit 65e82ed36c1efd9bd7172a35c8dcfd6e18aabfb2 +Subproject commit 89039b6078fd0c9f98bf3d6fcb5583fac6b0ecaf diff --git a/tests/test_extract_links.py b/tests/test_extract_links.py new file mode 100644 index 00000000..5995addf --- /dev/null +++ b/tests/test_extract_links.py @@ -0,0 +1,18 @@ +from pathlib import Path + +import pytest + +from .conftest import RESOURCES_ROOT, chdir, run_cli + + +def test_extract_links(capsys: pytest.CaptureFixture, tmp_path: Path) -> None: + with chdir(tmp_path): + run_cli( + [ + "extract-links", + str(RESOURCES_ROOT / "GeoBase_NHNC1_Data_Model_UML_EN.pdf"), + ] + ) + captured = capsys.readouterr() + assert not captured.err + assert "mailto:geoginfo@RNCan.gc.ca" in captured.out