Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
34 changes: 12 additions & 22 deletions Makefile
Original file line number Diff line number Diff line change
@@ -1,4 +1,4 @@
.PHONY: all pre-commit ty unit-test unit-test-junit unit-test-cov-html unit-test-cov-xml diff-cover unit-test-diff-cover
.PHONY: all pre-commit ty docs-api docs-build docs-build-pdf docs-build-all unit-test unit-test-junit unit-test-cov-html unit-test-cov-xml diff-cover unit-test-diff-cover

CMD:=uv run -m
PYMODULE:=pyrit
Expand All @@ -19,29 +19,19 @@ pre-commit:
ty:
$(CMD) ty check $(PYMODULE) $(UNIT_TESTS)

# Build the full documentation site:
# 1. Generate API reference JSON from Python source (griffe)
# 2. Convert API JSON to MyST markdown pages
# 3. Build the Jupyter Book site (HTML only — fast, no LaTeX needed)
# 4. Generate RSS feed
docs-build:
uv run python -m build_scripts.pydoc2json pyrit --submodules -o doc/_api/pyrit_all.json
uv run python -m build_scripts.gen_api_md
# --strict validates URLs and cross-refs; skips are configured in doc/myst.yml under error_rules
cd doc && uv run jupyter-book build --all --html --strict
# Build strict HTML and the RSS feed after generating the API reference.
# --all would also select the configured PDF export and require LaTeX.
docs-build: docs-api
cd doc && uv run jupyter-book build --html --strict
uv run python -m build_scripts.generate_rss

# Build the full documentation site including the PDF export.
# Mirrors the ReadTheDocs build (.readthedocs.yaml) so CI catches PDF-only issues
# such as missing images that the HTML-only build silently ignores.
# Requires xelatex / latexmk on PATH (texlive-xetex + texlive-fonts-recommended +
# texlive-plain-generic + latexmk on Ubuntu).
docs-build-all:
uv run python -m build_scripts.pydoc2json pyrit --submodules -o doc/_api/pyrit_all.json
uv run python -m build_scripts.gen_api_md
# --strict validates URLs and cross-refs; skips are configured in doc/myst.yml under error_rules
cd doc && uv run jupyter-book build --all --html --pdf --strict
uv run python -m build_scripts.generate_rss
# PDF is a separate, checked export requiring latexmk and xelatex.
docs-build-pdf: docs-api
uv run python -m build_scripts.build_docs_pdf

# Build HTML first, then check the PDF export without repeating API generation.
docs-build-all: docs-build
uv run python -m build_scripts.build_docs_pdf

# Regenerate only the API reference pages (without building the full site)
docs-api:
Expand Down
106 changes: 106 additions & 0 deletions build_scripts/build_docs_pdf.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,106 @@
# Copyright (c) Microsoft Corporation.
# Licensed under the MIT license.

"""Build the configured documentation PDF and verify the export, not just the exit code."""

from __future__ import annotations

import re
import shutil
import subprocess
import sys
from pathlib import Path

import yaml
from pypdf import PdfReader
from pypdf.errors import PdfReadError


def _pdf_output(doc_root: Path) -> Path:
config = yaml.safe_load((doc_root / "myst.yml").read_text(encoding="utf-8"))
project = config.get("project") if isinstance(config, dict) else None
exports = project.get("exports") if isinstance(project, dict) else None
if not isinstance(exports, list):
raise ValueError("myst.yml must declare a project PDF export.")
pdf_exports = [export for export in exports if isinstance(export, dict) and export.get("format") == "pdf"]
if len(pdf_exports) != 1:
raise ValueError("Expected exactly one project PDF export in myst.yml.")
export = pdf_exports[0]
if export.get("template") != "plain_latex_book":
raise ValueError("The checked PDF build supports plain_latex_book with xelatex.")
output = export.get("output")
if not isinstance(output, str) or not output or Path(output).suffix != ".pdf":
raise ValueError("The PDF export must declare an output ending in .pdf.")
output_path = (doc_root / output).resolve()
if not output_path.is_relative_to(doc_root.resolve()):
raise ValueError("The PDF output must stay inside the documentation directory.")
return output_path


def _signature(path: Path) -> tuple[int, int, int] | None:
if not path.exists():
return None
stat = path.stat()
return stat.st_mtime_ns, stat.st_ctime_ns, stat.st_size


def _require_fresh_file(*, path: Path, previous: tuple[int, int, int] | None) -> None:
if not path.is_file() or path.stat().st_size == 0:
raise ValueError(f"PDF export did not produce a nonempty file: {path}")
if _signature(path) == previous:
raise ValueError(f"PDF export left a stale file unchanged: {path}")


def _check_engine_log(path: Path) -> None:
fatal = re.compile(
r"^!|(?:LaTeX|Package \S+|Class \S+) Error:|Emergency stop|Fatal error occurred"
r"|^Latexmk: (?:Errors|Failure)|^Collected error summary"
)
for line_number, line in enumerate(path.read_text(encoding="utf-8", errors="replace").splitlines(), start=1):
if fatal.search(line):
raise ValueError(f"Fatal LaTeX diagnostic in {path}:{line_number}: {line}")


def build_pdf(doc_root: Path) -> int:
"""Build the repository's XeLaTeX PDF export and reject missing, stale, or failed output."""
output = _pdf_output(doc_root)
missing = [tool for tool in ("latexmk", "xelatex") if shutil.which(tool) is None]
if missing:
raise ValueError(
f"Missing PDF prerequisites on PATH: {', '.join(missing)}. "
"Provision LaTeX separately; use the HTML-only build if a PDF is not needed."
)
logs_dir = output.parent / f"{output.stem}_pdf_logs"
logs = [logs_dir / f"{output.stem}.log", logs_dir / f"{output.stem}.shell.log"]
previous = {path: _signature(path) for path in [output, *logs]}
result = subprocess.run(
[sys.executable, "-m", "jupyter_book", "build", "--site", "--pdf", "--strict", "--logs"],
cwd=doc_root,
check=False,
)
if result.returncode != 0:
print(f"ERROR: Jupyter Book PDF build exited with code {result.returncode}.", file=sys.stderr)
return result.returncode
for path in [output, *logs]:
_require_fresh_file(path=path, previous=previous[path])
for log in logs:
_check_engine_log(log)
reader = PdfReader(output, strict=True)
if not reader.pages:
raise ValueError(f"PDF export contains no pages: {output}")
print(f"[OK] Verified fresh PDF export and native LaTeX logs: {output}")
return 0


def main() -> int:
"""Run the checked export from this checkout's documentation directory."""
doc_root = Path(__file__).resolve().parents[1] / "doc"
try:
return build_pdf(doc_root)
except (OSError, ValueError, yaml.YAMLError, PdfReadError) as error:
print(f"ERROR: {error}", file=sys.stderr)
return 1


if __name__ == "__main__":
sys.exit(main())
4 changes: 2 additions & 2 deletions doc/code/executor/8_modality_feedback.ipynb
Original file line number Diff line number Diff line change
Expand Up @@ -172,8 +172,8 @@
"- `roakey.png` is loaded from the docs root.\n",
"- A modern color photo of a three-masted ship is loaded from a checked-in asset.\n",
"- Ship photo source: [Gorch Fock unter Segeln Kieler Foerde 2006](\n",
" https://en.wikipedia.org/wiki/German_training_ship_Gorch_Fock_%281958%29#/media/File:Gorch_Fock_unter_Segeln_Kieler_Foerde_2006.jpg\n",
" ) (Wikimedia Commons), licensed under CC BY-SA 2.5."
" https://commons.wikimedia.org/wiki/File:Gorch_Fock_unter_Segeln_Kieler_Foerde_2006.jpg\n",
" ) (Wikimedia Commons). Photo by Felix Koenig, cropped by Ibn Battuta, licensed under CC BY-SA 2.5."
]
},
{
Expand Down
4 changes: 2 additions & 2 deletions doc/code/executor/8_modality_feedback.py
Original file line number Diff line number Diff line change
Expand Up @@ -113,8 +113,8 @@
# - `roakey.png` is loaded from the docs root.
# - A modern color photo of a three-masted ship is loaded from a checked-in asset.
# - Ship photo source: [Gorch Fock unter Segeln Kieler Foerde 2006](
# https://en.wikipedia.org/wiki/German_training_ship_Gorch_Fock_%281958%29#/media/File:Gorch_Fock_unter_Segeln_Kieler_Foerde_2006.jpg
# ) (Wikimedia Commons), licensed under CC BY-SA 2.5.
# https://commons.wikimedia.org/wiki/File:Gorch_Fock_unter_Segeln_Kieler_Foerde_2006.jpg
# ) (Wikimedia Commons). Photo by Felix Koenig, cropped by Ibn Battuta, licensed under CC BY-SA 2.5.

# %%
roakey_seed_path = (Path(".") / ".." / ".." / "roakey.png").resolve()
Expand Down
1 change: 0 additions & 1 deletion doc/code/framework.md
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,6 @@
Learn how to use PyRIT's components to build red teaming workflows.

:::::{grid} 1 1 2 3
:gutter: 3

::::{card} 📦 Datasets
:link: ./datasets/0_dataset
Expand Down
69 changes: 69 additions & 0 deletions doc/contributing/7_notebooks.md
Original file line number Diff line number Diff line change
Expand Up @@ -18,3 +18,72 @@ Here are contributor guidelines:
- Before a release, re-generate all notebooks by using [pct_to_ipynb.py](../generate_docs/pct_to_ipynb.py). Because this executes against real systems, it can detect many issues.
- Please do not re-commit updated generated `.ipynb` files with slight changes if nothing has changed in the source
- We use [Jupyter-Book](https://jupyterbook.org/) with [Markedly Structured Text (MyST)](https://mystmd.org/).

## Building the documentation

Rendering the committed notebooks does not execute their code or call their AI
providers. Notebook execution is a separate step that requires configured
credentials. Keep the committed notebook outputs and JupyText metadata when
editing documentation.

Use this checkout's uv environment and locked development dependencies:

```powershell
uv sync --frozen --group dev --python 3.13
```

Generate the API reference before building the book. The TOC includes generated
pages, so a documentation-structure check alone is not a substitute for this
preparation:

```powershell
uv run --frozen --no-sync python -m build_scripts.pydoc2json pyrit --submodules -o doc\_api\pyrit_all.json
uv run --frozen --no-sync python -m build_scripts.gen_api_md
uv run --frozen --no-sync python -m build_scripts.validate_docs
```

For HTML only, run from `doc`:

```powershell
Set-Location doc
uv run --frozen --no-sync jupyter-book build --html --strict
```

Do not add `--all` to an HTML-only build. It selects the PDF export declared in
`myst.yml` too, even when `--html` is present. With the locked Jupyter Book 2
version, `--strict` fails on errors and reports warnings separately; a successful
strict build is not necessarily warning-free. Do not suppress errors to make a
build pass.

### PDF prerequisites and validation

The `plain_latex_book` export requires `latexmk`, `xelatex`, and the template's
LaTeX packages. Provision those tools separately before requesting a PDF build.
For example, Ubuntu needs `latexmk`, `texlive-xetex`,
`texlive-fonts-recommended`, and `texlive-plain-generic`. PDF image conversion can
also require ImageMagick. HTML builds do not need this toolchain.

After generating the API reference, run the checked PDF export from the
repository root:

```powershell
uv run --frozen --no-sync python -m build_scripts.build_docs_pdf
```

The helper checks prerequisites before building, runs
`jupyter-book build --site --pdf --strict --logs`, and requires a fresh, readable
`doc\exports\book.pdf` with fresh native logs free of fatal LaTeX diagnostics.
`--site` validates the book content in strict mode without building static HTML;
the locked Jupyter Book version only applies strict error checking to site
content, not to standalone exports.
Jupyter Book can log an export failure without returning a nonzero exit, so its
exit code alone is not sufficient PDF evidence. A stale PDF does not count as a
successful new export. Inspect the PDF's chapters, images, citations, and layout
as well; compilation does not prove that every interactive HTML element has a
correct print representation.

Where GNU Make is available, `make docs-build` prepares the API reference, builds
strict HTML, and generates RSS. `make docs-build-pdf` prepares the API reference
and runs the checked PDF export. `make docs-build-all` builds HTML first and then
checks the PDF export. These local targets do not change the hosted multiversion
publication workflow.
1 change: 0 additions & 1 deletion doc/getting_started/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,6 @@
Welcome to PyRIT! Getting up and running takes two steps: **install** the package, then **configure** your AI endpoints.

:::::{grid} 1 1 3 3
:gutter: 3

::::{card} 📦 Install PyRIT
:link: ./install
Expand Down
1 change: 0 additions & 1 deletion doc/getting_started/configuration.md
Original file line number Diff line number Diff line change
Expand Up @@ -44,7 +44,6 @@ This gives you an in-memory database with configured targets and scorers registe
For anything beyond a quick test — especially `pyrit_scan`, scenarios, and repeated use — you'll want to save your configuration to files in `~/.pyrit/`:

:::::{grid} 1 1 2 2
:gutter: 3

::::{card} 🔑 Populating Secrets
:link: ./populating_secrets
Expand Down
2 changes: 0 additions & 2 deletions doc/getting_started/install.md
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,6 @@ Choose the installation method that best fits your use case.
## For Users

:::::{grid} 1 1 2 2
:gutter: 3

::::{card} 🐋 User Docker Installation
:link: ./install_docker
Expand All @@ -26,7 +25,6 @@ Install with pip or uv. Best if you need to integrate PyRIT into existing Python
## For Contributors

:::::{grid} 1 1 2 2
:gutter: 3

::::{card} 🐋 Contributor Docker Installation
:link: ./install_devcontainers
Expand Down
Loading
Loading