corbel/scripts/gen_benign_realworld_pdf.py

163 lines
6.8 KiB
Python

#!/usr/bin/env python3
"""Generate a realistic benign-PDF fixture for the false-positive regression test.
The fixture mimics the structure of a technical book like *Linux from
Scratch*: a multi-page PDF with a table of contents containing
in-document links, body paragraphs containing URLs to kernel.org and
linuxfromscratch.org, mailing-list URLs that contain "support" in the
path, mailto: links to authors, and a page that mentions `wget`,
`exploit`, `payload`, etc. in prose.
This is the regression corpus for the false-positive fix. The integration
test asserts that scanning this PDF produces ZERO findings.
"""
from pathlib import Path
from reportlab.pdfgen import canvas
from reportlab.lib.pagesizes import letter
import pypdf
from pypdf.generic import (
ArrayObject,
DictionaryObject,
NameObject,
NumberObject,
TextStringObject,
)
FIXTURES_DIR = Path(__file__).resolve().parent.parent / "tests" / "fixtures"
FIXTURES_DIR.mkdir(parents=True, exist_ok=True)
def make_benign_realworld_pdf():
"""A multi-page PDF that exercises every URL type that USED TO be a false positive.
Contains:
- http:// URLs to kernel.org mirrors
- https:// URLs to github.com paths that mention brands ("microsoft")
- https:// URLs with "support", "verify", "account" in the path
- mailto: links to authors
- tel: phone links
- In-document cross-reference links (GoTo actions)
- A page that mentions wget / exploit / payload in prose
- A .ru URL (country-code TLD that used to fire PHISHING_TLD)
"""
out_path = FIXTURES_DIR / "benign_realworld.pdf"
# Step 1: draw the pages with reportlab.
base_path = FIXTURES_DIR / "_base_realworld.pdf"
c = canvas.Canvas(str(base_path), pagesize=letter)
c.setTitle("Linux from Scratch (Sample Chapter)")
c.setAuthor("Gerard Beekmans")
c.setSubject("Sample technical-document PDF for false-positive regression test")
# Page 1 — TOC-style page with prose containing URLs.
c.drawString(80, 720, "Chapter 1. Introduction")
c.drawString(80, 700, "See the official site at https://www.linuxfromscratch.org/")
c.drawString(80, 680, "Mailing lists: https://lists.linuxfromscratch.org/listinfo/lfs-support")
c.drawString(80, 660, "Source mirrors: http://ftp.osuosl.org/pub/lfs/")
c.drawString(80, 640, "Patches hosted at https://github.com/LFS-project/build-scripts")
c.drawString(80, 620, "Bug reports: mailto:lfs-support@linuxfromscratch.org")
c.drawString(80, 600, "Kernel sources: https://www.kernel.org/pub/linux/kernel/")
c.drawString(80, 580, "Phone: tel:+1-555-123-4567")
c.showPage()
# Page 2 — prose mentioning common security words.
c.drawString(80, 720, "Chapter 2. Building the System")
c.drawString(80, 700, "Run wget to download the package from the mirror.")
c.drawString(80, 680, "The exploit described in CVE-2024-1234 affects older kernels.")
c.drawString(80, 660, "The attacker's payload is delivered via a crafted document.")
c.drawString(80, 640, "Use /bin/sh as the login shell.")
c.drawString(80, 620, "On Windows, use PowerShell to install the module.")
c.drawString(80, 600, "Russian mirror: https://ftp.ru.debian.org/debian/")
c.drawString(80, 580, "Wikipedia: https://en.wikipedia.org/wiki/Microsoft_Windows")
c.drawString(80, 560, "Github org: https://github.com/microsoft/vscode")
c.showPage()
# Page 3 — page reference with GoTo (in-document link).
c.drawString(80, 720, "Chapter 3. Cross-references")
c.drawString(80, 700, "See Chapter 1 for introduction details.")
c.drawString(80, 680, "External resources:")
c.drawString(80, 660, "- https://www.ietf.org/rfc/rfc2616.txt")
c.drawString(80, 640, "- https://www.w3.org/TR/html5/")
c.drawString(80, 620, "- https://docs.python.org/3/library/")
c.drawString(80, 600, "- https://example.com/account/verify")
c.drawString(80, 580, "- https://example.com/login")
c.showPage()
c.save()
# Step 2: post-process with pypdf to inject URI-action annotations
# on the pages, mimicking real hyperlinks in a published book.
reader = pypdf.PdfReader(str(base_path))
writer = pypdf.PdfWriter()
for page in reader.pages:
writer.add_page(page)
# Hyperlinks to inject — one per page. Each entry is (page_idx, x1, y1, x2, y2, uri).
# The URI action is the structure that triggers the PdfUri vector in
# the parser. We deliberately include the URLs that USED TO be
# false positives.
hyperlinks = [
# Page 0 — TOC links.
(0, 80, 695, 400, 710, "https://www.linuxfromscratch.org/"),
(0, 80, 675, 400, 690, "https://lists.linuxfromscratch.org/listinfo/lfs-support"),
(0, 80, 655, 400, 670, "http://ftp.osuosl.org/pub/lfs/"),
(0, 80, 635, 400, 650, "https://github.com/LFS-project/build-scripts"),
(0, 80, 595, 400, 610, "mailto:lfs-support@linuxfromscratch.org"),
(0, 80, 575, 400, 590, "https://www.kernel.org/pub/linux/kernel/"),
(0, 80, 555, 400, 570, "tel:+1-555-123-4567"),
# Page 1 — body links.
(1, 80, 575, 400, 590, "https://ftp.ru.debian.org/debian/"),
(1, 80, 555, 400, 570, "https://en.wikipedia.org/wiki/Microsoft_Windows"),
(1, 80, 535, 400, 550, "https://github.com/microsoft/vscode"),
# Page 2 — external resources.
(2, 80, 655, 400, 670, "https://www.ietf.org/rfc/rfc2616.txt"),
(2, 80, 615, 400, 630, "https://example.com/account/verify"),
(2, 80, 595, 400, 610, "https://example.com/login"),
]
for page_idx, x1, y1, x2, y2, uri in hyperlinks:
page = writer.pages[page_idx]
# Build the link annotation.
uri_action = DictionaryObject({
NameObject("/Type"): NameObject("/Action"),
NameObject("/S"): NameObject("/URI"),
NameObject("/URI"): TextStringObject(uri),
})
uri_action_ref = writer._add_object(uri_action)
annot = DictionaryObject({
NameObject("/Type"): NameObject("/Annot"),
NameObject("/Subtype"): NameObject("/Link"),
NameObject("/Rect"): ArrayObject([
NumberObject(x1), NumberObject(y1),
NumberObject(x2), NumberObject(y2),
]),
NameObject("/A"): uri_action_ref,
NameObject("/Border"): ArrayObject([
NumberObject(0), NumberObject(0), NumberObject(0),
]),
})
annot_ref = writer._add_object(annot)
if "/Annots" not in page:
page[NameObject("/Annots")] = ArrayObject()
page[NameObject("/Annots")].append(annot_ref)
with open(out_path, "wb") as f:
writer.write(f)
base_path.unlink()
return out_path
def main():
path = make_benign_realworld_pdf()
print(f" wrote {path} ({path.stat().st_size} bytes)")
if __name__ == "__main__":
main()