#!/usr/bin/env python3 """Generate a realistic benign-PDF fixture for the false-positive regression test. The fixture mimics the structure of a technical book like *Linux from Scratch*: a multi-page PDF with a table of contents containing in-document links, body paragraphs containing URLs to kernel.org and linuxfromscratch.org, mailing-list URLs that contain "support" in the path, mailto: links to authors, and a page that mentions `wget`, `exploit`, `payload`, etc. in prose. This is the regression corpus for the false-positive fix. The integration test asserts that scanning this PDF produces ZERO findings. """ from pathlib import Path from reportlab.pdfgen import canvas from reportlab.lib.pagesizes import letter import pypdf from pypdf.generic import ( ArrayObject, DictionaryObject, NameObject, NumberObject, TextStringObject, ) FIXTURES_DIR = Path(__file__).resolve().parent.parent / "tests" / "fixtures" FIXTURES_DIR.mkdir(parents=True, exist_ok=True) def make_benign_realworld_pdf(): """A multi-page PDF that exercises every URL type that USED TO be a false positive. Contains: - http:// URLs to kernel.org mirrors - https:// URLs to github.com paths that mention brands ("microsoft") - https:// URLs with "support", "verify", "account" in the path - mailto: links to authors - tel: phone links - In-document cross-reference links (GoTo actions) - A page that mentions wget / exploit / payload in prose - A .ru URL (country-code TLD that used to fire PHISHING_TLD) """ out_path = FIXTURES_DIR / "benign_realworld.pdf" # Step 1: draw the pages with reportlab. base_path = FIXTURES_DIR / "_base_realworld.pdf" c = canvas.Canvas(str(base_path), pagesize=letter) c.setTitle("Linux from Scratch (Sample Chapter)") c.setAuthor("Gerard Beekmans") c.setSubject("Sample technical-document PDF for false-positive regression test") # Page 1 — TOC-style page with prose containing URLs. c.drawString(80, 720, "Chapter 1. Introduction") c.drawString(80, 700, "See the official site at https://www.linuxfromscratch.org/") c.drawString(80, 680, "Mailing lists: https://lists.linuxfromscratch.org/listinfo/lfs-support") c.drawString(80, 660, "Source mirrors: http://ftp.osuosl.org/pub/lfs/") c.drawString(80, 640, "Patches hosted at https://github.com/LFS-project/build-scripts") c.drawString(80, 620, "Bug reports: mailto:lfs-support@linuxfromscratch.org") c.drawString(80, 600, "Kernel sources: https://www.kernel.org/pub/linux/kernel/") c.drawString(80, 580, "Phone: tel:+1-555-123-4567") c.showPage() # Page 2 — prose mentioning common security words. c.drawString(80, 720, "Chapter 2. Building the System") c.drawString(80, 700, "Run wget to download the package from the mirror.") c.drawString(80, 680, "The exploit described in CVE-2024-1234 affects older kernels.") c.drawString(80, 660, "The attacker's payload is delivered via a crafted document.") c.drawString(80, 640, "Use /bin/sh as the login shell.") c.drawString(80, 620, "On Windows, use PowerShell to install the module.") c.drawString(80, 600, "Russian mirror: https://ftp.ru.debian.org/debian/") c.drawString(80, 580, "Wikipedia: https://en.wikipedia.org/wiki/Microsoft_Windows") c.drawString(80, 560, "Github org: https://github.com/microsoft/vscode") c.showPage() # Page 3 — page reference with GoTo (in-document link). c.drawString(80, 720, "Chapter 3. Cross-references") c.drawString(80, 700, "See Chapter 1 for introduction details.") c.drawString(80, 680, "External resources:") c.drawString(80, 660, "- https://www.ietf.org/rfc/rfc2616.txt") c.drawString(80, 640, "- https://www.w3.org/TR/html5/") c.drawString(80, 620, "- https://docs.python.org/3/library/") c.drawString(80, 600, "- https://example.com/account/verify") c.drawString(80, 580, "- https://example.com/login") c.showPage() c.save() # Step 2: post-process with pypdf to inject URI-action annotations # on the pages, mimicking real hyperlinks in a published book. reader = pypdf.PdfReader(str(base_path)) writer = pypdf.PdfWriter() for page in reader.pages: writer.add_page(page) # Hyperlinks to inject — one per page. Each entry is (page_idx, x1, y1, x2, y2, uri). # The URI action is the structure that triggers the PdfUri vector in # the parser. We deliberately include the URLs that USED TO be # false positives. hyperlinks = [ # Page 0 — TOC links. (0, 80, 695, 400, 710, "https://www.linuxfromscratch.org/"), (0, 80, 675, 400, 690, "https://lists.linuxfromscratch.org/listinfo/lfs-support"), (0, 80, 655, 400, 670, "http://ftp.osuosl.org/pub/lfs/"), (0, 80, 635, 400, 650, "https://github.com/LFS-project/build-scripts"), (0, 80, 595, 400, 610, "mailto:lfs-support@linuxfromscratch.org"), (0, 80, 575, 400, 590, "https://www.kernel.org/pub/linux/kernel/"), (0, 80, 555, 400, 570, "tel:+1-555-123-4567"), # Page 1 — body links. (1, 80, 575, 400, 590, "https://ftp.ru.debian.org/debian/"), (1, 80, 555, 400, 570, "https://en.wikipedia.org/wiki/Microsoft_Windows"), (1, 80, 535, 400, 550, "https://github.com/microsoft/vscode"), # Page 2 — external resources. (2, 80, 655, 400, 670, "https://www.ietf.org/rfc/rfc2616.txt"), (2, 80, 615, 400, 630, "https://example.com/account/verify"), (2, 80, 595, 400, 610, "https://example.com/login"), ] for page_idx, x1, y1, x2, y2, uri in hyperlinks: page = writer.pages[page_idx] # Build the link annotation. uri_action = DictionaryObject({ NameObject("/Type"): NameObject("/Action"), NameObject("/S"): NameObject("/URI"), NameObject("/URI"): TextStringObject(uri), }) uri_action_ref = writer._add_object(uri_action) annot = DictionaryObject({ NameObject("/Type"): NameObject("/Annot"), NameObject("/Subtype"): NameObject("/Link"), NameObject("/Rect"): ArrayObject([ NumberObject(x1), NumberObject(y1), NumberObject(x2), NumberObject(y2), ]), NameObject("/A"): uri_action_ref, NameObject("/Border"): ArrayObject([ NumberObject(0), NumberObject(0), NumberObject(0), ]), }) annot_ref = writer._add_object(annot) if "/Annots" not in page: page[NameObject("/Annots")] = ArrayObject() page[NameObject("/Annots")].append(annot_ref) with open(out_path, "wb") as f: writer.write(f) base_path.unlink() return out_path def main(): path = make_benign_realworld_pdf() print(f" wrote {path} ({path.stat().st_size} bytes)") if __name__ == "__main__": main()