From cf84eac593645f8c547aa31f07c2819dd4d455ac Mon Sep 17 00:00:00 2001 From: Jeremy Anderson Date: Sat, 29 Aug 2026 00:35:39 -0400 Subject: [PATCH] fixed a logic flaw --- FIX-NOTES-false-positive-redesign.md | 235 +++++ FIX-NOTES-indoc-links.md | 219 +++++ QUICKSTART.md | 353 +++++-- README.md | 257 +++-- scripts/gen_benign_realworld_pdf.py | 162 ++++ src/core/config.rs | 104 +- src/core/pipeline.rs | 122 ++- src/core/types.rs | 10 +- src/lib.rs | 6 +- src/main.rs | 56 +- src/parsers/docx_parser.rs | 80 +- src/parsers/pdf_parser.rs | 36 +- src/quarantine/mod.rs | 282 ++++-- src/scanner/context_filter.rs | 240 +++-- src/scanner/heuristics.rs | 1311 +++++++++++++++++--------- src/scanner/mod.rs | 52 +- src/scanner/signatures.rs | 645 +++++-------- tests/fixtures/benign_realworld.pdf | 366 +++++++ tests/pipeline_integration.rs | 96 +- 19 files changed, 3327 insertions(+), 1305 deletions(-) create mode 100644 FIX-NOTES-false-positive-redesign.md create mode 100644 FIX-NOTES-indoc-links.md create mode 100644 scripts/gen_benign_realworld_pdf.py create mode 100644 tests/fixtures/benign_realworld.pdf diff --git a/FIX-NOTES-false-positive-redesign.md b/FIX-NOTES-false-positive-redesign.md new file mode 100644 index 0000000..0d0a375 --- /dev/null +++ b/FIX-NOTES-false-positive-redesign.md @@ -0,0 +1,235 @@ +# Patch notes — false-positive redesign (master) + +## Problem + +Operators reported the scanner was generating ~130 findings on a clean +technical PDF (*Linux from Scratch*). The findings were mostly false +positives: every plain-HTTP URL, every URL whose path contained a word +like "support" or "account", every URL mentioning a brand in its path, +every `mailto:` link, every in-document cross-reference, and every +paragraph mentioning `wget` or `exploit` in prose. + +An earlier revision attempted to fix this by adding a hard-coded +allow-list of well-known documentation domains (`linuxfromscratch.org`, +`kernel.org`, `github.com`, etc.). This was correctly rejected by the +operator as a per-file band-aid — it made the Linux-from-Scratch PDF +stop alerting without solving the underlying problem, and it would +produce the same false positives on every other technical document the +scanner had never seen. + +This revision takes the principled approach: every detector must be +backed by a verifiable property, either of the document itself or of +an external authority. No thresholds, no per-file or per-domain +exceptions, no "suspicious" tier. + +## Design + +Every detector in the redesigned scanner falls into exactly one of two +categories. + +### Category 1 — Verifiable executable intent + +The vector contains a structure whose only purpose is to execute code +or spawn a process. Presence is the threat. There is no "benign +JavaScript in a PDF action" or "benign Launch action". + +| Detector | Triggers on | Verifiable property | +|---|---|---| +| Active script in PDF | `/JavaScript` or `/JS` action stream | The action dictionary has `S = JavaScript` | +| Program launch in PDF | `/Launch` action with `/F`, `/Win`, `/Mac`, `/Unix` | The action dictionary has `S = Launch` | +| External program exec in EPUB | ` // PoC for CVE-2024-1234", + ); + let finding = evaluate(&node, &Config::default()).unwrap(); + assert_eq!(finding.classification, ThreatClassification::EducationalContent); + } + + #[test] + fn cve_writeup_prose_without_signature_is_no_finding() { + // Text that mentions `eval()` in prose but does not contain a + // structural signature (`/JavaScript`, `` tag, DOCX VBA +//! project, embedded file whose bytes match a known executable +//! magic, URI using an executable scheme (`javascript:`, +//! `vbscript:`, `data:text/html`). +//! +//! - **Category 2 — Verifiable impersonation.** The vector lies about +//! identity in a way that is provably wrong. Examples: URL whose +//! host is byte-equal to a known homograph string (`micros0ft.com`), +//! URL whose authority section contains a `user:pass@` credential +//! pair (legitimate URLs never embed credentials), URL whose host +//! mixes Unicode scripts (Cyrillic 'о' inside an otherwise-Latin +//! "microsoft.com"). +//! +//! There is no Category 3 in the offline scanner — external +//! threat-intel feeds (URLhaus, PhishTank, OpenPhish) contribute +//! entries to the Category 1 / 2 tables via `load_external_rules` and +//! are consulted through the same match functions. +//! +//! ## Design invariant +//! +//! Every detector is backed by a verifiable property, either of the +//! document itself or of an external authority. No thresholds, no +//! per-file or per-domain exceptions, no "suspicious" tier. The +//! `Suspicious` classification is retained on the enum for API +//! compatibility but no default detector produces it; the +//! `emit_suspicious` config flag defaults to `false`. use crate::core::config::Config; use crate::core::types::{ @@ -10,20 +38,22 @@ use crate::core::types::{ }; /// Inspect a single executable vector and produce a [`Finding`] if it -/// is flagged as suspicious or malicious. +/// matches a Category 1 or Category 2 detector. /// -/// Returns `None` if the vector is considered safe (rare — most vectors -/// produce at least a `Suspicious` finding). +/// Returns `None` if no detector fires — the vector is Benign. pub fn inspect_vector(vector: &ExecutableVector, config: &Config) -> Option { let classification = classify_vector(vector, config); let recommendation = recommendation_for(&classification); - // Drop Suspicious findings if the operator disabled them. + // Drop Suspicious findings if the operator disabled them. (No + // detector in the default set produces Suspicious anymore, but the + // tier is retained for API compatibility and for future detectors + // that may produce genuinely indeterminate signals.) if !config.emit_suspicious && matches!(classification, ThreatClassification::Suspicious) { return None; } - // Drop Benign findings (rare, but possible for whitelisted URI schemes). + // Drop Benign findings. if matches!(classification, ThreatClassification::Benign) { return None; } @@ -51,84 +81,169 @@ pub fn inspect_vector(vector: &ExecutableVector, config: &Config) -> Option ThreatClassification { match vector.vector_type { - // PDF JavaScript in an executable hook is always malicious. + // ── Category 1: verifiable executable intent ─────────── + // + // Each of these vector types is structurally an executable + // hook. Presence of the structure is the threat — there is no + // "benign JavaScript in a PDF action" or "benign Launch action". + + // PDF /JavaScript or /JS action stream. VectorType::PdfJavaScript => ThreatClassification::Malicious( MaliciousType::ActiveJavaScriptInjection, ), - // PDF /Launch is always malicious. + // PDF /Launch action (execute an external program). VectorType::PdfLaunch => ThreatClassification::Malicious(MaliciousType::LaunchAction), - // PDF embedded file — inspect for executable / high-risk signatures. + // EPUB