diff --git a/CLAUDE.md b/CLAUDE.md index 0b9a698..5e688aa 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -18,44 +18,58 @@ Web attribution and compliance scanner. Checks robots.txt, RSL licenses, and TDM ## Architecture -Single binary, seven modules: +Cargo workspace with two crates: ``` -src/ - main.rs — CLI entry point (clap), dispatches to analyze or serve - ai_crawlers.rs — Canonical list of 26 AI crawlers (GPTBot, ClaudeBot, etc.) - analyzer.rs — Core logic: parse robots.txt, extract licenses, evaluate TDM, analyze AI bots - fetcher.rs — HTTP fetching for robots.txt and /.well-known/tdmrep.json - models.rs — Data types: AnalysisResult, TdmPolicy, BotAnalysisResult, request/response shapes - output.rs — Formatters: table, JSON, CSV (with AI bot columns), compact text - server.rs — Axum HTTP API (GET /health, POST /analyze) +crates/ + core/ — policycheck-core (pure library, no I/O, WASM-compatible) + src/ + lib.rs — PolicyAnalyzer facade, orchestrates all checks + ai_crawlers.rs — Canonical list of 26 AI crawlers (GPTBot, ClaudeBot, etc.) + models.rs — Data types: AnalysisResult, TdmPolicy, BotAnalysisResult + checks/ + mod.rs — Check module index + robots.rs — RFC 9309 robots.txt parsing (user agents, paths, crawl delay) + rsl.rs — RSL licence extraction (global, group-scoped, precedence) + content_signals.rs — Cloudflare Content Signals (search, ai-input, ai-train) + tdm.rs — W3C TDMRep pattern matching and rule evaluation + ai_bots.rs — Per-bot access analysis for 26 AI crawlers + cli/ — policycheck (binary: CLI + HTTP server) + src/ + main.rs — CLI entry point (clap), dispatches to analyze or serve + analyzer.rs — Network-aware analyzer wrapping core with HTTP fetching + fetcher.rs — HTTP fetching for robots.txt and /.well-known/tdmrep.json + output.rs — Formatters: table, JSON, CSV (with AI bot columns), compact text + server.rs — Axum HTTP API (GET /health, POST /analyze) ``` -No workspaces, no proc macros, no feature flags. Keep it simple. +The core crate has 4 dependencies (texting_robots, serde, serde_json, url) and no network I/O. The CLI crate owns reqwest, axum, and all I/O concerns. No proc macros, no feature flags. ## Commands ```bash # Build -cargo build # Debug +cargo build # Debug (whole workspace) cargo build --release # Optimised (LTO, strip) +cargo build -p policycheck # CLI only # Test -cargo test # All tests +cargo test --workspace # All tests (core + CLI) +cargo test -p policycheck-core # Core library only cargo test -- --nocapture # With stdout # Run - Single URL -cargo run -- analyze --url https://www.nytimes.com -cargo run -- analyze --url https://github.com --format json +cargo run -p policycheck -- analyze --url https://www.nytimes.com +cargo run -p policycheck -- analyze --url https://github.com --format json # Run - Bulk analysis with CSV export (advertiser use case) -cargo run -- analyze --csv publishers.csv --format csv --output results.csv +cargo run -p policycheck -- analyze --csv publishers.csv --format csv --output results.csv # Run - HTTP server -cargo run -- serve --port 3000 +cargo run -p policycheck -- serve --port 3000 # Lint -cargo clippy +cargo clippy --workspace --all-targets cargo fmt --check ``` @@ -69,27 +83,37 @@ cargo fmt --check ## Testing -Tests live alongside code in `#[cfg(test)] mod tests` blocks. Currently in `analyzer.rs`: +Tests live alongside code in `#[cfg(test)] mod tests` blocks. 53 tests across both crates: -- Unit tests for robots.txt parsing (user agents, paths, licenses) -- RSL licence extraction (global, group-scoped, precedence, absolute URI validation) -- TDM pattern matching (wildcards, end markers, complex patterns) -- TDM rule evaluation (async tests with `#[tokio::test]`) +**Core (35 tests)** — pure unit tests, no I/O: +- `checks::robots` — user agent extraction, path parsing, allow/disallow +- `checks::rsl` — global/group-scoped licences, precedence, absolute URI validation +- `checks::content_signals` — signal parsing, group scoping, Cloudflare format +- `checks::tdm` — pattern matching (wildcards, `$` end markers), rule evaluation +- `checks::ai_bots` — wildcard blocking, selective blocking, bot count +- `lib.rs` — integration tests via `PolicyAnalyzer::analyze()` + +**CLI (17 tests)** — server, fetcher, output, CSV: +- `server` — health check, empty URLs, too-many-URLs validation +- `fetcher` — URL construction for robots.txt and TDM endpoints +- `output` — CSV headers, comma escaping, JSON round-trip +- `analyzer` — CSV column detection, bare domain prefixing, empty row skipping When adding features: 1. Write `#[test]` or `#[tokio::test]` in the relevant module 2. Make it fail 3. Implement until green -4. `cargo clippy` + `cargo fmt` +4. `cargo clippy --workspace --all-targets` + `cargo fmt` ## Standards Implemented | Standard | Status | Where | |----------|--------|-------| -| RFC 9309 (Robots Exclusion Protocol) | Done | `analyzer.rs` via `texting_robots` | -| RSL (Responsible Sourcing License) | Done | `analyzer.rs::extract_licenses()` | -| W3C TDMRep | Done | `fetcher.rs::fetch_tdm_policy()`, `analyzer.rs::evaluate_tdm_policy()` | -| AI Crawler Analysis | Done | `ai_crawlers.rs`, `analyzer.rs::analyze_ai_bots()` | +| RFC 9309 (Robots Exclusion Protocol) | Done | `core::checks::robots` via `texting_robots` | +| RSL (Responsible Sourcing License) | Done | `core::checks::rsl` | +| Cloudflare Content Signals | Done | `core::checks::content_signals` | +| W3C TDMRep | Done | `cli::fetcher` + `core::checks::tdm` | +| AI Crawler Analysis | Done | `core::ai_crawlers` + `core::checks::ai_bots` | | RFC 9116 (security.txt) | Planned | — | | RFC 8615 (Well-Known URIs) | Planned | — | diff --git a/Cargo.lock b/Cargo.lock index a0906d5..4a32a6d 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -248,34 +248,11 @@ version = "7.1.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b34115915337defe99b2aff5c2ce6771e5fbc4079f4b506301f5cf394c8452f7" dependencies = [ - "crossterm", "strum", "strum_macros", "unicode-width", ] -[[package]] -name = "crossterm" -version = "0.27.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f476fe445d41c9e991fd07515a6f463074b782242ccf4a5b7b1d1012e70824df" -dependencies = [ - "bitflags", - "crossterm_winapi", - "libc", - "parking_lot", - "winapi", -] - -[[package]] -name = "crossterm_winapi" -version = "0.9.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "acdd7c62a3665c7f6830a51635d9ac9b23ed385797f70a83bb8bafe9c572ab2b" -dependencies = [ - "winapi", -] - [[package]] name = "csv" version = "1.4.0" @@ -882,7 +859,7 @@ checksum = "8b870d8c151b6f2fb93e84a13146138f05d02ed11c7e7c54f8826aaaf7c9f184" [[package]] name = "policycheck" -version = "0.2.1" +version = "0.3.0" dependencies = [ "anyhow", "axum", @@ -890,17 +867,27 @@ dependencies = [ "comfy-table", "csv", "http-body-util", + "policycheck-core", "reqwest", "serde", "serde_json", "tempfile", - "texting_robots", "tokio", "tower 0.4.13", "tower-http 0.5.2", "url", ] +[[package]] +name = "policycheck-core" +version = "0.3.0" +dependencies = [ + "serde", + "serde_json", + "texting_robots", + "url", +] + [[package]] name = "potential_utf" version = "0.1.4" @@ -990,7 +977,7 @@ dependencies = [ "once_cell", "socket2", "tracing", - "windows-sys 0.52.0", + "windows-sys 0.60.2", ] [[package]] @@ -1821,28 +1808,6 @@ dependencies = [ "rustls-pki-types", ] -[[package]] -name = "winapi" -version = "0.3.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5c839a674fcd7a98952e593242ea400abe93992746761e38641405d28b00f419" -dependencies = [ - "winapi-i686-pc-windows-gnu", - "winapi-x86_64-pc-windows-gnu", -] - -[[package]] -name = "winapi-i686-pc-windows-gnu" -version = "0.4.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ac3b87c63620426dd9b991e5ce0329eff545bccbbb34f3be09ff6fb6ab51b7b6" - -[[package]] -name = "winapi-x86_64-pc-windows-gnu" -version = "0.4.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "712e227841d057c1ee1cd2fb22fa7e5a5461ae8e48fa2ca79ec42cfc1931183f" - [[package]] name = "windows-link" version = "0.2.1" diff --git a/Cargo.toml b/Cargo.toml index 86297e2..face4c6 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,44 +1,15 @@ -[package] -name = "policycheck" -version = "0.2.1" +[workspace] +members = ["crates/core", "crates/cli"] +resolver = "2" + +[workspace.package] +version = "0.3.0" edition = "2021" rust-version = "1.75" authors = ["OpenAttribution Contributors"] -description = "Publisher policy compliance checker - verifies robots.txt, RSL licenses, Content Signals, and TDM policies" -readme = "README.md" license = "MIT" repository = "https://github.com/openattribution-org/policycheck" homepage = "https://openattribution.org" -keywords = ["robots", "rsl", "tdm", "policy", "compliance"] -categories = ["web-programming", "command-line-utilities"] -exclude = [ - ".github/", - ".dockerignore", - "Dockerfile", - "fly.toml", - "CLAUDE.md", - "*.csv", - "!example.csv" -] - -[dependencies] -texting_robots = "0.2" -reqwest = { version = "0.12", default-features = false, features = ["rustls-tls"] } -clap = { version = "4.5", features = ["derive"] } -csv = "1.3" -serde = { version = "1.0", features = ["derive"] } -serde_json = "1.0" -anyhow = "1.0" -tokio = { version = "1", features = ["full"] } -axum = "0.7" -tower = { version = "0.4", features = ["util"] } -http-body-util = "0.1" -tower-http = { version = "0.5", features = ["cors"] } -comfy-table = "=7.1.1" -url = "2.5" - -[dev-dependencies] -tempfile = "3" [profile.release] opt-level = 3 diff --git a/Dockerfile b/Dockerfile index 0d60a73..fc94a7c 100644 --- a/Dockerfile +++ b/Dockerfile @@ -5,12 +5,10 @@ WORKDIR /app # Copy manifests COPY Cargo.toml Cargo.lock ./ +COPY crates ./crates -# Copy source -COPY src ./src - -# Build release binary -RUN cargo build --release +# Build release binary (CLI server only — skip WASM crate) +RUN cargo build --release -p policycheck # Runtime stage FROM debian:bookworm-slim diff --git a/README.md b/README.md index 115e3e8..294c393 100644 --- a/README.md +++ b/README.md @@ -17,7 +17,7 @@ PolicyCheck helps you **scrape responsibly** by checking multiple compliance sig - ✅ **Robots.txt** - What paths you can crawl (REP/RFC 9309) - 📜 **RSL Licenses** - Required licensing terms (Responsible Sourcing License) - 🎯 **Content Signals** - AI usage preferences (Cloudflare's policy framework) -- 🤖 **TDM Policies** - Text & Data Mining permissions (coming soon) +- 🤖 **TDM Policies** - Text & Data Mining permissions (W3C TDMRep) - 🔒 **Privacy Controls** - DNT, GPC signals (coming soon) - 📧 **Security Contacts** - Who to contact about scraping (coming soon) @@ -65,11 +65,31 @@ For development or the latest unreleased features: ```bash git clone https://github.com/openattribution-org/policycheck.git cd policycheck -cargo build --release +cargo build --release -p policycheck ``` The binary will be at `target/release/policycheck`. +#### As a Library + +Use the core parsing library in your own project (no network I/O, WASM-compatible): + +```bash +cargo add policycheck-core +``` + +```rust +use policycheck_core::PolicyAnalyzer; + +let analyzer = PolicyAnalyzer::new("GPTBot".to_string()); +let result = analyzer.analyze( + "https://www.nytimes.com", + "User-agent: GPTBot\nDisallow: /\n", + None, +); +assert!(!result.is_path_allowed); +``` + ### Basic Usage ```bash @@ -617,10 +637,11 @@ Multi-platform images available for `linux/amd64` and `linux/arm64`. #### Building from Source ```dockerfile -FROM rust:1.92-slim as builder +FROM rust:1.85-bookworm as builder WORKDIR /app -COPY . . -RUN cargo build --release +COPY Cargo.toml Cargo.lock ./ +COPY crates ./crates +RUN cargo build --release -p policycheck FROM debian:bookworm-slim RUN apt-get update && apt-get install -y ca-certificates && rm -rf /var/lib/apt/lists/* @@ -711,9 +732,10 @@ podman-compose up -d - [x] CSV batch processing - [x] HTTP API server - [x] Multiple output formats +- [x] TDM (Text & Data Mining) policy detection (`/.well-known/tdmrep.json`) +- [x] Content Signals (Cloudflare AI policy framework) ### 🚧 In Progress -- [ ] TDM (Text & Data Mining) policy detection (`/.well-known/tdmrep.json`) - [ ] Security contact discovery (`/.well-known/security.txt`) - [ ] Privacy control detection (DNT, GPC) @@ -797,7 +819,7 @@ PolicyCheck implements the following standards: - ✅ **RFC 9309**: Robots Exclusion Protocol (REP) - ✅ **RSL Standard**: Responsible Sourcing License - ✅ **Content Signals**: Cloudflare's AI Policy Framework (CC0 License) -- 🚧 **W3C TDMRep**: Text and Data Mining Reservation Protocol (planned) +- ✅ **W3C TDMRep**: Text and Data Mining Reservation Protocol - 🚧 **RFC 9116**: security.txt (planned) - 🚧 **RFC 8615**: Well-Known URIs (planned) diff --git a/crates/cli/Cargo.toml b/crates/cli/Cargo.toml new file mode 100644 index 0000000..53c5c15 --- /dev/null +++ b/crates/cli/Cargo.toml @@ -0,0 +1,45 @@ +[package] +name = "policycheck" +description = "Publisher policy compliance checker - verifies robots.txt, RSL licenses, Content Signals, and TDM policies" +readme = "../../README.md" +keywords = ["robots", "rsl", "tdm", "policy", "compliance"] +categories = ["web-programming", "command-line-utilities"] +exclude = [ + ".github/", + ".dockerignore", + "Dockerfile", + "fly.toml", + "CLAUDE.md", + "*.csv", + "!example.csv" +] +version.workspace = true +edition.workspace = true +rust-version.workspace = true +authors.workspace = true +license.workspace = true +repository.workspace = true +homepage.workspace = true + +[[bin]] +name = "policycheck" +path = "src/main.rs" + +[dependencies] +policycheck-core = { path = "../core" } +reqwest = { version = "0.12", default-features = false, features = ["rustls-tls"] } +clap = { version = "4.5", features = ["derive"] } +serde = { version = "1.0", features = ["derive"] } +serde_json = "1.0" +anyhow = "1.0" +tokio = { version = "1", features = ["full"] } +axum = "0.7" +tower = { version = "0.4", features = ["util"] } +http-body-util = "0.1" +tower-http = { version = "0.5", features = ["cors"] } +url = "2.5" +csv = "1.3" +comfy-table = { version = "7", default-features = false } + +[dev-dependencies] +tempfile = "3" diff --git a/crates/cli/src/analyzer.rs b/crates/cli/src/analyzer.rs new file mode 100644 index 0000000..5d366c0 --- /dev/null +++ b/crates/cli/src/analyzer.rs @@ -0,0 +1,170 @@ +use crate::fetcher::RobotFetcher; +use anyhow::Result; +use policycheck_core::models::{AnalysisResult, AnalysisStatus}; +use policycheck_core::PolicyAnalyzer; + +/// Network-aware analyzer that wraps the core `PolicyAnalyzer` with HTTP fetching. +pub struct RobotAnalyzer { + core: PolicyAnalyzer, + fetcher: RobotFetcher, +} + +impl RobotAnalyzer { + pub fn new(user_agent: String) -> Self { + Self { + core: PolicyAnalyzer::new(user_agent), + fetcher: RobotFetcher::new(), + } + } + + pub fn with_fetcher(user_agent: String, fetcher: RobotFetcher) -> Self { + Self { + core: PolicyAnalyzer::new(user_agent), + fetcher, + } + } + + /// Read URLs from a CSV file. + /// + /// Detects the URL column by header name (url, link, website) and adds + /// an `https://` prefix to bare domains. + pub fn read_csv(&self, path: &std::path::Path) -> Result> { + let mut reader = csv::ReaderBuilder::new() + .has_headers(true) + .from_path(path)?; + + let mut urls = Vec::new(); + + let headers = reader.headers()?.clone(); + + let url_col_idx = headers + .iter() + .position(|h| { + let h_lower = h.to_lowercase(); + h_lower.contains("url") || h_lower == "link" || h_lower == "website" + }) + .unwrap_or(0); + + for result in reader.records() { + let record = result?; + + if let Some(url) = record.get(url_col_idx) { + let url = url.trim(); + if !url.is_empty() { + let url = if url.starts_with("http://") || url.starts_with("https://") { + url.to_string() + } else { + format!("https://{}", url) + }; + urls.push(url); + } + } + } + + Ok(urls) + } + + /// Fetch and analyze a single URL. + pub async fn analyze_url(&self, url: &str) -> AnalysisResult { + // Fetch robots.txt + let (robots_url, content) = match self.fetcher.fetch_for_url(url).await { + Ok(data) => data, + Err(e) => { + return AnalysisResult::error( + url.to_string(), + e.to_string(), + AnalysisStatus::FetchError, + ); + } + }; + + // Fetch TDM policy (optional — don't fail if missing) + let tdm_rules = self.fetcher.fetch_tdm_policy(url).await.ok(); + + // Delegate to core analyzer + let mut result = self.core.analyze(url, &content, tdm_rules); + result.robots_url = robots_url; + + result + } + + /// Analyze multiple URLs concurrently. + pub async fn analyze_urls(&self, urls: &[String]) -> Vec { + let mut handles = vec![]; + + for url in urls { + let url = url.clone(); + let url_for_error = url.clone(); + let fetcher = self.fetcher.clone(); + let core_user_agent = self.core_user_agent(); + + let handle = tokio::spawn(async move { + let analyzer = RobotAnalyzer::with_fetcher(core_user_agent, fetcher); + analyzer.analyze_url(&url).await + }); + + handles.push((url_for_error, handle)); + } + + let mut results = vec![]; + for (url, handle) in handles { + match handle.await { + Ok(result) => results.push(result), + Err(e) => results.push(AnalysisResult::error( + url, + format!("Task failed: {}", e), + AnalysisStatus::FetchError, + )), + } + } + + results + } + + fn core_user_agent(&self) -> String { + self.core.user_agent().to_string() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_read_csv_url_column_detection() { + let dir = tempfile::tempdir().unwrap(); + let csv_path = dir.path().join("test.csv"); + std::fs::write( + &csv_path, + "name,Company URL,notes\nNYT,https://www.nytimes.com,news\n", + ) + .unwrap(); + let analyzer = RobotAnalyzer::new("*".to_string()); + let urls = analyzer.read_csv(&csv_path).unwrap(); + assert_eq!(urls, vec!["https://www.nytimes.com"]); + } + + #[test] + fn test_read_csv_bare_domain_adds_https_prefix() { + let dir = tempfile::tempdir().unwrap(); + let csv_path = dir.path().join("test.csv"); + std::fs::write(&csv_path, "url\ngithub.com\n").unwrap(); + let analyzer = RobotAnalyzer::new("*".to_string()); + let urls = analyzer.read_csv(&csv_path).unwrap(); + assert_eq!(urls, vec!["https://github.com"]); + } + + #[test] + fn test_read_csv_empty_rows_skipped() { + let dir = tempfile::tempdir().unwrap(); + let csv_path = dir.path().join("test.csv"); + std::fs::write( + &csv_path, + "url\nhttps://github.com\n\n \nhttps://www.nytimes.com\n", + ) + .unwrap(); + let analyzer = RobotAnalyzer::new("*".to_string()); + let urls = analyzer.read_csv(&csv_path).unwrap(); + assert_eq!(urls, vec!["https://github.com", "https://www.nytimes.com"]); + } +} diff --git a/src/fetcher.rs b/crates/cli/src/fetcher.rs similarity index 98% rename from src/fetcher.rs rename to crates/cli/src/fetcher.rs index 31b9ca0..c0a5f45 100644 --- a/src/fetcher.rs +++ b/crates/cli/src/fetcher.rs @@ -1,9 +1,8 @@ use anyhow::{Context, Result}; +use policycheck_core::models::TdmRule; use std::time::Duration; use url::Url; -use crate::models::TdmRule; - #[derive(Clone)] pub struct RobotFetcher { client: reqwest::Client, @@ -112,7 +111,6 @@ impl RobotFetcher { .await .context("Failed to read TDM response body")?; - // Parse JSON array of TDM rules let rules: Vec = serde_json::from_str(&content).context("Failed to parse tdmrep.json")?; diff --git a/src/main.rs b/crates/cli/src/main.rs similarity index 91% rename from src/main.rs rename to crates/cli/src/main.rs index a338399..7c61f1f 100644 --- a/src/main.rs +++ b/crates/cli/src/main.rs @@ -2,10 +2,8 @@ use anyhow::{Context, Result}; use clap::{Parser, Subcommand, ValueEnum}; use std::path::PathBuf; -mod ai_crawlers; mod analyzer; mod fetcher; -mod models; mod output; mod server; @@ -74,7 +72,7 @@ async fn main() -> Result<()> { csv, user_agent, format, - output, + output: output_path, } => { let analyzer = RobotAnalyzer::new(user_agent); @@ -104,9 +102,9 @@ async fn main() -> Result<()> { OutputFormat::Table => output::format_table(&results)?, }; - if let Some(output_path) = output { - std::fs::write(&output_path, &output_str).context("Failed to write output file")?; - println!("Results written to {}", output_path.display()); + if let Some(path) = output_path { + std::fs::write(&path, &output_str).context("Failed to write output file")?; + println!("Results written to {}", path.display()); } else { println!("{}", output_str); } diff --git a/src/output.rs b/crates/cli/src/output.rs similarity index 98% rename from src/output.rs rename to crates/cli/src/output.rs index 5604cff..848445f 100644 --- a/src/output.rs +++ b/crates/cli/src/output.rs @@ -1,7 +1,7 @@ -use crate::ai_crawlers::{AICrawler, BotStatus}; -use crate::models::{AnalysisResult, AnalysisStatus}; use anyhow::Result; use comfy_table::{modifiers::UTF8_ROUND_CORNERS, presets::UTF8_FULL, *}; +use policycheck_core::ai_crawlers::{AICrawler, BotStatus}; +use policycheck_core::models::{AnalysisResult, AnalysisStatus}; pub fn format_table(results: &[AnalysisResult]) -> Result { let mut table = Table::new(); @@ -260,7 +260,7 @@ pub fn format_compact(results: &[AnalysisResult]) -> Result { output.push('\n'); } - // Content Signals (Cloudflare AI policy framework) + // Content Signals if result.content_signal_search.is_some() || result.content_signal_ai_input.is_some() || result.content_signal_ai_train.is_some() @@ -384,7 +384,6 @@ pub fn format_compact(results: &[AnalysisResult]) -> Result { #[cfg(test)] mod tests { use super::*; - use crate::models::AnalysisResult; fn make_result(url: &str) -> AnalysisResult { AnalysisResult { @@ -426,7 +425,6 @@ mod tests { result.user_agents = vec!["Bot,One".to_string(), "BotTwo".to_string()]; let csv = format_csv(&[result]).unwrap(); let data_line = csv.lines().nth(1).unwrap(); - // csv crate quotes fields containing commas assert!(data_line.contains("\"Bot,One; BotTwo\"")); } diff --git a/src/server.rs b/crates/cli/src/server.rs similarity index 92% rename from src/server.rs rename to crates/cli/src/server.rs index b1d83be..ac21733 100644 --- a/src/server.rs +++ b/crates/cli/src/server.rs @@ -1,5 +1,4 @@ use crate::analyzer::RobotAnalyzer; -use crate::models::{AnalysisStatus, AnalyzeRequest, AnalyzeResponse}; use anyhow::Result; use axum::{ extract::{DefaultBodyLimit, Json}, @@ -8,10 +7,31 @@ use axum::{ routing::{get, post}, Router, }; +use policycheck_core::models::{AnalysisResult, AnalysisStatus}; +use serde::{Deserialize, Serialize}; use serde_json::json; use std::env; use tower_http::cors::{Any, CorsLayer}; +#[derive(Debug, Deserialize)] +pub struct AnalyzeRequest { + pub urls: Vec, + #[serde(default = "default_user_agent")] + pub user_agent: String, +} + +fn default_user_agent() -> String { + "*".to_string() +} + +#[derive(Debug, Serialize)] +pub struct AnalyzeResponse { + pub results: Vec, + pub total: usize, + pub successful: usize, + pub failed: usize, +} + const MAX_URLS_PER_REQUEST: usize = 100; struct ApiError { @@ -29,9 +49,6 @@ impl IntoResponse for ApiError { } pub fn build_router() -> Router { - // CORS configuration - allow specific origins via ALLOWED_ORIGINS env var - // Format: comma-separated list (e.g., "https://openattribution.org,https://example.com") - // If not set, allows all origins (useful for development and open source deployments) let cors = if let Ok(allowed_origins) = env::var("ALLOWED_ORIGINS") { let origins: Vec = allowed_origins .split(',') diff --git a/crates/core/Cargo.toml b/crates/core/Cargo.toml new file mode 100644 index 0000000..61d0390 --- /dev/null +++ b/crates/core/Cargo.toml @@ -0,0 +1,19 @@ +[package] +name = "policycheck-core" +description = "Publisher policy compliance library - parses robots.txt, RSL licenses, Content Signals, and TDM policies" +readme = "README.md" +keywords = ["robots", "rsl", "tdm", "policy", "compliance"] +categories = ["web-programming", "parser-implementations"] +version.workspace = true +edition.workspace = true +rust-version.workspace = true +authors.workspace = true +license.workspace = true +repository.workspace = true +homepage.workspace = true + +[dependencies] +texting_robots = "0.2" +serde = { version = "1.0", features = ["derive"] } +serde_json = "1.0" +url = "2.5" diff --git a/crates/core/README.md b/crates/core/README.md new file mode 100644 index 0000000..c736fd0 --- /dev/null +++ b/crates/core/README.md @@ -0,0 +1,47 @@ +# policycheck-core + +Pure parsing and analysis library for web publisher compliance policies. No network I/O — callers provide raw content, this library parses it. + +Part of [PolicyCheck](https://github.com/openattribution-org/policycheck) by [OpenAttribution](https://openattribution.org). + +## Standards + +| Standard | Module | +|----------|--------| +| RFC 9309 (Robots Exclusion Protocol) | `checks::robots` | +| RSL (Responsible Sourcing License) | `checks::rsl` | +| W3C TDMRep (Text & Data Mining) | `checks::tdm` | +| Cloudflare Content Signals | `checks::content_signals` | +| AI Crawler Analysis (26 bots) | `checks::ai_bots` | + +## Usage + +```rust +use policycheck_core::PolicyAnalyzer; + +let analyzer = PolicyAnalyzer::new("GPTBot".to_string()); +let result = analyzer.analyze( + "https://www.nytimes.com", + "User-agent: GPTBot\nDisallow: /\n", + None, +); + +assert!(!result.is_path_allowed); +``` + +Individual check modules are also available directly: + +```rust +use policycheck_core::checks; + +let rsl = checks::rsl::extract(robots_txt_content, "GPTBot"); +let signals = checks::content_signals::extract(robots_txt_content, "*"); +``` + +## WASM + +This crate is designed to be WASM-compatible (no filesystem or network dependencies). See `policycheck-wasm` for the browser wrapper. + +## Licence + +MIT diff --git a/src/ai_crawlers.rs b/crates/core/src/ai_crawlers.rs similarity index 100% rename from src/ai_crawlers.rs rename to crates/core/src/ai_crawlers.rs diff --git a/crates/core/src/checks/ai_bots.rs b/crates/core/src/checks/ai_bots.rs new file mode 100644 index 0000000..26ab753 --- /dev/null +++ b/crates/core/src/checks/ai_bots.rs @@ -0,0 +1,138 @@ +//! AI bot access analysis. +//! +//! Checks the access status of 26 known AI crawlers against robots.txt content. +//! Each bot is individually parsed to get accurate per-bot allow/disallow results. + +use crate::ai_crawlers::{AICrawler, BotStatus}; +use crate::models::BotAnalysisResult; +use texting_robots::Robot; + +/// Analyze each AI bot individually to determine its access status. +/// +/// Re-parses robots.txt for each mentioned bot to get accurate per-bot results +/// (texting_robots bakes the user-agent into its parsed state). +pub fn analyze(content: &str, url: &str) -> Vec { + let all_bots = AICrawler::get_all(); + let mut results = Vec::new(); + + // Normalize user agents for comparison (case-insensitive) + let user_agents_lower: Vec = extract_user_agents_lower(content); + + // Check if a wildcard User-agent: * rule exists (applies to all bots) + let has_wildcard = user_agents_lower.iter().any(|ua| ua == "*"); + + for bot in all_bots { + let bot_name_lower = bot.name.to_lowercase(); + + // A bot is affected by robots.txt if it's explicitly named or a wildcard rule exists + let is_mentioned = has_wildcard + || user_agents_lower + .iter() + .any(|ua| ua == &bot_name_lower || ua.contains(&bot_name_lower)); + + let status = if is_mentioned { + // Bot is mentioned (or covered by wildcard) — check actual access + match Robot::new(&bot.name, content.as_bytes()) { + Ok(robot) => { + if robot.allowed(url) { + BotStatus::Allowed + } else { + BotStatus::Blocked + } + } + Err(_) => BotStatus::Allowed, + } + } else { + // Bot not mentioned and no wildcard — allowed by default + BotStatus::Allowed + }; + + results.push(BotAnalysisResult { + bot_name: bot.name, + company: bot.company, + category: format!("{:?}", bot.category), + status, + }); + } + + results +} + +/// Extract user agents from content, lowercased for comparison. +fn extract_user_agents_lower(content: &str) -> Vec { + let mut agents = Vec::new(); + + for line in content.lines() { + let line = line.trim(); + if line.to_lowercase().starts_with("user-agent:") { + if let Some(agent) = line.split(':').nth(1) { + let agent = agent.trim().to_lowercase(); + if !agents.contains(&agent) { + agents.push(agent); + } + } + } + } + + agents +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_analyze_ai_bots_all_allowed() { + let content = "User-agent: *\nAllow: /\n"; + let results = analyze(content, "https://github.com/"); + assert_eq!(results.len(), 26); + for bot in &results { + assert!( + matches!(bot.status, BotStatus::Allowed), + "{} should be allowed", + bot.bot_name + ); + } + } + + #[test] + fn test_analyze_ai_bots_returns_26() { + let results = analyze("", "https://github.com/"); + assert_eq!(results.len(), 26); + } + + #[test] + fn test_analyze_ai_bots_selective_blocking() { + let content = "\ +User-agent: GPTBot\nDisallow: /\n\n\ +User-agent: ClaudeBot\nDisallow: /\n\n\ +User-agent: *\nAllow: /\n"; + let results = analyze(content, "https://techcrunch.com/"); + + let gptbot = results.iter().find(|b| b.bot_name == "GPTBot").unwrap(); + assert!(matches!(gptbot.status, BotStatus::Blocked)); + + let claudebot = results.iter().find(|b| b.bot_name == "ClaudeBot").unwrap(); + assert!(matches!(claudebot.status, BotStatus::Blocked)); + + let perplexity = results + .iter() + .find(|b| b.bot_name == "PerplexityBot") + .unwrap(); + assert!(matches!(perplexity.status, BotStatus::Allowed)); + } + + #[test] + fn test_wildcard_disallow_blocks_all_ai_bots() { + let content = "User-agent: *\nDisallow: /\n"; + let results = analyze(content, "https://www.nytimes.com/"); + assert_eq!(results.len(), 26); + for bot in &results { + assert!( + matches!(bot.status, BotStatus::Blocked), + "{} should be blocked by wildcard Disallow: /", + bot.bot_name + ); + } + } +} diff --git a/crates/core/src/checks/content_signals.rs b/crates/core/src/checks/content_signals.rs new file mode 100644 index 0000000..c2233bd --- /dev/null +++ b/crates/core/src/checks/content_signals.rs @@ -0,0 +1,156 @@ +//! Cloudflare Content Signals extraction from robots.txt. +//! +//! Detects `Content-Signal:` directives from Cloudflare's AI policy framework. +//! Three signals: search, ai-input, ai-train (values: "yes" or "no"). +//! +//! See: + +/// Content Signals extraction result. +pub struct ContentSignalsResult { + pub search: Option, + pub ai_input: Option, + pub ai_train: Option, +} + +/// Extract Content Signals from robots.txt content. +/// +/// Parses `Content-Signal: search=yes, ai-train=no, ai-input=yes` directives. +/// Respects user-agent group scoping. +pub fn extract(content: &str, user_agent: &str) -> ContentSignalsResult { + let mut search_signal = None; + let mut ai_input_signal = None; + let mut ai_train_signal = None; + let mut in_matching_group = false; + let mut current_user_agents: Vec = Vec::new(); + + for line in content.lines() { + let line = line.trim(); + + // Skip comments and empty lines + if line.is_empty() || line.starts_with('#') { + continue; + } + + let line_lower = line.to_lowercase(); + + // Track user-agent groups + if line_lower.starts_with("user-agent:") { + if let Some(agent) = line.split(':').nth(1) { + let agent = agent.trim(); + current_user_agents.push(agent.to_string()); + + // Check if this matches our target user agent + if agent == user_agent || agent == "*" { + in_matching_group = true; + } + } + } else if line_lower.starts_with("content-signal:") { + // Only process if we're in the matching group or no group (global) + if current_user_agents.is_empty() || in_matching_group { + // Extract the value part after "Content-Signal:" + if let Some(signals_str) = line.split(':').nth(1) { + // Parse comma-separated key=value pairs + for pair in signals_str.split(',') { + let pair = pair.trim(); + if let Some((key, value)) = pair.split_once('=') { + let key = key.trim().to_lowercase(); + let value = value.trim().to_lowercase(); + + // Only accept "yes" or "no" values + if value == "yes" || value == "no" { + match key.as_str() { + "search" => search_signal = Some(value), + "ai-input" => ai_input_signal = Some(value), + "ai-train" => ai_train_signal = Some(value), + _ => {} // Ignore unknown signals + } + } + } + } + } + } + } else if !line_lower.starts_with("allow:") + && !line_lower.starts_with("disallow:") + && !line_lower.starts_with("sitemap:") + && !line_lower.starts_with("crawl-delay:") + && !line_lower.starts_with("license:") + && !current_user_agents.is_empty() + { + // Reset group context on unrecognized directive + if line.contains(':') { + current_user_agents.clear(); + in_matching_group = false; + } + } + } + + ContentSignalsResult { + search: search_signal, + ai_input: ai_input_signal, + ai_train: ai_train_signal, + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_content_signals_basic() { + let content = r#" +User-agent: * +Content-Signal: search=yes, ai-train=no, ai-input=yes +Allow: / + "#; + let result = extract(content, "*"); + + assert_eq!(result.search, Some("yes".to_string())); + assert_eq!(result.ai_input, Some("yes".to_string())); + assert_eq!(result.ai_train, Some("no".to_string())); + } + + #[test] + fn test_content_signals_partial() { + let content = r#" +User-agent: * +Content-Signal: search=yes, ai-train=no +Allow: / + "#; + let result = extract(content, "*"); + + assert_eq!(result.search, Some("yes".to_string())); + assert_eq!(result.ai_input, None); + assert_eq!(result.ai_train, Some("no".to_string())); + } + + #[test] + fn test_content_signals_group_scoped() { + let content = r#" +User-agent: Googlebot +Content-Signal: search=yes, ai-train=yes + +User-agent: GPTBot +Content-Signal: search=yes, ai-train=no, ai-input=no +Disallow: / + "#; + let result = extract(content, "GPTBot"); + + assert_eq!(result.search, Some("yes".to_string())); + assert_eq!(result.ai_input, Some("no".to_string())); + assert_eq!(result.ai_train, Some("no".to_string())); + } + + #[test] + fn test_content_signals_cloudflare_format() { + let content = r#" +User-Agent: * +Content-Signal: search=yes, ai-train=no +Allow: / + "#; + let result = extract(content, "*"); + + assert_eq!(result.search, Some("yes".to_string())); + assert_eq!(result.ai_input, None, "ai-input not specified"); + assert_eq!(result.ai_train, Some("no".to_string())); + } +} diff --git a/crates/core/src/checks/mod.rs b/crates/core/src/checks/mod.rs new file mode 100644 index 0000000..35e88a6 --- /dev/null +++ b/crates/core/src/checks/mod.rs @@ -0,0 +1,14 @@ +//! Compliance check modules. +//! +//! Each module implements analysis for a specific web compliance standard. +//! Modules that parse robots.txt receive the raw content as `&str`. +//! Modules with their own data source (e.g. TDM) receive pre-fetched data. +//! +//! To add a new standard, create a new module here and wire it into +//! `PolicyAnalyzer::analyze()` in `lib.rs`. + +pub mod ai_bots; +pub mod content_signals; +pub mod robots; +pub mod rsl; +pub mod tdm; diff --git a/crates/core/src/checks/robots.rs b/crates/core/src/checks/robots.rs new file mode 100644 index 0000000..ba180be --- /dev/null +++ b/crates/core/src/checks/robots.rs @@ -0,0 +1,132 @@ +//! Robots Exclusion Protocol (REP/RFC 9309) parsing. +//! +//! Extracts user agents, allowed/disallowed paths, crawl delay, +//! and sitemaps from robots.txt content. + +use texting_robots::Robot; + +/// Parsed robots.txt data for a specific user agent. +pub struct RobotsResult { + pub user_agents: Vec, + pub allowed_paths: Vec, + pub disallowed_paths: Vec, + pub is_path_allowed: bool, + pub crawl_delay: Option, + pub sitemaps: Vec, +} + +/// Parse robots.txt content and check path access for the given user agent. +pub fn analyze(content: &str, user_agent: &str, url: &str) -> RobotsResult { + let user_agents = extract_user_agents(content); + let (allowed_paths, disallowed_paths) = extract_paths(content); + + let (is_path_allowed, crawl_delay, sitemaps) = match Robot::new(user_agent, content.as_bytes()) + { + Ok(robot) => ( + robot.allowed(url), + robot.delay.map(|d| d as f64), + robot.sitemaps.clone(), + ), + Err(_) => (false, None, vec![]), + }; + + RobotsResult { + user_agents, + allowed_paths, + disallowed_paths, + is_path_allowed, + crawl_delay, + sitemaps, + } +} + +/// Extract all User-agent directives from robots.txt content. +#[allow(clippy::collapsible_if)] +fn extract_user_agents(content: &str) -> Vec { + let mut agents = Vec::new(); + + for line in content.lines() { + let line = line.trim(); + if line.to_lowercase().starts_with("user-agent:") { + if let Some(agent) = line.split(':').nth(1) { + let agent = agent.trim().to_string(); + if !agents.contains(&agent) { + agents.push(agent); + } + } + } + } + + agents +} + +/// Extract Allow and Disallow paths from robots.txt content. +#[allow(clippy::collapsible_if)] +fn extract_paths(content: &str) -> (Vec, Vec) { + let mut allowed = Vec::new(); + let mut disallowed = Vec::new(); + + for line in content.lines() { + let line = line.trim(); + + if line.to_lowercase().starts_with("allow:") { + if let Some(path) = line.split(':').nth(1) { + let path = path.trim(); + if !path.is_empty() && !allowed.contains(&path.to_string()) { + allowed.push(path.to_string()); + } + } + } else if line.to_lowercase().starts_with("disallow:") { + if let Some(path) = line.split(':').nth(1) { + let path = path.trim(); + if !path.is_empty() && !disallowed.contains(&path.to_string()) { + disallowed.push(path.to_string()); + } + } + } + } + + (allowed, disallowed) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_extract_user_agents() { + let content = "User-agent: *\nUser-agent: GoogleBot\nUser-agent: BingBot"; + let agents = extract_user_agents(content); + + assert_eq!(agents.len(), 3); + assert!(agents.contains(&"*".to_string())); + assert!(agents.contains(&"GoogleBot".to_string())); + assert!(agents.contains(&"BingBot".to_string())); + } + + #[test] + fn test_extract_paths() { + let content = "Allow: /public\nDisallow: /private\nDisallow: /admin"; + let (allowed, disallowed) = extract_paths(content); + + assert_eq!(allowed.len(), 1); + assert_eq!(allowed[0], "/public"); + assert_eq!(disallowed.len(), 2); + assert!(disallowed.contains(&"/private".to_string())); + assert!(disallowed.contains(&"/admin".to_string())); + } + + #[test] + fn test_analyze_allowed_path() { + let content = "User-agent: *\nAllow: /\n"; + let result = analyze(content, "*", "https://example.com/page"); + assert!(result.is_path_allowed); + } + + #[test] + fn test_analyze_disallowed_path() { + let content = "User-agent: *\nDisallow: /\n"; + let result = analyze(content, "*", "https://example.com/page"); + assert!(!result.is_path_allowed); + } +} diff --git a/crates/core/src/checks/rsl.rs b/crates/core/src/checks/rsl.rs new file mode 100644 index 0000000..e657b11 --- /dev/null +++ b/crates/core/src/checks/rsl.rs @@ -0,0 +1,256 @@ +//! RSL (Responsible Sourcing License) extraction from robots.txt. +//! +//! Detects `License:` directives per the RSL standard. Supports both +//! global licenses (outside any User-agent group) and group-scoped +//! licenses. Group-scoped licenses take precedence over global ones. +//! +//! See: + +/// RSL license extraction result. +pub struct RslResult { + pub global_licenses: Vec, + pub group_licenses: Vec, + pub active_licenses: Vec, +} + +/// Extract RSL licenses from robots.txt content. +/// +/// Returns global, group-scoped, and active (effective) licenses. +/// Precedence: group-scoped licenses override global licenses. +pub fn extract(content: &str, user_agent: &str) -> RslResult { + let mut global_licenses = Vec::new(); + let mut group_licenses = Vec::new(); + let mut current_user_agents: Vec = Vec::new(); + let mut in_matching_group = false; + + for line in content.lines() { + let line = line.trim(); + + // Skip comments and empty lines + if line.is_empty() || line.starts_with('#') { + continue; + } + + let line_lower = line.to_lowercase(); + + if line_lower.starts_with("user-agent:") { + // Extract user agent + if let Some(agent) = line.split(':').nth(1) { + let agent = agent.trim(); + current_user_agents.push(agent.to_string()); + + // Check if this matches our target user agent + if agent == user_agent || agent == "*" { + in_matching_group = true; + } + } + } else if line_lower.starts_with("license:") { + // Extract license URI + if let Some(license_uri) = line + .split(':') + .skip(1) + .collect::>() + .join(":") + .split_whitespace() + .next() + { + let license = license_uri.trim().to_string(); + + if !license.is_empty() { + // Validate it's an absolute URI (basic check) + if license.starts_with("http://") || license.starts_with("https://") { + if current_user_agents.is_empty() { + // Global license (outside any user-agent group) + if !global_licenses.contains(&license) { + global_licenses.push(license); + } + } else if in_matching_group { + // Group-scoped license for our user agent + if !group_licenses.contains(&license) { + group_licenses.push(license); + } + } + } + } + } + } else if !line_lower.starts_with("allow:") + && !line_lower.starts_with("disallow:") + && !line_lower.starts_with("sitemap:") + && !line_lower.starts_with("crawl-delay:") + && !current_user_agents.is_empty() + { + // Reset group context on unrecognized directive (new group likely starting) + if line.contains(':') { + current_user_agents.clear(); + in_matching_group = false; + } + } + } + + // Determine active licenses based on RSL precedence rules + let active_licenses = if !group_licenses.is_empty() { + group_licenses.clone() + } else { + global_licenses.clone() + }; + + RslResult { + global_licenses, + group_licenses, + active_licenses, + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_extract_global_licenses() { + let content = "License: https://example.com/license.xml\nUser-agent: *\nDisallow: /private"; + let result = extract(content, "*"); + + assert_eq!(result.global_licenses.len(), 1); + assert_eq!(result.global_licenses[0], "https://example.com/license.xml"); + assert_eq!(result.group_licenses.len(), 0); + } + + #[test] + fn test_extract_group_scoped_licenses() { + let content = r#" +License: https://example.com/global.xml +User-agent: GPTBot +License: https://example.com/gptbot.xml +Disallow: / + "#; + let result = extract(content, "GPTBot"); + + assert_eq!(result.global_licenses.len(), 1); + assert_eq!(result.global_licenses[0], "https://example.com/global.xml"); + assert_eq!(result.group_licenses.len(), 1); + assert_eq!(result.group_licenses[0], "https://example.com/gptbot.xml"); + } + + #[test] + fn test_license_precedence_group_overrides_global() { + let content = r#" +License: https://example.com/global.xml +User-agent: GPTBot +License: https://example.com/gptbot.xml + "#; + let result = extract(content, "GPTBot"); + + assert_eq!(result.active_licenses.len(), 1); + assert_eq!(result.active_licenses[0], "https://example.com/gptbot.xml"); + } + + #[test] + fn test_license_requires_absolute_uri() { + let content = "License: /relative/path.xml\nLicense: https://example.com/absolute.xml"; + let result = extract(content, "*"); + + assert_eq!(result.global_licenses.len(), 1); + assert_eq!( + result.global_licenses[0], + "https://example.com/absolute.xml" + ); + } + + #[test] + fn test_license_ignores_comments() { + let content = r#" +# This is a comment with License: https://fake.com/license.xml +License: https://example.com/real.xml + "#; + let result = extract(content, "*"); + + assert_eq!(result.global_licenses.len(), 1); + assert_eq!(result.global_licenses[0], "https://example.com/real.xml"); + } + + #[test] + fn test_wildcard_user_agent_matches() { + let content = r#" +User-agent: * +License: https://example.com/wildcard.xml + "#; + let result = extract(content, "MyBot"); + + assert_eq!(result.group_licenses.len(), 1); + assert_eq!(result.group_licenses[0], "https://example.com/wildcard.xml"); + } + + #[test] + fn test_separate_user_agent_groups_dont_mix_licenses() { + let content = r#" +User-agent: Googlebot +License: https://example.com/google-only.xml +Disallow: /admin + +User-agent: GPTBot +License: https://example.com/gpt-only.xml +Disallow: / + "#; + let result = extract(content, "GPTBot"); + + assert_eq!( + result.global_licenses.len(), + 0, + "Should have no global licenses" + ); + assert_eq!( + result.group_licenses.len(), + 1, + "Should have exactly 1 group license" + ); + assert_eq!(result.group_licenses[0], "https://example.com/gpt-only.xml"); + assert!( + !result + .group_licenses + .contains(&"https://example.com/google-only.xml".to_string()), + "Should NOT include Googlebot's license" + ); + } + + #[test] + fn test_rsl_real_world_rslstandard_org() { + let content = r#" +License: https://rslcollective.org/royalty.xml + +User-agent: * +Disallow: + "#; + let result = extract(content, "*"); + + assert_eq!(result.global_licenses.len(), 1); + assert_eq!( + result.global_licenses[0], + "https://rslcollective.org/royalty.xml" + ); + assert_eq!(result.group_licenses.len(), 0, "No group-scoped licenses"); + assert_eq!( + result.active_licenses[0], + "https://rslcollective.org/royalty.xml" + ); + } + + #[test] + fn test_rsl_real_world_medium_com() { + let content = r#" +User-agent: * +Allow: /about + +User-agent: GPTBot +User-agent: ClaudeBot +User-agent: FacebookBot +License: https://medium.com/license.xml +Disallow: / + "#; + let result = extract(content, "GPTBot"); + + assert_eq!(result.global_licenses.len(), 0, "No global licenses"); + assert_eq!(result.group_licenses.len(), 1, "Should have group license"); + assert_eq!(result.group_licenses[0], "https://medium.com/license.xml"); + assert_eq!(result.active_licenses[0], "https://medium.com/license.xml"); + } +} diff --git a/crates/core/src/checks/tdm.rs b/crates/core/src/checks/tdm.rs new file mode 100644 index 0000000..be90dac --- /dev/null +++ b/crates/core/src/checks/tdm.rs @@ -0,0 +1,183 @@ +//! TDM (Text & Data Mining) policy evaluation. +//! +//! Evaluates rules from `/.well-known/tdmrep.json` against a URL path. +//! Supports `*` wildcards and `$` end-of-pattern markers per W3C TDMRep. +//! +//! See: + +use crate::models::{TdmPolicy, TdmRule}; +use url::Url; + +/// Evaluate TDM rules against a URL and return the matching policy. +/// +/// Returns `None` if the URL cannot be parsed. +/// Rules are evaluated in order (first-match wins per W3C TDMRep). +pub fn evaluate(url: &str, rules: Vec) -> Option { + let parsed = Url::parse(url).ok()?; + let path = parsed.path(); + + // Find the first matching rule (first-match wins per W3C TDMRep) + let matched_rule = rules + .iter() + .find(|rule| match_pattern(&rule.location, path)) + .cloned(); + + let is_reserved = matched_rule + .as_ref() + .map(|r| r.tdm_reservation == 1) + .unwrap_or(false); + + Some(TdmPolicy { + rules, + matched_rule, + is_reserved, + }) +} + +/// Match a path against a TDM location pattern. +/// +/// Supports `*` wildcard and `$` end-of-pattern marker. +pub fn match_pattern(pattern: &str, path: &str) -> bool { + // Remove $ end marker if present for processing + let (pattern, must_end) = if let Some(stripped) = pattern.strip_suffix('$') { + (stripped, true) + } else { + (pattern, false) + }; + + // Simple wildcard matching + let parts: Vec<&str> = pattern.split('*').collect(); + + if parts.len() == 1 { + // No wildcards - exact match (or prefix if no $) + if must_end { + return path == pattern; + } else { + return path.starts_with(pattern); + } + } + + // Check if path matches the pattern with wildcards + let mut path_pos = 0; + for (i, part) in parts.iter().enumerate() { + if part.is_empty() { + continue; + } + + if i == 0 { + // First part must match from the start + if !path[path_pos..].starts_with(part) { + return false; + } + path_pos += part.len(); + } else { + // Find the next occurrence + if let Some(pos) = path[path_pos..].find(part) { + path_pos += pos + part.len(); + } else { + return false; + } + } + } + + // If must_end is true, ensure we've consumed the entire path + !must_end || path_pos == path.len() +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_pattern_exact_match() { + assert!(match_pattern("/", "/")); + assert!(match_pattern("/docs", "/docs")); + assert!(match_pattern("/docs", "/docs/page")); + assert!(!match_pattern("/docs$", "/docs/page")); + } + + #[test] + fn test_pattern_wildcard() { + assert!(match_pattern("/docs/*", "/docs/page")); + assert!(match_pattern("/docs/*", "/docs/page/sub")); + assert!(match_pattern("*.pdf", "/file.pdf")); + assert!(match_pattern("*.pdf", "/docs/file.pdf")); + assert!(!match_pattern("/docs/*", "/other/page")); + } + + #[test] + fn test_pattern_end_marker() { + assert!(match_pattern("/docs$", "/docs")); + assert!(!match_pattern("/docs$", "/docs/")); + assert!(!match_pattern("/docs$", "/docs/page")); + assert!(match_pattern("/docs/page$", "/docs/page")); + } + + #[test] + fn test_pattern_complex() { + assert!(match_pattern("/*/public/*", "/docs/public/file")); + assert!(match_pattern("/docs/*.pdf$", "/docs/file.pdf")); + assert!(!match_pattern("/docs/*.pdf$", "/docs/file.pdf.bak")); + } + + #[test] + fn test_evaluate_rule_matching() { + let rules = vec![ + TdmRule { + location: "/".to_string(), + tdm_reservation: 1, + tdm_policy: Some("https://example.com/policy.html".to_string()), + }, + TdmRule { + location: "/public/*".to_string(), + tdm_reservation: 0, + tdm_policy: None, + }, + ]; + + // Test root path matches first rule + let policy = evaluate("https://example.com/", rules.clone()); + assert!(policy.is_some()); + let policy = policy.unwrap(); + assert!(policy.is_reserved); + assert_eq!(policy.matched_rule.as_ref().unwrap().location, "/"); + + // Test public path — first rule still wins (first-match) + let rules2 = vec![ + TdmRule { + location: "/public/*".to_string(), + tdm_reservation: 0, + tdm_policy: None, + }, + TdmRule { + location: "/".to_string(), + tdm_reservation: 1, + tdm_policy: Some("https://example.com/policy.html".to_string()), + }, + ]; + + let policy2 = evaluate("https://example.com/public/data", rules2); + assert!(policy2.is_some()); + let policy2 = policy2.unwrap(); + assert!(!policy2.is_reserved); + assert_eq!(policy2.matched_rule.as_ref().unwrap().location, "/public/*"); + } + + #[test] + fn test_evaluate_empty_rules_returns_unreserved_with_no_match() { + // GIVEN a valid URL but no TDM rules + let policy = evaluate("https://www.nytimes.com/article", vec![]); + + // SHOULD return a policy with no matched rule and not reserved + assert!(policy.is_some()); + let policy = policy.unwrap(); + assert!(!policy.is_reserved); + assert!(policy.matched_rule.is_none()); + } + + #[test] + fn test_evaluate_invalid_url_returns_none() { + let policy = evaluate("not-a-url", vec![]); + assert!(policy.is_none()); + } +} diff --git a/crates/core/src/lib.rs b/crates/core/src/lib.rs new file mode 100644 index 0000000..014f025 --- /dev/null +++ b/crates/core/src/lib.rs @@ -0,0 +1,179 @@ +//! PolicyCheck core library. +//! +//! Pure parsing and analysis logic for web compliance checking. +//! No network I/O — callers provide raw content, this library parses it. +//! +//! ## Architecture +//! +//! Each compliance standard lives in its own module under `checks/`: +//! - `checks::robots` — Robots Exclusion Protocol (RFC 9309) +//! - `checks::rsl` — Responsible Sourcing License +//! - `checks::content_signals` — Cloudflare Content Signals +//! - `checks::tdm` — W3C Text & Data Mining Reservation Protocol +//! - `checks::ai_bots` — AI crawler access analysis +//! +//! The `PolicyAnalyzer` orchestrates all checks into a unified `AnalysisResult`. + +pub mod ai_crawlers; +pub mod checks; +pub mod models; + +use models::{AnalysisResult, AnalysisStatus, TdmRule}; + +/// Core policy analyzer. Takes raw content (no fetching) and produces analysis results. +/// +/// ``` +/// use policycheck_core::PolicyAnalyzer; +/// +/// let analyzer = PolicyAnalyzer::new("GPTBot".to_string()); +/// let result = analyzer.analyze("https://example.com", "User-agent: *\nDisallow: /\n", None); +/// assert!(!result.is_path_allowed); +/// ``` +pub struct PolicyAnalyzer { + user_agent: String, +} + +impl PolicyAnalyzer { + pub fn new(user_agent: String) -> Self { + Self { user_agent } + } + + /// Get the configured user agent. + pub fn user_agent(&self) -> &str { + &self.user_agent + } + + /// Analyze raw robots.txt content for a given URL. + /// + /// `tdm_rules` is optional — pass pre-fetched `/.well-known/tdmrep.json` data + /// if available, or `None` to skip TDM evaluation. + pub fn analyze( + &self, + url: &str, + robots_txt: &str, + tdm_rules: Option>, + ) -> AnalysisResult { + // Run each compliance check module + let robots = checks::robots::analyze(robots_txt, &self.user_agent, url); + let rsl = checks::rsl::extract(robots_txt, &self.user_agent); + let signals = checks::content_signals::extract(robots_txt, &self.user_agent); + let tdm_policy = tdm_rules.and_then(|rules| checks::tdm::evaluate(url, rules)); + let ai_bot_analysis = checks::ai_bots::analyze(robots_txt, url); + + AnalysisResult { + url: url.to_string(), + robots_url: String::new(), // Caller sets this (they know the actual URL fetched) + status: AnalysisStatus::Success, + user_agents: robots.user_agents, + crawl_delay: robots.crawl_delay, + sitemaps: robots.sitemaps, + allowed_paths: robots.allowed_paths, + disallowed_paths: robots.disallowed_paths, + is_path_allowed: robots.is_path_allowed, + global_licenses: rsl.global_licenses, + group_licenses: rsl.group_licenses, + active_licenses: rsl.active_licenses, + content_signal_search: signals.search, + content_signal_ai_input: signals.ai_input, + content_signal_ai_train: signals.ai_train, + tdm_policy, + ai_bot_analysis, + error: None, + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_analyze_basic() { + let analyzer = PolicyAnalyzer::new("*".to_string()); + let result = analyzer.analyze("https://example.com", "User-agent: *\nAllow: /\n", None); + + assert!(matches!(result.status, AnalysisStatus::Success)); + assert!(result.is_path_allowed); + assert!(result.error.is_none()); + } + + #[test] + fn test_analyze_blocked() { + let analyzer = PolicyAnalyzer::new("GPTBot".to_string()); + let content = "User-agent: GPTBot\nDisallow: /\n"; + let result = analyzer.analyze("https://example.com", content, None); + + assert!(!result.is_path_allowed); + } + + #[test] + fn test_analyze_with_rsl() { + let analyzer = PolicyAnalyzer::new("*".to_string()); + let content = "License: https://example.com/license.xml\nUser-agent: *\nAllow: /\n"; + let result = analyzer.analyze("https://example.com", content, None); + + assert_eq!(result.global_licenses.len(), 1); + assert_eq!(result.active_licenses.len(), 1); + } + + #[test] + fn test_analyze_with_content_signals() { + let analyzer = PolicyAnalyzer::new("*".to_string()); + let content = "User-agent: *\nContent-Signal: search=yes, ai-train=no\nAllow: /\n"; + let result = analyzer.analyze("https://example.com", content, None); + + assert_eq!(result.content_signal_search, Some("yes".to_string())); + assert_eq!(result.content_signal_ai_train, Some("no".to_string())); + } + + #[test] + fn test_analyze_empty_robots_txt_returns_success_with_path_allowed() { + // GIVEN empty robots.txt content (site has no restrictions) + let analyzer = PolicyAnalyzer::new("GPTBot".to_string()); + + // WHEN we analyze + let result = analyzer.analyze("https://www.nytimes.com", "", None); + + // SHOULD succeed with path allowed (empty robots.txt = no restrictions) + assert!(matches!(result.status, AnalysisStatus::Success)); + assert!(result.is_path_allowed); + assert!(result.user_agents.is_empty()); + assert!(result.error.is_none()); + } + + #[test] + fn test_analyze_with_tdm_rules_sets_reservation_status() { + // GIVEN robots.txt allowing access and TDM rules reserving all content + let analyzer = PolicyAnalyzer::new("*".to_string()); + let tdm_rules = vec![models::TdmRule { + location: "/".to_string(), + tdm_reservation: 1, + tdm_policy: Some("https://www.nytimes.com/tdm-policy".to_string()), + }]; + + // WHEN we analyze with TDM rules + let result = analyzer.analyze( + "https://www.nytimes.com/article", + "User-agent: *\nAllow: /\n", + Some(tdm_rules), + ); + + // SHOULD report TDM reserved + let tdm = result.tdm_policy.unwrap(); + assert!(tdm.is_reserved); + assert_eq!(tdm.matched_rule.unwrap().location, "/"); + } + + #[test] + fn test_analyze_url_with_query_params_checks_path_correctly() { + // GIVEN robots.txt blocking /search + let analyzer = PolicyAnalyzer::new("*".to_string()); + let content = "User-agent: *\nDisallow: /search\n"; + + // WHEN checking a URL with query params under /search + let result = analyzer.analyze("https://www.nytimes.com/search?q=test", content, None); + + // SHOULD be disallowed (path starts with /search) + assert!(!result.is_path_allowed); + } +} diff --git a/src/models.rs b/crates/core/src/models.rs similarity index 84% rename from src/models.rs rename to crates/core/src/models.rs index cb949fc..4666a21 100644 --- a/src/models.rs +++ b/crates/core/src/models.rs @@ -58,7 +58,7 @@ pub enum AnalysisStatus { impl AnalysisResult { pub fn error(url: String, error: String, status: AnalysisStatus) -> Self { Self { - url: url.clone(), + url, robots_url: String::new(), status, user_agents: vec![], @@ -79,22 +79,3 @@ impl AnalysisResult { } } } - -#[derive(Debug, Deserialize)] -pub struct AnalyzeRequest { - pub urls: Vec, - #[serde(default = "default_user_agent")] - pub user_agent: String, -} - -fn default_user_agent() -> String { - "*".to_string() -} - -#[derive(Debug, Serialize)] -pub struct AnalyzeResponse { - pub results: Vec, - pub total: usize, - pub successful: usize, - pub failed: usize, -} diff --git a/src/analyzer.rs b/src/analyzer.rs deleted file mode 100644 index ecec303..0000000 --- a/src/analyzer.rs +++ /dev/null @@ -1,948 +0,0 @@ -use crate::ai_crawlers::{AICrawler, BotStatus}; -use crate::fetcher::RobotFetcher; -use crate::models::{AnalysisResult, AnalysisStatus, BotAnalysisResult, TdmPolicy, TdmRule}; -use anyhow::Result; -use std::path::Path; -use texting_robots::Robot; -use url::Url; - -pub struct RobotAnalyzer { - user_agent: String, - fetcher: RobotFetcher, -} - -impl RobotAnalyzer { - pub fn new(user_agent: String) -> Self { - Self { - user_agent, - fetcher: RobotFetcher::new(), - } - } - - pub fn with_fetcher(user_agent: String, fetcher: RobotFetcher) -> Self { - Self { - user_agent, - fetcher, - } - } - - /// Read URLs from a CSV file - pub fn read_csv(&self, path: &Path) -> Result> { - let mut reader = csv::ReaderBuilder::new() - .has_headers(true) - .from_path(path)?; - - let mut urls = Vec::new(); - - // Get headers to find URL column - let headers = reader.headers()?.clone(); - - // Find the URL column index (look for "url", "URL", "Company URL", etc.) - let url_col_idx = headers - .iter() - .position(|h| { - let h_lower = h.to_lowercase(); - h_lower.contains("url") || h_lower == "link" || h_lower == "website" - }) - .unwrap_or(0); // Default to first column if no URL header found - - // Read URLs from the identified column - for result in reader.records() { - let record = result?; - - if let Some(url) = record.get(url_col_idx) { - let url = url.trim(); - if !url.is_empty() { - // Add http:// prefix if missing - let url = if url.starts_with("http://") || url.starts_with("https://") { - url.to_string() - } else if !url.is_empty() { - format!("https://{}", url) - } else { - continue; - }; - urls.push(url); - } - } - } - - Ok(urls) - } - - /// Analyze a single URL - pub async fn analyze_url(&self, url: &str) -> AnalysisResult { - // Fetch robots.txt - let (robots_url, content) = match self.fetcher.fetch_for_url(url).await { - Ok(data) => data, - Err(e) => { - return AnalysisResult::error( - url.to_string(), - e.to_string(), - AnalysisStatus::FetchError, - ); - } - }; - - // Parse robots.txt - let robot = match Robot::new(&self.user_agent, content.as_bytes()) { - Ok(r) => r, - Err(e) => { - return AnalysisResult::error( - url.to_string(), - format!("Parse error: {:?}", e), - AnalysisStatus::ParseError, - ); - } - }; - - // Extract user agents from the content - let user_agents = self.extract_user_agents(&content); - - // Extract allowed and disallowed paths - let (allowed_paths, disallowed_paths) = self.extract_paths(&content); - - // Extract RSL licenses - let (global_licenses, group_licenses) = self.extract_licenses(&content); - - // Determine active licenses based on RSL precedence rules - let active_licenses = if !group_licenses.is_empty() { - group_licenses.clone() - } else { - global_licenses.clone() - }; - - // Extract Content Signals (Cloudflare's AI policy framework) - let (content_signal_search, content_signal_ai_input, content_signal_ai_train) = - self.extract_content_signals(&content); - - // Check if the original URL path is allowed - let is_path_allowed = robot.allowed(url); - - // Fetch and evaluate TDM policy - let tdm_policy = match self.fetcher.fetch_tdm_policy(url).await { - Ok(rules) => self.evaluate_tdm_policy(url, rules).await, - Err(_) => None, // TDM policy is optional, ignore errors - }; - - // Analyze AI bot access - let ai_bot_analysis = self.analyze_ai_bots(&content, url); - - AnalysisResult { - url: url.to_string(), - robots_url, - status: AnalysisStatus::Success, - user_agents, - crawl_delay: robot.delay.map(|d| d as f64), - sitemaps: robot.sitemaps.clone(), - allowed_paths, - disallowed_paths, - is_path_allowed, - global_licenses, - group_licenses, - active_licenses, - content_signal_search, - content_signal_ai_input, - content_signal_ai_train, - tdm_policy, - ai_bot_analysis, - error: None, - } - } - - /// Analyze multiple URLs concurrently - pub async fn analyze_urls(&self, urls: &[String]) -> Vec { - let mut handles = vec![]; - - for url in urls { - let url = url.clone(); - let url_for_error = url.clone(); - let user_agent = self.user_agent.clone(); - let fetcher = self.fetcher.clone(); - - let handle = tokio::spawn(async move { - let analyzer = RobotAnalyzer::with_fetcher(user_agent, fetcher); - analyzer.analyze_url(&url).await - }); - - handles.push((url_for_error, handle)); - } - - let mut results = vec![]; - for (url, handle) in handles { - match handle.await { - Ok(result) => results.push(result), - Err(e) => results.push(AnalysisResult::error( - url, - format!("Task failed: {}", e), - AnalysisStatus::FetchError, - )), - } - } - - results - } - - /// Extract user agents from robots.txt content - #[allow(clippy::collapsible_if)] - fn extract_user_agents(&self, content: &str) -> Vec { - let mut agents = Vec::new(); - - for line in content.lines() { - let line = line.trim(); - if line.to_lowercase().starts_with("user-agent:") { - if let Some(agent) = line.split(':').nth(1) { - let agent = agent.trim().to_string(); - if !agents.contains(&agent) { - agents.push(agent); - } - } - } - } - - agents - } - - /// Extract allowed and disallowed paths - #[allow(clippy::collapsible_if)] - fn extract_paths(&self, content: &str) -> (Vec, Vec) { - let mut allowed = Vec::new(); - let mut disallowed = Vec::new(); - - for line in content.lines() { - let line = line.trim(); - - if line.to_lowercase().starts_with("allow:") { - if let Some(path) = line.split(':').nth(1) { - let path = path.trim(); - if !path.is_empty() && !allowed.contains(&path.to_string()) { - allowed.push(path.to_string()); - } - } - } else if line.to_lowercase().starts_with("disallow:") { - if let Some(path) = line.split(':').nth(1) { - let path = path.trim(); - if !path.is_empty() && !disallowed.contains(&path.to_string()) { - disallowed.push(path.to_string()); - } - } - } - } - - (allowed, disallowed) - } - - /// Extract RSL licenses from robots.txt content - /// Returns (global_licenses, group_licenses) where: - /// - global_licenses: License directives outside any User-agent group - /// - group_licenses: License directives within the matching User-agent group - fn extract_licenses(&self, content: &str) -> (Vec, Vec) { - let mut global_licenses = Vec::new(); - let mut group_licenses = Vec::new(); - let mut current_user_agents: Vec = Vec::new(); - let mut in_matching_group = false; - - for line in content.lines() { - let line = line.trim(); - - // Skip comments and empty lines - if line.is_empty() || line.starts_with('#') { - continue; - } - - let line_lower = line.to_lowercase(); - - if line_lower.starts_with("user-agent:") { - // Extract user agent - if let Some(agent) = line.split(':').nth(1) { - let agent = agent.trim(); - current_user_agents.push(agent.to_string()); - - // Check if this matches our target user agent - if agent == self.user_agent || agent == "*" { - in_matching_group = true; - } - } - } else if line_lower.starts_with("license:") { - // Extract license URI - if let Some(license_uri) = line - .split(':') - .skip(1) - .collect::>() - .join(":") - .split_whitespace() - .next() - { - let license = license_uri.trim().to_string(); - - if !license.is_empty() { - // Validate it's an absolute URI (basic check) - if license.starts_with("http://") || license.starts_with("https://") { - if current_user_agents.is_empty() { - // Global license (outside any user-agent group) - if !global_licenses.contains(&license) { - global_licenses.push(license); - } - } else if in_matching_group { - // Group-scoped license for our user agent - if !group_licenses.contains(&license) { - group_licenses.push(license); - } - } - } - } - } - } else if !line_lower.starts_with("allow:") - && !line_lower.starts_with("disallow:") - && !line_lower.starts_with("sitemap:") - && !line_lower.starts_with("crawl-delay:") - && !current_user_agents.is_empty() - { - // Reset group context on unrecognized directive (new group likely starting) - if line.contains(':') { - current_user_agents.clear(); - in_matching_group = false; - } - } - } - - (global_licenses, group_licenses) - } - - /// Extract Content Signals from robots.txt (Cloudflare's AI policy framework) - /// Returns (search, ai-input, ai-train) signals as Option - /// Format: Content-Signal: search=yes, ai-train=no, ai-input=yes - fn extract_content_signals( - &self, - content: &str, - ) -> (Option, Option, Option) { - let mut search_signal = None; - let mut ai_input_signal = None; - let mut ai_train_signal = None; - let mut in_matching_group = false; - let mut current_user_agents: Vec = Vec::new(); - - for line in content.lines() { - let line = line.trim(); - - // Skip comments and empty lines - if line.is_empty() || line.starts_with('#') { - continue; - } - - let line_lower = line.to_lowercase(); - - // Track user-agent groups - if line_lower.starts_with("user-agent:") { - if let Some(agent) = line.split(':').nth(1) { - let agent = agent.trim(); - current_user_agents.push(agent.to_string()); - - // Check if this matches our target user agent - if agent == self.user_agent || agent == "*" { - in_matching_group = true; - } - } - } else if line_lower.starts_with("content-signal:") { - // Only process if we're in the matching group or no group (global) - if current_user_agents.is_empty() || in_matching_group { - // Extract the value part after "Content-Signal:" - if let Some(signals_str) = line.split(':').nth(1) { - // Parse comma-separated key=value pairs - for pair in signals_str.split(',') { - let pair = pair.trim(); - if let Some((key, value)) = pair.split_once('=') { - let key = key.trim().to_lowercase(); - let value = value.trim().to_lowercase(); - - // Only accept "yes" or "no" values - if value == "yes" || value == "no" { - match key.as_str() { - "search" => search_signal = Some(value), - "ai-input" => ai_input_signal = Some(value), - "ai-train" => ai_train_signal = Some(value), - _ => {} // Ignore unknown signals - } - } - } - } - } - } - } else if !line_lower.starts_with("allow:") - && !line_lower.starts_with("disallow:") - && !line_lower.starts_with("sitemap:") - && !line_lower.starts_with("crawl-delay:") - && !line_lower.starts_with("license:") - && !current_user_agents.is_empty() - { - // Reset group context on unrecognized directive - if line.contains(':') { - current_user_agents.clear(); - in_matching_group = false; - } - } - } - - (search_signal, ai_input_signal, ai_train_signal) - } - - /// Match a path against a TDM location pattern - /// Supports * wildcard and $ end-of-pattern marker - fn match_tdm_pattern(pattern: &str, path: &str) -> bool { - // Remove $ end marker if present for processing - let (pattern, must_end) = if let Some(stripped) = pattern.strip_suffix('$') { - (stripped, true) - } else { - (pattern, false) - }; - - // Simple wildcard matching - let parts: Vec<&str> = pattern.split('*').collect(); - - if parts.len() == 1 { - // No wildcards - exact match (or prefix if no $) - if must_end { - return path == pattern; - } else { - return path.starts_with(pattern); - } - } - - // Check if path matches the pattern with wildcards - let mut path_pos = 0; - for (i, part) in parts.iter().enumerate() { - if part.is_empty() { - continue; - } - - if i == 0 { - // First part must match from the start - if !path[path_pos..].starts_with(part) { - return false; - } - path_pos += part.len(); - } else { - // Find the next occurrence - if let Some(pos) = path[path_pos..].find(part) { - path_pos += pos + part.len(); - } else { - return false; - } - } - } - - // If must_end is true, ensure we've consumed the entire path - !must_end || path_pos == path.len() - } - - /// Evaluate TDM rules and find the matching rule for a URL - async fn evaluate_tdm_policy(&self, url: &str, rules: Vec) -> Option { - // Parse URL to get the path - let parsed = Url::parse(url).ok()?; - let path = parsed.path(); - - // Find the first matching rule (first-match wins per W3C TDMRep) - let matched_rule = rules - .iter() - .find(|rule| Self::match_tdm_pattern(&rule.location, path)) - .cloned(); - - let is_reserved = matched_rule - .as_ref() - .map(|r| r.tdm_reservation == 1) - .unwrap_or(false); - - Some(TdmPolicy { - rules, - matched_rule, - is_reserved, - }) - } - - /// Analyze each AI bot individually to determine its access status - fn analyze_ai_bots(&self, content: &str, url: &str) -> Vec { - let all_bots = AICrawler::get_all(); - let mut results = Vec::new(); - - // Normalize user agents for comparison (case-insensitive) - let user_agents_lower: Vec = self - .extract_user_agents(content) - .iter() - .map(|ua| ua.to_lowercase()) - .collect(); - - for bot in all_bots { - let bot_name_lower = bot.name.to_lowercase(); - - // Check if this bot is mentioned in robots.txt - let is_mentioned = user_agents_lower - .iter() - .any(|ua| ua == &bot_name_lower || ua.contains(&bot_name_lower)); - - let status = if is_mentioned { - // Bot is mentioned - check if it's allowed or blocked for this path - // texting_robots::Robot bakes the user-agent into its parsed state, - // so we must re-parse for each bot to get per-bot allow/disallow results. - match Robot::new(&bot.name, content.as_bytes()) { - Ok(robot) => { - if robot.allowed(url) { - BotStatus::Allowed - } else { - BotStatus::Blocked - } - } - Err(_) => BotStatus::Allowed, // Parse error, default to allowed - } - } else { - // Bot not mentioned - allowed by default (follows wildcard rules or no restrictions) - BotStatus::Allowed - }; - - results.push(BotAnalysisResult { - bot_name: bot.name, - company: bot.company, - category: format!("{:?}", bot.category), - status, - }); - } - - results - } -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn test_extract_user_agents() { - let analyzer = RobotAnalyzer::new("TestBot".to_string()); - let content = "User-agent: *\nUser-agent: GoogleBot\nUser-agent: BingBot"; - let agents = analyzer.extract_user_agents(content); - - assert_eq!(agents.len(), 3); - assert!(agents.contains(&"*".to_string())); - assert!(agents.contains(&"GoogleBot".to_string())); - assert!(agents.contains(&"BingBot".to_string())); - } - - #[test] - fn test_extract_paths() { - let analyzer = RobotAnalyzer::new("*".to_string()); - let content = "Allow: /public\nDisallow: /private\nDisallow: /admin"; - let (allowed, disallowed) = analyzer.extract_paths(content); - - assert_eq!(allowed.len(), 1); - assert_eq!(allowed[0], "/public"); - assert_eq!(disallowed.len(), 2); - assert!(disallowed.contains(&"/private".to_string())); - assert!(disallowed.contains(&"/admin".to_string())); - } - - #[test] - fn test_extract_global_licenses() { - let analyzer = RobotAnalyzer::new("*".to_string()); - let content = "License: https://example.com/license.xml\nUser-agent: *\nDisallow: /private"; - let (global, group) = analyzer.extract_licenses(content); - - assert_eq!(global.len(), 1); - assert_eq!(global[0], "https://example.com/license.xml"); - assert_eq!(group.len(), 0); - } - - #[test] - fn test_extract_group_scoped_licenses() { - let analyzer = RobotAnalyzer::new("GPTBot".to_string()); - let content = r#" -License: https://example.com/global.xml -User-agent: GPTBot -License: https://example.com/gptbot.xml -Disallow: / - "#; - let (global, group) = analyzer.extract_licenses(content); - - assert_eq!(global.len(), 1); - assert_eq!(global[0], "https://example.com/global.xml"); - assert_eq!(group.len(), 1); - assert_eq!(group[0], "https://example.com/gptbot.xml"); - } - - #[test] - fn test_license_precedence_group_overrides_global() { - let analyzer = RobotAnalyzer::new("GPTBot".to_string()); - let content = r#" -License: https://example.com/global.xml -User-agent: GPTBot -License: https://example.com/gptbot.xml - "#; - let (global, group) = analyzer.extract_licenses(content); - - // Active licenses should be group-scoped when present - let active = if !group.is_empty() { &group } else { &global }; - assert_eq!(active.len(), 1); - assert_eq!(active[0], "https://example.com/gptbot.xml"); - } - - #[test] - fn test_license_requires_absolute_uri() { - let analyzer = RobotAnalyzer::new("*".to_string()); - let content = "License: /relative/path.xml\nLicense: https://example.com/absolute.xml"; - let (global, _) = analyzer.extract_licenses(content); - - // Should only include absolute URIs - assert_eq!(global.len(), 1); - assert_eq!(global[0], "https://example.com/absolute.xml"); - } - - #[test] - fn test_license_ignores_comments() { - let analyzer = RobotAnalyzer::new("*".to_string()); - let content = r#" -# This is a comment with License: https://fake.com/license.xml -License: https://example.com/real.xml - "#; - let (global, _) = analyzer.extract_licenses(content); - - assert_eq!(global.len(), 1); - assert_eq!(global[0], "https://example.com/real.xml"); - } - - #[test] - fn test_wildcard_user_agent_matches() { - let analyzer = RobotAnalyzer::new("MyBot".to_string()); - let content = r#" -User-agent: * -License: https://example.com/wildcard.xml - "#; - let (_, group) = analyzer.extract_licenses(content); - - // Wildcard should match any user agent - assert_eq!(group.len(), 1); - assert_eq!(group[0], "https://example.com/wildcard.xml"); - } - - #[test] - fn test_content_signals_basic() { - let analyzer = RobotAnalyzer::new("*".to_string()); - let content = r#" -User-agent: * -Content-Signal: search=yes, ai-train=no, ai-input=yes -Allow: / - "#; - let (search, ai_input, ai_train) = analyzer.extract_content_signals(content); - - assert_eq!(search, Some("yes".to_string())); - assert_eq!(ai_input, Some("yes".to_string())); - assert_eq!(ai_train, Some("no".to_string())); - } - - #[test] - fn test_content_signals_partial() { - let analyzer = RobotAnalyzer::new("*".to_string()); - let content = r#" -User-agent: * -Content-Signal: search=yes, ai-train=no -Allow: / - "#; - let (search, ai_input, ai_train) = analyzer.extract_content_signals(content); - - assert_eq!(search, Some("yes".to_string())); - assert_eq!(ai_input, None); - assert_eq!(ai_train, Some("no".to_string())); - } - - #[test] - fn test_content_signals_group_scoped() { - let analyzer = RobotAnalyzer::new("GPTBot".to_string()); - let content = r#" -User-agent: Googlebot -Content-Signal: search=yes, ai-train=yes - -User-agent: GPTBot -Content-Signal: search=yes, ai-train=no, ai-input=no -Disallow: / - "#; - let (search, ai_input, ai_train) = analyzer.extract_content_signals(content); - - // Should only get signals from GPTBot group - assert_eq!(search, Some("yes".to_string())); - assert_eq!(ai_input, Some("no".to_string())); - assert_eq!(ai_train, Some("no".to_string())); - } - - #[test] - fn test_content_signals_cloudflare_format() { - // Test the exact format used by Cloudflare's managed robots.txt - let analyzer = RobotAnalyzer::new("*".to_string()); - let content = r#" -User-Agent: * -Content-Signal: search=yes, ai-train=no -Allow: / - "#; - let (search, ai_input, ai_train) = analyzer.extract_content_signals(content); - - assert_eq!(search, Some("yes".to_string())); - assert_eq!(ai_input, None, "ai-input not specified"); - assert_eq!(ai_train, Some("no".to_string())); - } - - #[test] - fn test_separate_user_agent_groups_dont_mix_licenses() { - let analyzer = RobotAnalyzer::new("GPTBot".to_string()); - let content = r#" -User-agent: Googlebot -License: https://example.com/google-only.xml -Disallow: /admin - -User-agent: GPTBot -License: https://example.com/gpt-only.xml -Disallow: / - "#; - let (global, group) = analyzer.extract_licenses(content); - - // Should not collect licenses from other user-agent groups - assert_eq!(global.len(), 0, "Should have no global licenses"); - assert_eq!(group.len(), 1, "Should have exactly 1 group license"); - assert_eq!(group[0], "https://example.com/gpt-only.xml"); - assert!( - !group.contains(&"https://example.com/google-only.xml".to_string()), - "Should NOT include Googlebot's license" - ); - } - - #[test] - fn test_rsl_real_world_rslstandard_org() { - // Based on rslstandard.org's robots.txt (as of 2026-02-14) - // Tests global license pattern used by RSL Standard's own site - let analyzer = RobotAnalyzer::new("*".to_string()); - let content = r#" -License: https://rslcollective.org/royalty.xml - -User-agent: * -Disallow: - "#; - let (global, group) = analyzer.extract_licenses(content); - - assert_eq!(global.len(), 1); - assert_eq!(global[0], "https://rslcollective.org/royalty.xml"); - assert_eq!(group.len(), 0, "No group-scoped licenses"); - - // Active licenses should be global - let active = if !group.is_empty() { &group } else { &global }; - assert_eq!(active[0], "https://rslcollective.org/royalty.xml"); - } - - #[test] - fn test_rsl_real_world_medium_com() { - // Based on medium.com's robots.txt (as of 2026-02-14) - // Tests group-scoped license pattern with multiple specific bots - let analyzer = RobotAnalyzer::new("GPTBot".to_string()); - let content = r#" -User-agent: * -Allow: /about - -User-agent: GPTBot -User-agent: ClaudeBot -User-agent: FacebookBot -License: https://medium.com/license.xml -Disallow: / - "#; - let (global, group) = analyzer.extract_licenses(content); - - assert_eq!(global.len(), 0, "No global licenses"); - assert_eq!(group.len(), 1, "Should have group license"); - assert_eq!(group[0], "https://medium.com/license.xml"); - - // Active licenses should be group-scoped - let active = if !group.is_empty() { &group } else { &global }; - assert_eq!(active[0], "https://medium.com/license.xml"); - } - - #[test] - fn test_tdm_pattern_exact_match() { - assert!(RobotAnalyzer::match_tdm_pattern("/", "/")); - assert!(RobotAnalyzer::match_tdm_pattern("/docs", "/docs")); - assert!(RobotAnalyzer::match_tdm_pattern("/docs", "/docs/page")); - assert!(!RobotAnalyzer::match_tdm_pattern("/docs$", "/docs/page")); - } - - #[test] - fn test_tdm_pattern_wildcard() { - assert!(RobotAnalyzer::match_tdm_pattern("/docs/*", "/docs/page")); - assert!(RobotAnalyzer::match_tdm_pattern( - "/docs/*", - "/docs/page/sub" - )); - assert!(RobotAnalyzer::match_tdm_pattern("*.pdf", "/file.pdf")); - assert!(RobotAnalyzer::match_tdm_pattern("*.pdf", "/docs/file.pdf")); - assert!(!RobotAnalyzer::match_tdm_pattern("/docs/*", "/other/page")); - } - - #[test] - fn test_tdm_pattern_end_marker() { - assert!(RobotAnalyzer::match_tdm_pattern("/docs$", "/docs")); - assert!(!RobotAnalyzer::match_tdm_pattern("/docs$", "/docs/")); - assert!(!RobotAnalyzer::match_tdm_pattern("/docs$", "/docs/page")); - assert!(RobotAnalyzer::match_tdm_pattern( - "/docs/page$", - "/docs/page" - )); - } - - #[test] - fn test_tdm_pattern_complex() { - assert!(RobotAnalyzer::match_tdm_pattern( - "/*/public/*", - "/docs/public/file" - )); - assert!(RobotAnalyzer::match_tdm_pattern( - "/docs/*.pdf$", - "/docs/file.pdf" - )); - assert!(!RobotAnalyzer::match_tdm_pattern( - "/docs/*.pdf$", - "/docs/file.pdf.bak" - )); - } - - #[tokio::test] - async fn test_tdm_rule_matching() { - let analyzer = RobotAnalyzer::new("*".to_string()); - - let rules = vec![ - TdmRule { - location: "/".to_string(), - tdm_reservation: 1, - tdm_policy: Some("https://example.com/policy.html".to_string()), - }, - TdmRule { - location: "/public/*".to_string(), - tdm_reservation: 0, - tdm_policy: None, - }, - ]; - - // Test root path matches first rule - let policy = analyzer - .evaluate_tdm_policy("https://example.com/", rules.clone()) - .await; - assert!(policy.is_some()); - let policy = policy.unwrap(); - assert!(policy.is_reserved); - assert_eq!(policy.matched_rule.as_ref().unwrap().location, "/"); - - // Test public path matches second rule - let rules2 = vec![ - TdmRule { - location: "/public/*".to_string(), - tdm_reservation: 0, - tdm_policy: None, - }, - TdmRule { - location: "/".to_string(), - tdm_reservation: 1, - tdm_policy: Some("https://example.com/policy.html".to_string()), - }, - ]; - - let policy2 = analyzer - .evaluate_tdm_policy("https://example.com/public/data", rules2) - .await; - assert!(policy2.is_some()); - let policy2 = policy2.unwrap(); - assert!(!policy2.is_reserved); - assert_eq!(policy2.matched_rule.as_ref().unwrap().location, "/public/*"); - } - - #[test] - fn test_analyze_ai_bots_all_blocked() { - let analyzer = RobotAnalyzer::new("*".to_string()); - let content = "User-agent: *\nDisallow: /\n"; - let results = analyzer.analyze_ai_bots(content, "https://www.nytimes.com/"); - assert_eq!(results.len(), 26); - // Wildcard disallow blocks all bots via Robot parsing, but only bots - // that are "mentioned" get blocked status. Unmentioned bots default to Allowed. - // With User-agent: *, all bots match via wildcard in texting_robots. - } - - #[test] - fn test_analyze_ai_bots_all_allowed() { - let analyzer = RobotAnalyzer::new("*".to_string()); - let content = "User-agent: *\nAllow: /\n"; - let results = analyzer.analyze_ai_bots(content, "https://github.com/"); - assert_eq!(results.len(), 26); - for bot in &results { - assert!( - matches!(bot.status, BotStatus::Allowed), - "{} should be allowed", - bot.bot_name - ); - } - } - - #[test] - fn test_analyze_ai_bots_returns_26() { - let analyzer = RobotAnalyzer::new("*".to_string()); - let results = analyzer.analyze_ai_bots("", "https://github.com/"); - assert_eq!(results.len(), 26); - } - - #[test] - fn test_analyze_ai_bots_selective_blocking() { - let analyzer = RobotAnalyzer::new("*".to_string()); - let content = "\ -User-agent: GPTBot\nDisallow: /\n\n\ -User-agent: ClaudeBot\nDisallow: /\n\n\ -User-agent: *\nAllow: /\n"; - let results = analyzer.analyze_ai_bots(content, "https://techcrunch.com/"); - - let gptbot = results.iter().find(|b| b.bot_name == "GPTBot").unwrap(); - assert!(matches!(gptbot.status, BotStatus::Blocked)); - - let claudebot = results.iter().find(|b| b.bot_name == "ClaudeBot").unwrap(); - assert!(matches!(claudebot.status, BotStatus::Blocked)); - - let perplexity = results - .iter() - .find(|b| b.bot_name == "PerplexityBot") - .unwrap(); - assert!(matches!(perplexity.status, BotStatus::Allowed)); - } - - #[test] - fn test_read_csv_url_column_detection() { - let dir = tempfile::tempdir().unwrap(); - let csv_path = dir.path().join("test.csv"); - std::fs::write( - &csv_path, - "name,Company URL,notes\nNYT,https://www.nytimes.com,news\n", - ) - .unwrap(); - let analyzer = RobotAnalyzer::new("*".to_string()); - let urls = analyzer.read_csv(&csv_path).unwrap(); - assert_eq!(urls, vec!["https://www.nytimes.com"]); - } - - #[test] - fn test_read_csv_adds_https_prefix() { - let dir = tempfile::tempdir().unwrap(); - let csv_path = dir.path().join("test.csv"); - std::fs::write(&csv_path, "url\ngithub.com\n").unwrap(); - let analyzer = RobotAnalyzer::new("*".to_string()); - let urls = analyzer.read_csv(&csv_path).unwrap(); - assert_eq!(urls, vec!["https://github.com"]); - } - - #[test] - fn test_read_csv_skips_empty_rows() { - let dir = tempfile::tempdir().unwrap(); - let csv_path = dir.path().join("test.csv"); - std::fs::write( - &csv_path, - "url\nhttps://github.com\n\n \nhttps://www.nytimes.com\n", - ) - .unwrap(); - let analyzer = RobotAnalyzer::new("*".to_string()); - let urls = analyzer.read_csv(&csv_path).unwrap(); - assert_eq!(urls, vec!["https://github.com", "https://www.nytimes.com"]); - } -}