diff --git a/.agents/mcp_config.json b/.agents/mcp_config.json new file mode 100644 index 0000000..aae89f6 --- /dev/null +++ b/.agents/mcp_config.json @@ -0,0 +1,10 @@ +{ + "mcpServers": { + "seolens": { + "command": "seolens", + "args": [ + "mcp" + ] + } + } +} diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..4b1fb94 --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,60 @@ +name: CI + +on: + push: + pull_request: + +concurrency: + group: ${{ github.workflow }}-${{ github.head_ref || github.ref }} + cancel-in-progress: true + +permissions: + contents: read + +jobs: + lint: + name: Code Formatting & Clippy Lints + runs-on: ubuntu-latest + steps: + - name: Checkout repository + uses: actions/checkout@v4 + + - name: Install Rust toolchain + uses: dtolnay/rust-toolchain@stable + with: + components: rustfmt, clippy + + - name: Setup Rust caching + uses: Swatinem/rust-cache@v2 + + - name: Check code formatting + run: cargo fmt --all -- --check + + - name: Run Clippy lints + run: cargo clippy --workspace --all-targets -- -D warnings + + test: + name: Test Suite (${{ matrix.os }}) + runs-on: ${{ matrix.os }} + strategy: + fail-fast: false + matrix: + os: [ubuntu-latest, macos-latest, windows-latest] + steps: + - name: Checkout repository + uses: actions/checkout@v4 + + - name: Install Rust toolchain + uses: dtolnay/rust-toolchain@stable + + - name: Setup Rust caching + uses: Swatinem/rust-cache@v2 + + - name: Install cargo-nextest + uses: taiki-e/install-action@nextest + + - name: Run unit & integration tests + run: cargo nextest run --workspace + + - name: Run documentation tests + run: cargo test --doc diff --git a/.gitignore b/.gitignore index 9f555fb..2f044c8 100644 --- a/.gitignore +++ b/.gitignore @@ -21,6 +21,5 @@ Thumbs.db .vscode/ *.swp -# Local inspirations and blueprints -inspiration/ -ARCHITECTURE_BLUEPRINT.md \ No newline at end of file +# Local inspirations +inspiration/ \ No newline at end of file diff --git a/AGENTS.md b/AGENTS.md index f0b68df..56b3839 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -30,7 +30,7 @@ The repository is structured as a **2-Member Cargo Workspace**: ## Working Principles - **Strict Test-Driven Development (TDD)**: Always write failing automated tests in `tests/` before writing minimal implementation code in `src/`. -- **Micro-Phase Progression**: Follow the 13 micro-phases defined in `docs/PRODUCTION_SPEC.md` sequentially. Complete and verify each phase before moving to the next. +- **Architectural Progression**: Follow the architecture and roadmap defined in [`docs/architecture.md`](docs/architecture.md). Complete and verify each component before moving to the next. - **Zero Panics in Library Code**: Never use `unwrap()` or `expect()` in `src/` library modules. All fallible operations must return `Result` using `thiserror`. - **Minimal, Targeted Changes**: Make focused edits. Preserve existing comments, docstrings, and unrelated code. - **Empirical Measurement**: Do not make up unverified performance claims or benchmark figures. Profile and measure memory and speed empirically. @@ -52,7 +52,7 @@ The repository is structured as a **2-Member Cargo Workspace**: - **Mandatory Formatting**: After making any code changes and before handing over to the user, **always run `cargo fmt --all`**. - **Test Runner Preference**: If `cargo-nextest` is installed on the system (check via `cargo nextest --version`), always prioritize using `cargo nextest run` (or `cargo nextest run --workspace`) for running unit and integration tests because it is faster and provides superior UI output. Fall back to standard `cargo test` if `nextest` is unavailable. Note that doc tests are executed with `cargo test --doc`. - **Minor / Trivial Tasks**: Run `cargo fmt --all`, `cargo check`, and `cargo clippy`. -- **Phase Work & Major Features**: Run `cargo fmt --all`, run the test suite (preferring `cargo nextest run` if installed, otherwise `cargo test`), and execute the manual verification gate defined in `docs/PRODUCTION_SPEC.md`. +- **Phase Work & Major Features**: Run `cargo fmt --all`, run the test suite (preferring `cargo nextest run` if installed, otherwise `cargo test`), and verify changes against [`docs/architecture.md`](docs/architecture.md). - **Super Simple Edits** (e.g. typos, comments): Run `cargo fmt --all`. --- @@ -74,13 +74,12 @@ The repository is structured as a **2-Member Cargo Workspace**: Read the relevant specification in `docs/` before implementing or changing any component: -1. **Roadmap, Architecture & TDD Protocol**: [`docs/PRODUCTION_SPEC.md`](file:///mnt/Code/PROJECTS/seo-lens/docs/PRODUCTION_SPEC.md) -2. **120 SEO Rules Catalog (Heuristics & Fixes)**: [`docs/SEO_RULES_CATALOG.md`](file:///mnt/Code/PROJECTS/seo-lens/docs/SEO_RULES_CATALOG.md) -3. **Domain Models & SQLite WAL Schema**: [`docs/DATA_MODELS_AND_SCHEMA.md`](file:///mnt/Code/PROJECTS/seo-lens/docs/DATA_MODELS_AND_SCHEMA.md) -4. **Native Desktop Application (Tauri v2 + React 19)**: [`docs/DESKTOP_APP_SPEC.md`](file:///mnt/Code/PROJECTS/seo-lens/docs/DESKTOP_APP_SPEC.md) -5. **Model Context Protocol (MCP) Server**: [`docs/MCP_SPECIFICATION.md`](file:///mnt/Code/PROJECTS/seo-lens/docs/MCP_SPECIFICATION.md) -6. **CLI Commands, Flags & Exporters**: [`docs/CLI_AND_REPORTS_SPEC.md`](file:///mnt/Code/PROJECTS/seo-lens/docs/CLI_AND_REPORTS_SPEC.md) -7. **Crawler Architecture & Congestion Control**: [`docs/CRAWLER_SPEC.md`](file:///mnt/Code/PROJECTS/seo-lens/docs/CRAWLER_SPEC.md) +1. **Roadmap & Architecture**: [`docs/architecture.md`](docs/architecture.md) +2. **120 SEO Rules Catalog (Heuristics & Fixes)**: [`docs/rules.md`](docs/rules.md) +3. **Storage & SQLite Architecture**: [`docs/storage.md`](docs/storage.md) +4. **Model Context Protocol (MCP) Server**: [`docs/mcp.md`](docs/mcp.md) +5. **CLI Commands, Flags & Exporters**: [`docs/cli.md`](docs/cli.md) +6. **Crawler Architecture & Congestion Control**: [`docs/crawler.md`](docs/crawler.md) --- diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md new file mode 100644 index 0000000..62f85d0 --- /dev/null +++ b/CONTRIBUTING.md @@ -0,0 +1,175 @@ +# Contributing to SEO Lens + +Thank you for your interest in contributing to **SEO Lens**! We welcome bug reports, feature requests, documentation improvements, and code contributions from developers of all experience levels. + +--- + +## 1. Code of Conduct + +We are committed to providing a welcoming, inclusive, and harassment-free environment. Please be respectful and constructive in all interactions, issues, and pull requests. + +--- + +## 2. Getting Started + +### Prerequisites + +1. **Rust Toolchain (1.80+)**: + + - **Linux & macOS**: + + ```bash + curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh + source "$HOME/.cargo/env" + ``` + + - **Windows**: Download and run the official installer from [rustup.rs](https://rustup.rs/). + - **Existing Installations**: Verify or update to the latest stable release: + + ```bash + rustup update stable + ``` + +2. **C Build Tools & OpenSSL** (Required to compile bundled SQLite and OpenSSL dependencies): + + - **Ubuntu / Debian**: + + ```bash + sudo apt update && sudo apt install -y build-essential pkg-config libssl-dev + ``` + + - **Fedora / RHEL**: + + ```bash + sudo dnf install -y gcc pkg-config openssl-devel + ``` + + - **macOS**: + + ```bash + xcode-select --install + ``` + + - **Windows**: Install the **Desktop development with C++** workload via the Visual Studio Installer. + +3. **Cargo Nextest (Recommended)**: + Used for fast parallel integration testing: + + ```bash + cargo install cargo-nextest --locked + ``` + +4. **Git** + +### Setting Up Your Local Repository + +```bash +# 1. Fork and clone the repository +git clone https://github.com/Shantodotdev/seo-lens.git +cd seo-lens + +# 2. Verify that everything builds and tests pass +cargo check +cargo nextest run # or 'cargo test' +``` + +--- + +## 3. Development Workflow + +1. **Create a branch**: + + ```bash + git checkout -b feat/my-feature + # or + git checkout -b fix/issue-description + ``` + +2. **Follow Test-Driven Development (TDD)**: + - Write failing automated tests in `tests/` before writing minimal implementation code in `src/`. + +3. **Format and lint before committing**: + + ```bash + # Format all Rust files + cargo fmt --all + + # Check for compiler and clippy warnings + cargo check + cargo clippy -- -D warnings + + # Run the full test suite + cargo nextest run + ``` + +4. **Commit using Conventional Commits**: + - `feat(crawler): add support for Brotli compression` + - `fix(parser): handle malformed self-closing tags` + - `docs(rules): clarify remediation advice for canonical loops` + - `test(graph): add cyclic redirect test fixture` + +--- + +## 4. How to Add a New SEO Rule + +SEO Lens has a modular rules engine. Adding a new rule takes just 4 steps: + +### Step 1: Register the Rule in the Catalog (`src/rules/catalog.rs`) + +Add your strongly typed `RuleId` and definition with severity, category, description, and fix advice: + +```rust +// In src/rules/catalog.rs: +RuleDefinition { + id: RuleId::WarnCustomCheck, + category: IssueCategory::Links, + severity: Severity::Warning, + title: "Custom link defect detected", + description: "Explanation of why this defect harms technical SEO.", + fix_advice: "Actionable instructions for the developer on how to fix it.", +} +``` + +### Step 2: Implement the Evaluation Logic + +- **In-Flight Document Rule**: Implement in `src/rules/page/` (e.g. `src/rules/page/links.rs`). It evaluates a single `ParsedPage` streamingly during crawling. +- **Site-Wide Graph Rule**: Implement in `src/rules/graph/` (e.g. `src/rules/graph/orphans.rs`). It evaluates the `petgraph` site graph post-crawl. + +### Step 3: Write Integration Tests (`tests/rules_tests.rs`) + +Add a synthetic HTML fixture in `tests/fixtures/` and verify that your rule triggers as expected: + +```rust +#[tokio::test] +async fn test_custom_rule_detection() { + let report = parse_and_audit("tests/fixtures/sample_page.html").await; + assert!(report.issues.iter().any(|i| i.code == "WARN_CUSTOM_CHECK")); +} +``` + +### Step 4: Document the Rule in `docs/rules.md` + +Add your rule code, severity, detection heuristic, and fix instructions under the relevant category in [`docs/rules.md`](./docs/rules.md). + +--- + +## 5. Coding Standards & Conventions + +- **Zero Panics in Library Code**: Never use `unwrap()` or `expect()` in `src/` library modules. Return `Result` using `thiserror`. +- **Memory Optimization**: + - Use `compact_str::CompactString` for strings $\le 24$ bytes (URLs, tags, MIME types) to keep memory stack-inlined. + - Use `bitflags` for boolean flags and robots directives. + - Parse HTML streamingly with `lol_html` rather than constructing large in-memory DOMs. +- **Documentation**: All `pub` structs, traits, and functions must have concise doc comments (`///`). +- **Relative Links**: Always use relative paths when linking files in markdown documents. + +--- + +## 6. Pull Request Guidelines + +Before submitting your PR, verify: + +- [ ] `cargo fmt --all` produces no diffs. +- [ ] `cargo clippy -- -D warnings` reports zero warnings. +- [ ] `cargo nextest run` (or `cargo test`) passes completely. +- [ ] Any new public APIs or CLI flags are documented in `docs/`. diff --git a/Cargo.lock b/Cargo.lock index d8b4f85..3750f93 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -345,6 +345,27 @@ dependencies = [ "syn 2.0.119", ] +[[package]] +name = "csv" +version = "1.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "52cd9d68cf7efc6ddfaaee42e7288d3a99d613d4b50f76ce9827ae0c6e14f938" +dependencies = [ + "csv-core", + "itoa", + "ryu", + "serde_core", +] + +[[package]] +name = "csv-core" +version = "0.1.13" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "704a3c26996a80471189265814dbc2c257598b96b8a7feae2d31ace646bb9782" +dependencies = [ + "memchr", +] + [[package]] name = "deadpool" version = "0.12.3" @@ -384,6 +405,27 @@ dependencies = [ "syn 2.0.119", ] +[[package]] +name = "dirs" +version = "6.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c3e8aa94d75141228480295a7d0e7feb620b1a5ad9f12bc40be62411e38cce4e" +dependencies = [ + "dirs-sys", +] + +[[package]] +name = "dirs-sys" +version = "0.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e01a3366d27ee9890022452ee61b2b63a67e6f13f58900b651ff5665f0bb1fab" +dependencies = [ + "libc", + "option-ext", + "redox_users", + "windows-sys 0.61.2", +] + [[package]] name = "displaydoc" version = "0.2.7" @@ -980,6 +1022,15 @@ version = "0.2.189" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2" +[[package]] +name = "libredox" +version = "0.1.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8d8f1ea3f21fd3405dcaf6c9b5c1630af9afc422d9073ea39c5f6d6c772e08ed" +dependencies = [ + "libc", +] + [[package]] name = "libsqlite3-sys" version = "0.31.0" @@ -1176,6 +1227,12 @@ dependencies = [ "vcpkg", ] +[[package]] +name = "option-ext" +version = "0.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "04744f49eae99ab78e0d5c0b603ab218f515ea8cfe5a456d7629ad883a3b6e7d" + [[package]] name = "parking_lot" version = "0.12.5" @@ -1345,6 +1402,17 @@ dependencies = [ "bitflags", ] +[[package]] +name = "redox_users" +version = "0.5.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a4e608c6638b9c18977b00b475ac1f28d14e84b27d8d42f70e0bf1e3dec127ac" +dependencies = [ + "getrandom 0.2.17", + "libredox", + "thiserror", +] + [[package]] name = "regex" version = "1.13.1" @@ -1587,6 +1655,8 @@ dependencies = [ "bitflags", "clap", "compact_str", + "csv", + "dirs", "flate2", "hashbrown 0.15.5", "lol_html", diff --git a/Cargo.toml b/Cargo.toml index c67c783..5210989 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -12,6 +12,8 @@ edition = "2021" authors = ["KR Shanto"] description = "High-performance, local-first website crawler and technical SEO audit engine in Rust" license = "MIT OR Apache-2.0" +repository = "https://github.com/Shantodotdev/seo-lens" +homepage = "https://github.com/Shantodotdev/seo-lens" [lib] name = "seo_lens" @@ -29,7 +31,7 @@ thiserror = "2.0" anyhow = "1.0" tracing = "0.1" tracing-subscriber = { version = "0.3", features = ["env-filter"] } -clap = { version = "4.5", features = ["derive"] } +clap = { version = "4.5", features = ["derive", "color"] } compact_str = { version = "0.8", features = ["serde"] } bitflags = { version = "2.8", features = ["serde"] } url = "2.5" @@ -42,6 +44,8 @@ flate2 = "1.0" petgraph = "0.8" regex = "1.11" rusqlite = { version = "0.33", features = ["bundled"] } +dirs = "6.0" +csv = "1.3" [dev-dependencies] diff --git a/LICENSE-APACHE b/LICENSE-APACHE new file mode 100644 index 0000000..dc93d05 --- /dev/null +++ b/LICENSE-APACHE @@ -0,0 +1,201 @@ + Apache License + Version 2.0, January 2004 + http://www.apache.org/licenses/ + + TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION + + 1. Definitions. + + "License" shall mean the terms and conditions for use, reproduction, + and distribution as defined by Sections 1 through 9 of this document. + + "Licensor" shall mean the copyright owner or entity authorized by + the copyright owner that is granting the License. + + "Legal Entity" shall mean the union of the acting entity and all + other entities that control, are controlled by, or are under common + control with that entity. For the purposes of this definition, + "control" means (i) the power, direct or indirect, to cause the + direction or management of such entity, whether by contract or + otherwise, or (ii) ownership of fifty percent (50%) or more of the + outstanding shares, or (iii) beneficial ownership of such entity. + + "You" (or "Your") shall mean an individual or Legal Entity + exercising permissions granted by this License. + + "Source" form shall mean the preferred form for making modifications, + including but not limited to software source code, documentation + source, and configuration files. + + "Object" form shall mean any form resulting from mechanical + transformation or translation of a Source form, including but + not limited to compiled object code, generated documentation, + and conversions to other media types. + + "Work" shall mean the work of authorship, whether in Source or + Object form, made available under the License, as indicated by a + copyright notice that is included in or attached to the work + (an example is provided in the Appendix below). + + "Derivative Works" shall mean any work, whether in Source or Object + form, that is based on (or derived from) the Work and for which the + editorial revisions, annotations, elaborations, or other modifications + represent, as a whole, an original work of authorship. For the purposes + of this License, Derivative Works shall not include works that remain + separable from, or merely link (or bind by name) to the interfaces of, + the Work and Derivative Works thereof. + + "Contribution" shall mean any work of authorship, including + the original version of the Work and any modifications or additions + to that Work or Derivative Works thereof, that is intentionally + submitted to Licensor for inclusion in the Work by the copyright owner + or by an individual or Legal Entity authorized to submit on behalf of + the copyright owner. For the purposes of this definition, "submitted" + means any form of electronic, verbal, or written communication sent + to the Licensor or its representatives, including but not limited to + communication on electronic mailing lists, source code control systems, + and issue tracking systems that are managed by, or on behalf of, the + Licensor for the purpose of discussing and improving the Work, but + excluding communication that is conspicuously marked or otherwise + designated in writing by the copyright owner as "Not a Contribution." + + "Contributor" shall mean Licensor and any individual or Legal Entity + on behalf of whom a Contribution has been received by Licensor and + subsequently incorporated within the Work. + + 2. Grant of Copyright License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + copyright license to reproduce, prepare Derivative Works of, + publicly display, publicly perform, sublicense, and distribute the + Work and such Derivative Works in Source or Object form. + + 3. Grant of Patent License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + (except as stated in this section) patent license to make, have made, + use, offer to sell, sell, import, and otherwise transfer the Work, + where such license applies only to those patent claims licensable + by such Contributor that are necessarily infringed by their + Contribution(s) alone or by combination of their Contribution(s) + with the Work to which such Contribution(s) was submitted. If You + institute patent litigation against any entity (including a + cross-claim or counterclaim in a lawsuit) alleging that the Work + or a Contribution incorporated within the Work constitutes direct + or contributory patent infringement, then any patent licenses + granted to You under this License for that Work shall terminate + as of the date such litigation is filed. + + 4. Redistribution. You may reproduce and distribute copies of the + Work or Derivative Works thereof in any medium, with or without + modifications, and in Source or Object form, provided that You + meet the following conditions: + + (a) You must give any other recipients of the Work or + Derivative Works a copy of this License; and + + (b) You must cause any modified files to carry prominent notices + stating that You changed the files; and + + (c) You must retain, in the Source form of any Derivative Works + that You distribute, all copyright, patent, trademark, and + attribution notices from the Source form of the Work, + excluding those notices that do not pertain to any part of + the Derivative Works; and + + (d) If the Work includes a "NOTICE" text file as part of its + distribution, then any Derivative Works that You distribute must + include a readable copy of the attribution notices contained + within such NOTICE file, excluding those notices that do not + pertain to any part of the Derivative Works, in at least one + of the following places: within a NOTICE text file distributed + as part of the Derivative Works; within the Source form or + documentation, if provided along with the Derivative Works; or, + within a display generated by the Derivative Works, if and + wherever such third-party notices normally appear. The contents + of the NOTICE file are for informational purposes only and + do not modify the License. You may add Your own attribution + notices within Derivative Works that You distribute, alongside + or as an addendum to the NOTICE text from the Work, provided + that such additional attribution notices cannot be construed + as modifying the License. + + You may add Your own copyright statement to Your modifications and + may provide additional or different license terms and conditions + for use, reproduction, or distribution of Your modifications, or + for any such Derivative Works as a whole, provided Your use, + reproduction, and distribution of the Work otherwise complies with + the conditions stated in this License. + + 5. Submission of Contributions. Unless You explicitly state otherwise, + any Contribution intentionally submitted for inclusion in the Work + by You to the Licensor shall be under the terms and conditions of + this License, without any additional terms or conditions. + Notwithstanding the above, nothing herein shall supersede or modify + the terms of any separate license agreement you may have executed + with Licensor regarding such Contributions. + + 6. Trademarks. This License does not grant permission to use the trade + names, trademarks, service marks, or product names of the Licensor, + except as required for reasonable and customary use in describing the + origin of the Work and reproducing the content of the NOTICE file. + + 7. Disclaimer of Warranty. Unless required by applicable law or + agreed to in writing, Licensor provides the Work (and each + Contributor provides its Contributions) on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or + implied, including, without limitation, any warranties or conditions + of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A + PARTICULAR PURPOSE. You are solely responsible for determining the + appropriateness of using or redistributing the Work and assume any + risks associated with Your exercise of permissions under this License. + + 8. Limitation of Liability. In no event and under no legal theory, + whether in tort (including negligence), contract, or otherwise, + unless required by applicable law (such as deliberate and grossly + negligent acts) or agreed to in writing, shall any Contributor be + liable to You for damages, including any direct, indirect, special, + incidental, or consequential damages of any character arising as a + result of this License or out of the use or inability to use the + Work (including but not limited to damages for loss of goodwill, + work stoppage, computer failure or malfunction, or any and all + other commercial damages or losses), even if such Contributor + has been advised of the possibility of such damages. + + 9. Accepting Warranty or Additional Liability. While redistributing + the Work or Derivative Works thereof, You may choose to offer, + and charge a fee for, acceptance of support, warranty, indemnity, + or other liability obligations and/or rights consistent with this + License. However, in accepting such obligations, You may act only + on Your own behalf and on Your sole responsibility, not on behalf + of any other Contributor, and only if You agree to indemnify, + defend, and hold each Contributor harmless for any liability + incurred by, or claims asserted against, such Contributor by reason + of your accepting any such warranty or additional liability. + + END OF TERMS AND CONDITIONS + + APPENDIX: How to apply the Apache License to your work. + + To apply the Apache License to your work, attach the following + boilerplate notice, with the fields enclosed by brackets "[]" + replaced with your own identifying information. (Don't include + the brackets!) The text should be enclosed in the appropriate + comment syntax for the file format. We also recommend that a + file or class name and description of purpose be included on the + same "printed page" as the copyright notice for easier + identification within third-party archives. + + Copyright 2026 KR Shanto + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. diff --git a/LICENSE-MIT b/LICENSE-MIT new file mode 100644 index 0000000..b607d75 --- /dev/null +++ b/LICENSE-MIT @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2026 KR Shanto + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/README.md b/README.md new file mode 100644 index 0000000..ae29075 --- /dev/null +++ b/README.md @@ -0,0 +1,176 @@ +# SEO Lens + +**High-performance, local-first website crawler, 120-rule technical SEO audit engine, and AI-native auditor written in Rust.** + +[![License: MIT / Apache-2.0](https://img.shields.io/badge/license-MIT%2FApache--2.0-blue.svg)](./LICENSE-MIT) +![Rust 2021](https://img.shields.io/badge/rust-2021_edition-orange.svg) +![Model Context Protocol](https://img.shields.io/badge/mcp-compliant-green.svg) +![Local First SQLite WAL](https://img.shields.io/badge/local--first-sqlite_wal-purple.svg) + +--- + +## Overview + +**SEO Lens** is a lightweight, blazing-fast website crawler and technical SEO auditor designed as a modern, local-first alternative to heavy legacy tools like Screaming Frog and expensive cloud crawlers. + +Built from the ground up in Rust, SEO Lens streams HTML responses through Cloudflare's zero-copy `lol_html` parser, dynamically adjusts crawl speed using network congestion algorithms (AIMD), persists audit history in an embedded SQLite database, and exposes a native **Model Context Protocol (MCP)** server for autonomous AI coding agents. + +![SEO Lens CLI Audit in Terminal](./docs/assets/cli_terminal_audit.png) + +![SEO Lens Interactive HTML Report](./docs/assets/html_report_dashboard.png) + +--- + +## Key Features + +- ⚑ **Blazingly Fast**: Crawls ~500–800 pages/second on raw HTTP using under 50 MB of RAM. +- πŸ”’ **Local-First & Private**: All data stays on your machine in an embedded SQLite WAL database. No accounts, no cloud dependencies, zero telemetry. +- 🧠 **AI-Native (MCP)**: Native stdio Model Context Protocol server. AI agents in Claude Desktop, Cursor, and Windsurf can launch audits and inspect findings autonomously. +- πŸ›‘οΈ **AIMD Congestion Politeness**: Additive-Increase / Multiplicative-Decrease rate controller actively protects target servers from overload. +- πŸ” **120 Comprehensive Rules**: Validates HTTP transport, metadata, headings, directives, canonicalization, links, security headers, CLS image dimensions, hreflang reciprocity, and Google Rich Results. +- πŸ€– **GEO & AI Search Readiness**: Audits `/llms.txt`, probes AI crawler permissions (`GPTBot`, `ClaudeBot`), and flags bot-challenge firewall screens. +- πŸ•ΈοΈ **Topology & Internal PageRank**: Ingests link graphs into `petgraph` to detect orphan pages, canonical loops, and distribute internal link equity. +- πŸ“Š **Multi-Format Reporting**: Generates interactive standalone TUI-style HTML reports, Screaming Frog-compatible CSV suites, Markdown for LLMs, and structured JSON. + +--- + +## Quickstart + +> **Prerequisites**: Requires **Rust 1.80+** ([rustup.rs](https://rustup.rs/)). On Linux, ensure `build-essential`, `pkg-config`, and `libssl-dev` are installed (see [CONTRIBUTING.md](./CONTRIBUTING.md) for OS-specific details). + +### 1. Clone the Repository + +```bash +git clone https://github.com/Shantodotdev/seo-lens.git +cd seo-lens +``` + +### 2. Choose How to Run + +You have three flexible options depending on your workflow: + +#### Option A: Run Directly with Cargo (Best for quick testing) + +Run audits immediately without installing anything to your system PATH: + +```bash +# Cargo compiles and runs seolens on the fly +cargo run -- audit https://example.com --max-pages 100 +``` + +#### Option B: Build the Standalone Binary (Best for production & scripts) + +Build a self-contained, optimized release binary into `./target/release/seolens`: + +```bash +# Compile optimized release binary +cargo build --release + +# Run the binary directly +./target/release/seolens audit https://example.com --max-pages 500 +``` + +#### Option C: Install Globally as a System Command (Best for daily use) + +`cargo install --path .` compiles the release binary and copies it to your global Cargo binary directory (`~/.cargo/bin`). This makes `seolens` available from **any folder** in your terminal like any standard Unix tool: + +```bash +# Install to ~/.cargo/bin/seolens +cargo install --path . + +# Now available globally from any terminal folder +seolens audit https://example.com --max-pages 500 +``` + +--- + +### Basic Usage + +#### 1. Audit a Website + +```bash +# Run a 500-page crawl with concurrency limit of 10 +seolens audit https://example.com --max-pages 500 -c 10 +``` + +#### 2. Quick Single-Page Inspection + +```bash +# Inspect a single URL instantly without crawling the whole site +seolens inspect https://example.com/blog/post-1 +``` + +#### 3. Inspect Issues by Severity + +```bash +# List all critical and alert issues from the latest session +seolens issues --severity critical,alert +``` + +#### 4. Export Reports + +```bash +# Export an interactive HTML dashboard and Screaming Frog-compatible CSVs +seolens report -f html,csv -o ./reports +``` + +#### 5. Check AI Search Readiness + +```bash +# Check /llms.txt and AI bot permissions in robots.txt +seolens check-ai https://example.com +``` + +--- + +## Use with AI Agents (MCP) + +SEO Lens includes a built-in Model Context Protocol (MCP) server running pure-Rust JSON-RPC 2.0 over `stdio`. + +### Quick Setup + +Tell your AI coding agent (Claude, Cursor, Windsurf) to register SEO Lens: + +> _"Add an MCP server named `seolens` with the command `seolens` and args `['mcp']`."_ + +### What Agents Can Do + +- **`seo_start_audit`**: Spawns non-blocking background crawls (returns a session token in $<1$s). +- **`seo_audit_status`**: Polls live crawl progress, page counts, and real-time health score. +- **`seo_get_markdown_report`**: Generates executive Markdown summaries tailored for LLM context windows. +- **`seo_query_issues`**: Filters issues by category, severity, and rule code. +- **`seo_quick_page_check`**: Instantly audits single URLs during local development. + +_(See [Model Context Protocol Guide](./docs/mcp.md) for full tool schemas and configurations)._ + +--- + +## Documentation + +Explore the complete technical documentation in [`docs/`](./docs/README.md): + +| Guide | Description | +| ---------------------------------------------------- | ------------------------------------------------------------------------------- | +| [**Architecture & Roadmap**](./docs/architecture.md) | Pipeline design, 2-member workspace, streaming parsing, and project milestones. | +| [**CLI Commands & Flags**](./docs/cli.md) | Exhaustive reference for all 10 subcommands, CLI options, and exporters. | +| [**Crawler Engine & AIMD**](./docs/crawler.md) | Asynchronous crawler mechanics, AIMD rate tuning, and RFC 9309 compliance. | +| [**Model Context Protocol (MCP)**](./docs/mcp.md) | Agent configuration, stdio JSON-RPC architecture, and 8 structured tools. | +| [**120 SEO Rules Catalog**](./docs/rules.md) | Complete reference of all 120 technical checks, heuristics, and fix advice. | +| [**Storage & SQLite Schema**](./docs/storage.md) | Database architecture, WAL mode, 7 relational tables, and SQL query recipes. | + +--- + +## Contributing + +We welcome contributions! Please review [CONTRIBUTING.md](./CONTRIBUTING.md) for development setup, testing guidelines, and instructions on how to add new SEO rules. + +--- + +## License + +SEO Lens is dual-licensed under either: + +- **MIT License** ([LICENSE-MIT](./LICENSE-MIT)) +- **Apache License, Version 2.0** ([LICENSE-APACHE](./LICENSE-APACHE)) + +at your option. diff --git a/docs/CLI_AND_REPORTS_SPEC.md b/docs/CLI_AND_REPORTS_SPEC.md deleted file mode 100644 index 2df7bc4..0000000 --- a/docs/CLI_AND_REPORTS_SPEC.md +++ /dev/null @@ -1,224 +0,0 @@ -# SEO Lens: CLI Interface & Report Exporters Specification - -Command-Line Interface (Clap v4), Terminal UI (Indicatif), CI/CD Exit Codes, and Multi-Format Exporters (JSON, Markdown, HTML, Screaming Frog CSVs) - ---- - -## 1. CLI Command Hierarchy & Subcommands - -`SEO Lens` uses `clap v4` with the derive macro to expose a modern, intuitive subcommand interface via the `seolens` binary: - -```bash -seolens -β”œβ”€β”€ audit # Run a full or partial website crawl and audit -β”œβ”€β”€ mcp # Start the native Model Context Protocol server (stdio for Cursor/Claude) -β”œβ”€β”€ report # Re-export or inspect an existing audit from the SQLite database -└── list # List all historical audit sessions stored locally -``` - -_(Note: The visual dashboard is provided as a dedicated native desktop application via Tauri v2, eliminating browser port conflicts and providing a double-clickable experience for non-technical users and CMS creators.)_ - ---- - -## 2. Command Details & Flag Specification - -### 2.1 `seolens audit ` - -The primary command for technical SEO auditing. - -```bash -seolens audit https://example.com [FLAGS] [OPTIONS] -``` - -#### Arguments & Options: - -| Flag / Option | Short | Type | Default | Description | -| --------------- | ----- | -------- | ------------------ | ------------------------------------------------------------------------------------- | -| `` | | `String` | _(Required)_ | Root URL to crawl (e.g. `https://client.com`). | -| `--max-pages` | `-p` | `u32` | `500` | Maximum pages to crawl (`0` = unlimited). | -| `--max-depth` | `-d` | `u16` | `5` | Maximum crawl depth from start URL. | -| `--concurrency` | `-c` | `usize` | `10` | Number of concurrent fetch tasks. | -| `--delay` | | `u64` | `0` | Delay between requests in milliseconds (0 = auto-AIMD). | -| `--render-js` | | `bool` | `false` | Enable Headless Chrome CDP for JavaScript rendering. | -| `--chrome-ws` | | `String` | `auto` | Remote Chrome WebSocket URL (e.g. `ws://127.0.0.1:9222`). | -| `--user-agent` | `-u` | `String` | `SEOLens/1.0` | Custom User-Agent string. | -| `--format` | `-f` | `String` | `terminal,json,md` | Comma-separated outputs: `terminal,json,md,html,csv,all`. | -| `--output-dir` | `-o` | `Path` | `./reports` | Directory where export artifacts are saved. | -| `--fail-on` | | `String` | `none` | CI/CD threshold: `critical`, `alert`, or `warning`. Returns exit code `1` if matched. | -| `--no-robots` | | `bool` | `false` | Ignore `/robots.txt` disallow rules. | -| `--ephemeral` | | `bool` | `false` | Do not persist results to SQLite; auto-cleanup on finish. | - ---- - -### 2.2 `seolens mcp` - -Launches the Model Context Protocol server for AI coding assistants (Claude Desktop, Cursor, Windsurf). - -```bash -seolens mcp [OPTIONS] -``` - -#### Options - -| Option | Default | Description | -| ------------- | ------- | ------------------------------------------------------------------- | -| `--transport` | `stdio` | Transport mechanism: `stdio` (local agents) or `sse` (remote HTTP). | -| `--port` | `8080` | Port to bind for HTTP/SSE transport (when `--transport sse`). | - ---- - -### 2.3 `seolens report ` - -Inspects or re-exports an existing audit session from SQLite. - -```bash -seolens report --format csv,json,md -o ./exports -``` - ---- - -### 2.4 `seolens list` - -Lists all audit sessions currently stored in the local SQLite database. - -```bash -seolens list -``` - ---- - -## 3. Exit Codes (CI/CD Pipeline Integration) - -`SEO Lens` is designed to run in automated GitHub Actions, GitLab CI, and deployment pipelines: - -| Exit Code | Meaning | Condition | -| --------- | --------------------------- | ---------------------------------------------------------------------------------------------------------------------------- | -| `0` | **Success / Clean** | Audit completed successfully with no violations above `--fail-on` threshold. | -| `1` | **SEO Threshold Violation** | Found one or more issues matching the `--fail-on` flag (e.g. `--fail-on critical` failed on broken links or missing titles). | -| `2` | **Runtime / Network Error** | Target URL unreachable, DNS resolution failed, invalid flags, or disk full. | -| `130` | **Interrupted** | Gracefully terminated by user via `SIGINT` (`Ctrl+C`). SQLite state remains intact. | - ---- - -## 4. Terminal UI Specification (`indicatif`) - -During execution, `seolens audit` renders a live, colored ANSI terminal dashboard: - -```bash -πŸ” SEO Lens v0.1.0 β€” Crawling: https://example.com -──────────────────────────────────────────────────────────────────────── -[00:00:18] [β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‘β–‘β–‘β–‘β–‘] 342/500 pages (19.0 p/s) -Active Delay: 75ms (AIMD) | p95 TTFB: 240ms | Memory: 32MB - -Found Issues: 🚨 2 Critical | ⚠️ 8 Alerts | ⚑ 24 Warnings -──────────────────────────────────────────────────────────────────────── -Current: https://example.com/products/wireless-headphones -``` - -### Post-Crawl Terminal Scorecard - -Upon completion, the terminal displays an executive scorecard: - -```bash -======================================================================== - SEO LENS AUDIT SCORECARD -======================================================================== -Target: https://example.com -Health Score: 84 / 100 -Duration: 24.6s (482 pages crawled, 1,420 internal links) -Avg TTFB: 185ms (p95: 310ms) - -HTTP Status Breakdown: - βœ” 200 OK: 468 (97.1%) - β„Ή 301 Redirect: 10 (2.1%) - βœ– 404 Not Found: 4 (0.8%) - -Top Priority Issues: - 🚨 [ERR_CANONICAL_TO_4XX_5XX] (3 pages) - Canonical points to dead 404 URL - 🚨 [ERR_H1_MISSING] (1 page) - https://example.com/checkout - ⚠️ [ALERT_GEO_AI_RETRIEVAL_BOT_BLOCKED] (Site-wide) - robots.txt blocks PerplexityBot - ⚑ [WARN_IMG_MISSING_ALT] (14 images) - Missing descriptive alt attributes - -Exported Artifacts: - πŸ“„ Markdown: ./reports/example_com_audit.md - πŸ“Š JSON: ./reports/example_com_audit.json - 🌐 HTML: ./reports/example_com_audit.html - πŸ“‘ CSVs: ./reports/csv/ (internal_all.csv, issues_all.csv) -======================================================================== -``` - ---- - -## 5. Report Exporters Specification - ---- - -### 5.1 Standalone HTML Report (`report.html`) - -- **Self-Contained**: CSS and JS are compiled directly into the HTML file using `rust-embed`. -- **Zero External Dependencies**: Renders completely offline without calling Google Fonts, external CDNs, or third-party trackers. -- **Interactive Features**: - - Live search bar filtering by URL, title, or status code. - - Severity filter chips (Critical, Alert, Warning). - - Issue accordion with copyable code remediation instructions. - - Interactive link equity chart and crawl depth histogram. - ---- - -### 5.2 Screaming Frog Compatible CSV Suite - -To enable immediate compatibility with existing client spreadsheet workflows, `seolens` exports four industry-standard CSVs under `--format csv`: - -#### 1. `internal_all.csv` - -Columns matching Screaming Frog standard export: -`Address`, `Status Code`, `Status`, `Content Type`, `Size (Bytes)`, `Word Count`, `Title 1`, `Title 1 Length`, `Meta Description 1`, `Meta Description 1 Length`, `H1-1`, `H1-1 Length`, `Canonical Link Element 1`, `Indexability`, `Indexability Status`, `Inlinks`, `Outlinks`, `Crawl Depth`, `Response Time (ms)`. - -#### 2. `issues_all.csv` - -Summary of all triggered rules: -`Issue Code`, `Issue Name`, `Severity`, `Category`, `URL`, `Source URL`, `Details`, `Recommendation`. - -#### 3. `response_codes.csv` - -URL routing map: -`URL`, `Status Code`, `Status`, `Redirect URL`, `Redirect Type`, `Inlinks Count`. - -#### 4. `external_all.csv` - -Outbound link audit: -`Source URL`, `Destination URL`, `Anchor Text`, `Status Code`, `Is Nofollow`. - ---- - -### 5.3 Machine-Readable JSON (`audit.json`) - -The complete, lossless structured schema containing: - -- `summary`: Crawl statistics, health score, duration, timing percentiles. -- `crawled_pages`: Array of full `PageReport` objects. -- `issues`: Grouped issue arrays with occurrence counts and affected URLs. -- `site_graph`: Nodes (URLs) and edges (links with anchor text and attributes). - ---- - -### 5.4 Executive Markdown (`report.md`) - -Designed for human executive review and client emails: - -- Clean GitHub Flavored Markdown with badge formatting. -- Organized into: Executive Summary $\rightarrow$ Critical Blockers $\rightarrow$ Optimization Opportunities $\rightarrow$ Technical Action Items. - ---- - -## 6. Summary - -This specification guarantees: - -- A polished, developer-friendly CLI with standard UNIX exit codes for CI/CD integration. -- 100% interoperability with agency client workflows via Screaming Frog compatible CSVs. -- Self-contained, beautiful offline HTML dashboards that clients can open directly in any browser. diff --git a/docs/CRAWLER_SPEC.md b/docs/CRAWLER_SPEC.md deleted file mode 100644 index 0a7f667..0000000 --- a/docs/CRAWLER_SPEC.md +++ /dev/null @@ -1,251 +0,0 @@ -# SEO Lens: Crawler Engine & Politeness Specification - -**Scope**: Network Concurrency, AIMD Adaptive Politeness, URL Normalization Pipeline, Frontier Scheduling, and RFC 9309 Robots/Sitemap Ingestion - ---- - -## 1. Concurrency Architecture & Task Flow - -`SEO Lens` uses an asynchronous producer-consumer pipeline built on the Tokio runtime: - -```mermaid -flowchart TD - subgraph Discovery ["URL Discovery & Frontier"] - Queue["URL Frontier (FIFO mpsc / VecDeque)"] - VisitedSet["Visited Set (SwissTable 64-bit Hashes)"] - SitemapEngine["Sitemap Ingestion Engine (quick-xml)"] - RobotsEngine["Robots.txt Engine (RFC 9309)"] - end - - SitemapEngine -->|Seed URLs| Queue - - subgraph ConcurrencyPipeline ["Concurrency & Politeness"] - Queue -->|Pop (URL, Depth)| Worker["Worker Task (tokio::spawn)"] - Worker -->|Check Permission| RobotsEngine - Worker --> RateGate["AIMD Adaptive Rate Controller (Semaphore + Token Bucket)"] - RateGate --> HTTPClient["reqwest HTTP/2 Client Pool"] - end - - subgraph ResponseProcessing ["Stream Ingestion & Feedback"] - HTTPClient --> Telemetry["Latency & Error Rate Feedback Loop"] - Telemetry -->|Tune Delay & Concurrency| RateGate - HTTPClient --> Parser["Streaming lol_html Tokenizer"] - Parser --> LinkFilter["Domain Boundary & Link Filter"] - LinkFilter -->|Unseen URLs| VisitedSet - VisitedSet -->|New URLs| Queue - end -``` - ---- - -## 2. The AIMD Adaptive Politeness Engine - -Crawling client websites without rate limiting risks causing CPU spikes, database connection exhaustion, or triggering Cloudflare/WAF IP blocks. - -`SEO Lens` adapts the **Additive-Increase/Multiplicative-Decrease (AIMD)** congestion control algorithm (inspired by TCP Reno and `librecrawl-mcp`): - -### 2.1 Mathematical Model & Parameters - -``` - β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β” - β”‚ Sample Window (N = 50) β”‚ - β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”¬β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜ - β”‚ - Compute: Error Rate (E) & p95 TTFB - β”‚ - β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”΄β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β” - β–Ό β–Ό - Degradation Trigger Healthy Recovery - (E > 8% OR p95 > 1.5x baseline) (E == 0% AND p95 < 500ms) - β”‚ β”‚ - β–Ό β–Ό - Multiplicative Backoff: Additive Recovery: - delay = min(delay * 2.0, 10,000ms) delay = max(delay - 25ms, delay_floor) - concurrency = max(floor(c * 0.5), 1) concurrency = min(c + 1, c_max) -``` - -| Parameter | Symbol | Value | Rationale | -| ------------------------- | ----------------------------- | ---------------------------------------- | -------------------------------------------------------------------------------------- | -| **Sample Window** | $W$ | `50 requests` | Large enough to smooth statistical outliers, small enough to react in $<3$ seconds. | -| **Error Rate Ceiling** | $E_{\text{thresh}}$ | `0.08` (8%) | If $>4$ requests out of 50 fail (429, 500, 502, 503, 504), origin is in distress. | -| **Latency Multiplier** | $L_{\text{factor}}$ | `1.5` | If p95 TTFB exceeds $1.5\times$ baseline average, backend database is queuing queries. | -| **Multiplicative Factor** | $\beta$ | `2.0` | Instantly halves origin load when distress is detected. | -| **Additive Step** | $\Delta_{\text{add}}$ | `25 ms` | Gently steps up crawl rate without shocking the origin. | -| **Max Delay Ceiling** | $\text{delay}_{\text{max}}$ | `10,000 ms` | Prevents infinite stalls; caps wait at 10 seconds per request. | -| **Delay Floor** | $\text{delay}_{\text{floor}}$ | $\max(\text{robots\_delay}, 0\text{ms})$ | Strictly obeys `Crawl-Delay` specified in `/robots.txt`. | - ---- - -## 3. URL Normalization Pipeline - -To eliminate crawl duplication (e.g. `https://example.com`, `http://example.com/`, `https://example.com/?utm_source=fb` must all resolve to the same canonical identifier), all discovered links pass through an 8-stage normalization pipeline in `src/core/url.rs`: - -``` -Raw Href ──> [1. Scheme] ──> [2. Hostname] ──> [3. Port] ──> [4. Path] - ──> [5. Trailing Slash] ──> [6. Strip Fragments] - ──> [7. Strip Tracking Query] ──> [8. Sort Query] ──> Normalized URL -``` - -### Stage-by-Stage Transformation Specification: - -1. **Scheme Normalization**: - - Convert scheme to lowercase (`HTTP` $\rightarrow$ `http`, `HTTPS` $\rightarrow$ `https`). - - Resolve protocol-relative URLs (`//cdn.example.com/a` $\rightarrow$ `https://cdn.example.com/a`). -2. **Hostname Normalization**: - - Convert host to lowercase (`Example.COM` $\rightarrow$ `example.com`). - - Remove trailing root dot if present (`example.com.` $\rightarrow$ `example.com`). -3. **Default Port Stripping**: - - Strip standard ports (`http://example.com:80/` $\rightarrow$ `http://example.com/`). - - Strip `:443` for HTTPS (`https://example.com:443/` $\rightarrow$ `https://example.com/`). -4. **Path Segment Resolution**: - - Resolve relative path dots (`/a/b/../c` $\rightarrow$ `/a/c`). - - Deduplicate consecutive internal slashes (`/blog//post` $\rightarrow$ `/blog/post`). -5. **Root Path Enforcement**: - - If path is completely empty, supply `/` (`https://example.com` $\rightarrow$ `https://example.com/`). -6. **Fragment Removal**: - - Strip hash fragments (`/page#reviews` $\rightarrow$ `/page`). -7. **Tracking Parameter Stripping**: - - Strip marketing, analytics, and session query parameters: - - `utm_source`, `utm_medium`, `utm_campaign`, `utm_term`, `utm_content` - - `fbclid`, `gclid`, `msclkid`, `mc_eid`, `_ga`, `_gl`, `ref` -8. **Deterministic Query Parameter Sorting**: - - Lexicographically sort remaining query keys (`?b=2&a=1` $\rightarrow$ `?a=1&b=2`). - -### Hash-Based Deduplication - -Once normalized, the URL string is converted into a **64-bit AHash (`u64`)**. - -- The `VisitedSet` in `src/crawler/frontier.rs` stores only `u64` hashes in a SwissTable (`hashbrown::HashSet`). -- Storing 50,000 URLs consumes only **400 KB of RAM** compared to >10 MB for raw strings. - ---- - -## 4. Frontier Queue & Depth Management - -The frontier schedules pending URLs and enforces crawl boundaries: - -```rust -pub struct FrontierEntry { - pub url: String, - pub depth: u16, - pub source_url: Option, -} -``` - -### Scheduling Policies: - -1. **Breadth-First Search (BFS) (Default)**: - - Uses `VecDeque` (FIFO queue). - - Ensures shallow, high-PageRank category pages are audited before deep product catalog leaves. -2. **Depth Limiting**: - - When a link is discovered on page at `depth = d`, its candidate entry is assigned `depth = d + 1`. - - If `d + 1 > max_depth`, the URL is added to the link graph as an edge but is **never enqueued for crawling**. -3. **Domain Boundary Control**: - - **Internal Domain**: Host matches starting domain or its subdomains (if `--include-subdomains` is on). Crawled recursively. - - **External Domain**: Outbound link. Validated with a lightweight HTTP `HEAD` or single `GET` to verify HTTP status, but child links are not extracted. - ---- - -## 5. RFC 9309 Robots.txt & Sitemap Compliance - -### 5.1 Robots.txt Rules (`src/crawler/robots.rs`) - -Complies strictly with **RFC 9309 (Robots Exclusion Protocol)**: - -1. **User-Agent Matching**: - - Specific user-agent matches take precedence over wildcard `*`. - - Priority order: `SEOLens` $\rightarrow$ `Googlebot` $\rightarrow$ `*`. -2. **Longest Match Precedence**: - - When multiple rules match a path, the rule with the longest matching character length wins: - ``` - Allow: /products/ - Disallow: /products/archived/ - # URL: /products/archived/item-1 -> DISALLOWED (20 chars vs 10 chars) - ``` -3. **Allow Overrides Disallow on Equal Length**: - - If `Allow: /blog` and `Disallow: /blog` have identical length, `Allow` wins per RFC 9309. -4. **Wildcards**: Supports `*` (zero or more characters) and `$` (end of URL). - -### 5.2 Streaming Sitemap Ingestion (`src/crawler/sitemap.rs`) - -Uses `quick-xml` for zero-allocation streaming XML parsing: - -1. **Auto-Discovery**: - - Checks `Sitemap:` directives declared in `/robots.txt`. - - Checks common standard paths: `/sitemap.xml`, `/sitemap_index.xml`, `/wp-sitemap.xml`. -2. **Sitemap Index Recursion**: - - When a `` document is parsed, its child `` URLs are recursively fetched and parsed up to 3 levels deep. -3. **Compressed Sitemaps (`.xml.gz`)**: - - Automatically detects `.gz` extensions and decompresses streams on the fly using `flate2`. -4. **Orphan Identification Hook**: - - All URLs extracted from sitemaps are tagged with `is_sitemap_url = true`. - - If post-crawl analysis finds that a sitemap URL has 0 incoming internal links from HTML pages, `ALERT_GRAPH_ORPHAN_PAGE` is raised. - ---- - -## 6. WAF & Anti-Bot Fingerprint Probes (`src/crawler/waf.rs`) - -When crawling client sites protected by Cloudflare, Akamai, or DataDome, firewalls often return HTTP 200 OK or 403 with a JavaScript challenge screen. If parsed blindly, the crawler reports "0 words, missing H1, missing title" when the site is actually healthy. - -`SEO Lens` probes raw responses for known challenge signatures: - -```rust -pub struct WafProbe { - pub provider: &'static str, - pub signatures: &'static [&'static str], -} - -pub const WAF_SIGNATURES: &[WafProbe] = &[ - WafProbe { - provider: "Cloudflare", - signatures: &[ - "cf-browser-verification", - "checking your browser before accessing", - "cloudflare ray id", - "/cdn-cgi/challenge-platform/", - ], - }, - WafProbe { - provider: "Akamai", - signatures: &[ - "akamai bot manager", - "_abck=", - "reference number:", - "bm-sz=", - ], - }, - WafProbe { - provider: "DataDome", - signatures: &[ - "geo.captcha-delivery.com", - "datadome", - "dd_cookie_test", - ], - }, - WafProbe { - provider: "Imperva", - signatures: &[ - "incapsula incident id", - "_incap_ses", - "visid_incap", - ], - }, -]; -``` - -When a signature matches, `SEO Lens`: - -1. Flags `ALERT_WAF_BOT_CHALLENGE` on the URL. -2. Does **not** trigger false-positive alerts for missing titles, H1s, or content. -3. Automatically instructs the user/agent in the report: _"Blocked by Cloudflare/Akamai WAF challenge. Whitelist the crawler IP or pass a valid session cookie."_ - ---- - -## 7. Summary - -This specification guarantees: - -- Origin servers are actively protected via empirical AIMD rate-tuning. -- URLs are deduplicated with zero ambiguity through an 8-stage pipeline. -- RFC 9309 robots compliance and XML sitemap recursion operate reliably. -- Anti-bot firewall screens are accurately detected without polluting client audit reports. diff --git a/docs/DATA_MODELS_AND_SCHEMA.md b/docs/DATA_MODELS_AND_SCHEMA.md deleted file mode 100644 index 4bd3de0..0000000 --- a/docs/DATA_MODELS_AND_SCHEMA.md +++ /dev/null @@ -1,361 +0,0 @@ -# SEO Lens: Data Models & Storage Schema Specification -**Document Status**: Permanent Technical Specification -**Scope**: Rust Domain Models, Memory Layout, and SQLite WAL Database Schema - ---- - -## 1. Architectural Philosophy & Memory Strategy - -In a large crawl (10,000+ pages with 500,000+ internal links and assets), naive string allocations cause heap fragmentation, GC/allocator pauses, and gigabytes of wasted RAM. - -`SEO Lens` enforces strict memory principles: -1. **Stack-Inlined Strings (`compact_str::CompactString`)**: - - Standard Rust `String` is 24 bytes pointing to a heap buffer. - - `CompactString` stores up to 24 bytes directly on the stack (covering >80% of URL paths, titles, tags, and status codes) without touching the heap allocator. -2. **Compact Numeric Primitives**: - - HTTP status codes are stored as `u16` (2 bytes). - - Word counts, character lengths, and response times (TTFB in milliseconds) are stored as `u32` (4 bytes). -3. **Bitflags for Directives**: - - Robots directives (`noindex`, `nofollow`, `noarchive`, `nosnippet`, `noimageindex`) are packed into a single 1-byte bitfield (`u8`) rather than 5 separate booleans or heap-allocated strings. -4. **Relational SQLite with Write-Ahead Logging (WAL)**: - - Synchronous disk writes on every page destroy crawl throughput. - - SQLite is configured in `WAL` mode with `PRAGMA synchronous = NORMAL`, writing in atomic batches of 250 pages. - ---- - -## 2. Core Rust Domain Models - -Below is the definitive specification of data structures in `src/core/models.rs`: - -```rust -use compact_str::CompactString; -use serde::{Deserialize, Serialize}; -use std::time::Duration; - -/// Severity classification for technical SEO findings. -#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] -#[serde(rename_all = "snake_case")] -pub enum Severity { - Critical = 1, - Alert = 2, - Warning = 3, - Notice = 4, -} - -/// Functional categories mapping to the 120-check catalog. -#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)] -#[serde(rename_all = "snake_case")] -pub enum IssueCategory { - HttpTransport, - TitleMetadata, - Headings, - Indexability, - Canonicalization, - Links, - Security, - MobileUx, - Internationalization, - StructuredData, - GeoAiSearch, - SiteGraph, - JsDiff, -} - -bitflags::bitflags! { - /// Memory-efficient bitfield for robots and indexing directives. - #[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)] - pub struct RobotsFlags: u8 { - const NONE = 0b0000_0000; - const NOINDEX = 0b0000_0001; - const NOFOLLOW = 0b0000_0010; - const NOSNIPPET = 0b0000_0100; - const NOIMAGEINDEX = 0b0000_1000; - const NOARCHIVE = 0b0001_0000; - } -} - -/// Represents the complete audit report for a single crawled URL. -#[derive(Debug, Clone, Serialize, Deserialize)] -pub struct PageReport { - /// Unique incremental identifier (primary key in SQLite) - pub id: Option, - /// Associated crawl session identifier - pub crawl_id: CompactString, - - // --- Network & Transport --- - pub url: String, - pub url_hash: u64, - pub final_url: Option, - pub status_code: u16, - pub content_type: CompactString, - pub size_bytes: u32, - pub ttfb_ms: u32, - pub crawl_depth: u16, - - // --- Metadata --- - pub title: Option, - pub title_length: u16, - pub meta_description: Option, - pub meta_desc_length: u16, - pub canonical_url: Option, - pub html_lang: Option, - pub charset: Option, - pub viewport: Option, - - // --- Directives --- - pub robots_flags: RobotsFlags, - pub is_sitemap_url: bool, - pub is_internal: bool, - - // --- Headings --- - pub h1_primary: Option, - pub h1_count: u16, - pub h2_headings: Vec, - pub h3_headings: Vec, - - // --- Content & Quality --- - pub word_count: u32, - pub content_hash: u64, - pub simhash: u64, - pub is_soft_404: bool, - pub has_lorem_ipsum: bool, - - // --- Security --- - pub is_https: bool, - pub has_hsts: bool, - pub has_csp: bool, - pub has_x_frame: bool, - pub has_x_content_type: bool, - pub mixed_content_count: u16, - - // --- Child Collections (stored relationally) --- - pub links: Vec, - pub images: Vec, - pub schemas: Vec, - pub hreflangs: Vec, - pub issues: Vec, -} - -/// A hyperlink discovered in an HTML document. -#[derive(Debug, Clone, Serialize, Deserialize)] -pub struct DiscoveredLink { - pub source_url: String, - pub target_url: String, - pub target_url_hash: u64, - pub anchor_text: String, - pub is_internal: bool, - pub is_nofollow: bool, - pub is_image_link: bool, - pub status_code: Option, -} - -/// An image asset referenced on a page. -#[derive(Debug, Clone, Serialize, Deserialize)] -pub struct ImageResource { - pub src_url: String, - pub alt_text: Option, - pub width: Option, - pub height: Option, - pub size_bytes: Option, - pub has_dimensions: bool, - pub is_broken: bool, -} - -/// JSON-LD or Microdata structured data block. -#[derive(Debug, Clone, Serialize, Deserialize)] -pub struct SchemaRecord { - pub schema_type: CompactString, - pub raw_json: String, - pub is_valid_json: bool, - pub is_google_eligible: bool, - pub missing_required_fields: Vec, -} - -/// Hreflang alternate language tag. -#[derive(Debug, Clone, Serialize, Deserialize)] -pub struct HreflangTag { - pub lang_code: CompactString, - pub target_url: String, - pub is_reciprocal: bool, -} - -/// A specific technical SEO defect identified by the rules engine. -#[derive(Debug, Clone, Serialize, Deserialize)] -pub struct IssueFinding { - pub code: CompactString, // e.g. "ERR_TITLE_MISSING" - pub category: IssueCategory, - pub severity: Severity, - pub title: CompactString, // Human-readable headline - pub message: String, // Context-specific detail - pub target_url: String, - pub source_page_url: Option, // Which page linked here (for 404s/broken links) -} - -/// Summary metrics for an entire crawl session. -#[derive(Debug, Clone, Default, Serialize, Deserialize)] -pub struct CrawlSummary { - pub session_id: String, - pub target_url: String, - pub started_at: String, - pub finished_at: Option, - pub total_pages_crawled: u32, - pub total_links_discovered: u32, - pub total_errors: u32, - pub total_alerts: u32, - pub total_warnings: u32, - pub total_notices: u32, - pub average_ttfb_ms: u32, - pub p95_ttfb_ms: u32, - pub health_score: u8, // 0-100 score calculated by weighted severity -} -``` - ---- - -## 3. SQLite Relational Database Schema (DDL) - -The schema is initialized via `src/storage/schema.sql`. It is tuned for write throughput during ingestion and fast indexed queries during reporting and MCP filtering. - -```sql --- Enables Write-Ahead Logging for high concurrent read/write throughput -PRAGMA journal_mode = WAL; -PRAGMA synchronous = NORMAL; -PRAGMA foreign_keys = ON; -PRAGMA cache_size = -64000; -- 64MB page cache - --- Crawl Sessions table -CREATE TABLE IF NOT EXISTS crawls ( - session_id TEXT PRIMARY KEY, - target_url TEXT NOT NULL, - status TEXT NOT NULL CHECK(status IN ('queued', 'crawling', 'analyzing_graph', 'completed', 'failed', 'paused')), - total_max_pages INTEGER NOT NULL DEFAULT 10000, - max_depth INTEGER NOT NULL DEFAULT 5, - respect_robots BOOLEAN NOT NULL DEFAULT 1, - render_js BOOLEAN NOT NULL DEFAULT 0, - started_at TEXT NOT NULL, - finished_at TEXT, - total_pages INTEGER NOT NULL DEFAULT 0, - error_count INTEGER NOT NULL DEFAULT 0, - alert_count INTEGER NOT NULL DEFAULT 0, - warning_count INTEGER NOT NULL DEFAULT 0, - health_score INTEGER DEFAULT NULL -); - --- Crawled Pages table -CREATE TABLE IF NOT EXISTS pages ( - id INTEGER PRIMARY KEY AUTOINCREMENT, - crawl_id TEXT NOT NULL REFERENCES crawls(session_id) ON DELETE CASCADE, - url TEXT NOT NULL, - url_hash INTEGER NOT NULL, - final_url TEXT, - status_code INTEGER NOT NULL, - content_type TEXT NOT NULL, - size_bytes INTEGER NOT NULL DEFAULT 0, - ttfb_ms INTEGER NOT NULL DEFAULT 0, - crawl_depth INTEGER NOT NULL DEFAULT 0, - title TEXT, - title_length INTEGER NOT NULL DEFAULT 0, - meta_description TEXT, - meta_desc_length INTEGER NOT NULL DEFAULT 0, - canonical_url TEXT, - html_lang TEXT, - charset TEXT, - viewport TEXT, - robots_flags INTEGER NOT NULL DEFAULT 0, - is_sitemap_url BOOLEAN NOT NULL DEFAULT 0, - is_internal BOOLEAN NOT NULL DEFAULT 1, - h1_primary TEXT, - h1_count INTEGER NOT NULL DEFAULT 0, - word_count INTEGER NOT NULL DEFAULT 0, - content_hash INTEGER NOT NULL DEFAULT 0, - simhash INTEGER NOT NULL DEFAULT 0, - is_https BOOLEAN NOT NULL DEFAULT 1, - has_hsts BOOLEAN NOT NULL DEFAULT 0, - has_csp BOOLEAN NOT NULL DEFAULT 0, - has_x_frame BOOLEAN NOT NULL DEFAULT 0, - has_x_content_type BOOLEAN NOT NULL DEFAULT 0, - created_at TEXT NOT NULL DEFAULT (datetime('now')) -); - --- Links Graph table (Internal and External) -CREATE TABLE IF NOT EXISTS links ( - id INTEGER PRIMARY KEY AUTOINCREMENT, - crawl_id TEXT NOT NULL REFERENCES crawls(session_id) ON DELETE CASCADE, - source_url TEXT NOT NULL, - target_url TEXT NOT NULL, - target_url_hash INTEGER NOT NULL, - anchor_text TEXT NOT NULL DEFAULT '', - is_internal BOOLEAN NOT NULL DEFAULT 1, - is_nofollow BOOLEAN NOT NULL DEFAULT 0, - status_code INTEGER DEFAULT NULL -); - --- Technical Issues table -CREATE TABLE IF NOT EXISTS issues ( - id INTEGER PRIMARY KEY AUTOINCREMENT, - crawl_id TEXT NOT NULL REFERENCES crawls(session_id) ON DELETE CASCADE, - page_id INTEGER REFERENCES pages(id) ON DELETE CASCADE, - code TEXT NOT NULL, - category TEXT NOT NULL, - severity INTEGER NOT NULL CHECK(severity IN (1, 2, 3, 4)), - title TEXT NOT NULL, - message TEXT NOT NULL, - target_url TEXT NOT NULL, - source_page_url TEXT -); - --- Structured Data (JSON-LD) table -CREATE TABLE IF NOT EXISTS schemas ( - id INTEGER PRIMARY KEY AUTOINCREMENT, - crawl_id TEXT NOT NULL REFERENCES crawls(session_id) ON DELETE CASCADE, - page_id INTEGER NOT NULL REFERENCES pages(id) ON DELETE CASCADE, - schema_type TEXT NOT NULL, - raw_json TEXT NOT NULL, - is_valid_json BOOLEAN NOT NULL DEFAULT 1, - is_google_eligible BOOLEAN NOT NULL DEFAULT 1, - missing_required TEXT NOT NULL DEFAULT '[]' -); - --- Images table -CREATE TABLE IF NOT EXISTS images ( - id INTEGER PRIMARY KEY AUTOINCREMENT, - crawl_id TEXT NOT NULL REFERENCES crawls(session_id) ON DELETE CASCADE, - page_id INTEGER NOT NULL REFERENCES pages(id) ON DELETE CASCADE, - src_url TEXT NOT NULL, - alt_text TEXT, - width INTEGER, - height INTEGER, - has_dimensions BOOLEAN NOT NULL DEFAULT 0, - is_broken BOOLEAN NOT NULL DEFAULT 0 -); - --- Indexes for Microsecond Querying -CREATE INDEX IF NOT EXISTS idx_pages_crawl_hash ON pages(crawl_id, url_hash); -CREATE INDEX IF NOT EXISTS idx_pages_status ON pages(crawl_id, status_code); -CREATE INDEX IF NOT EXISTS idx_links_crawl_target ON links(crawl_id, target_url_hash); -CREATE INDEX IF NOT EXISTS idx_links_source ON links(crawl_id, source_url); -CREATE INDEX IF NOT EXISTS idx_issues_crawl_severity ON issues(crawl_id, severity); -CREATE INDEX IF NOT EXISTS idx_issues_crawl_category ON issues(crawl_id, category); -CREATE INDEX IF NOT EXISTS idx_issues_code ON issues(crawl_id, code); -``` - ---- - -## 4. Batch Transaction Ingestion Contract - -To ensure the database layer never blocks the crawler: -1. **Batch Accumulator**: - - As worker green tasks complete, `PageReport` structs are sent over a `tokio::sync::mpsc::channel(1000)` to a dedicated SQLite writer task. - - The writer accumulates up to **250 pages** or flushes every **2.0 seconds** (whichever occurs first). -2. **Transaction Batching**: - - `BEGIN TRANSACTION` $\rightarrow$ bulk insert pages, bulk insert links, bulk insert issues $\rightarrow$ `COMMIT`. - - Bypasses disk flush thrashing, sustaining thousands of page writes per second on standard SSDs. - ---- - -## 5. Summary - -This specification guarantees: -- Zero data model ambiguity between the crawler, parser, rules engine, and SQLite storage. -- Microscopic memory overhead using stack-allocated `CompactString` and packed `RobotsFlags`. -- Deterministic, high-throughput relational persistence for all 120 SEO checks. diff --git a/docs/DESKTOP_APP_SPEC.md b/docs/DESKTOP_APP_SPEC.md deleted file mode 100644 index 911ba9a..0000000 --- a/docs/DESKTOP_APP_SPEC.md +++ /dev/null @@ -1,368 +0,0 @@ -# SEO Lens: Native Desktop Application Specification (Tauri v2) -**Document Status**: Permanent Technical Specification -**Architecture**: Native Cross-Platform Desktop App (Tauri v2 + React 19 + Tailwind CSS + Vite) -**Deployment**: Native Installers (.dmg for macOS, .msi/.exe for Windows, .AppImage/.deb for Linux), 100% Offline, Zero Port Conflicts, Zero External Runtime - ---- - -## 1. System Vision & Persona Experience - -While technical developers and CI/CD pipelines interact with the `seolens` CLI via headless commands, and autonomous AI agents interact via the Model Context Protocol (MCP), **non-technical clients, agency marketers, WordPress/Webflow designers, and vibe coders require a native desktop experience**. - -Rather than running a local web server (`localhost:8080`) that creates port conflicts, firewall alerts, and terminal confusion, `SEO Lens` packages a **native desktop application powered by Tauri v2**. - -### Target Personas & Tailored Features - -``` - β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β” - β”‚ SEO Lens Core Engine (Rust) β”‚ - β”‚ Crawler β€’ 120 Rules β€’ SQLite β€’ Parser β€’ Graph β€’ Diff β”‚ - β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”¬β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜ - β”‚ - β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”΄β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β” - β”‚ β”‚ - β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β–Όβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β” β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β–Όβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β” - β”‚ CLI & MCP Mode β”‚ β”‚ Tauri v2 Desktop App β”‚ - β”‚ (Headless Single Binary)β”‚ β”‚ (Shared React 19 UI) β”‚ - β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”¬β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜ β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”¬β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜ - β”‚ β”‚ - β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”΄β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β” β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”΄β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β” - β–Ό β–Ό β–Ό β–Ό -1. Web Developers 2. Vibe Coders 3. CMS Creators 4. Non-Tech Clients -β€’ CLI terminal output β€’ MCP stdio with Cursor/ β€’ Double-click desktop app β€’ Visual Health Score (0-100) -β€’ CI/CD exit codes Claude Desktop β€’ CMS Detection (WP/Framer)β€’ 1-Click PDF / Print export -β€’ Headless Docker β€’ "Copy AI Fix Prompt" β€’ Platform-specific fixes β€’ Plain-English explanations -``` - -#### 1. "Vibe Coders" (Build with Agents, Never Read Code) -- **The Workflow**: They generate sites with Lovable, Bolt, v0, Cursor Composer, or Windsurf. They do not write Rust or dig through terminal logs. -- **"Copy AI Fix Prompt" Button**: Next to every triggered issue (e.g. missing canonical, unoptimized OpenGraph, invalid schema JSON-LD), the UI provides a 1-click button: - > *"Act as an expert web engineer. Fix this technical SEO issue on my website: Canonical URL is missing on `/pricing`. Current HTML head: `[...]`. Generate the exact code patch."* -- **MCP Integration**: Vibe coders can alternatively connect `seolens mcp` directly to Cursor or Claude Desktop to let their agent audit and fix issues autonomously. - -#### 2. WordPress, Webflow & Framer Creators -- **The Workflow**: Freelancers, agency designers, and marketers who use visual site builders or CMSs and are familiar with tools like Screaming Frog. -- **CMS Auto-Detection**: The crawler detects meta tags, script signatures, and headers for WordPress, Webflow, Framer, Shopify, Wix, and Next.js. -- **Platform-Specific Remediation Tabs**: In the Issue Detail drawer, fixes are translated into actionable CMS steps: - - *General*: RFC standard explanation and code. - - *WordPress*: Step-by-step guidance for Yoast SEO or RankMath settings. - - *Webflow / Framer*: Settings navigation (e.g., Page Settings $\rightarrow$ Custom Code / SEO). - -#### 3. Non-Technical Clients & Business Stakeholders -- **Double-Click Installer**: Standard `.dmg` on macOS, `.msi`/`.exe` on Windows. Zero terminal knowledge required. -- **Executive Health Score**: Instant visual 0–100 score ring with plain-English ratings ("Good", "Needs Attention", "Critical Issues Found"). -- **Business Impact Translations**: Translates cryptic errors into ROI/business consequences (e.g., instead of just "Canonical points to 404", explains "Google drops this page because it cannot find the preferred master version"). -- **1-Click Executive PDF & Print Export**: Generates a clean, client-ready summary document to share with teams. - -#### 4. Technical Developers & SEO Specialists -- **High-Performance Virtual Grid**: Smooth 60 FPS scrolling across 50,000+ pages via `@tanstack/react-virtual`. -- **Deep Technical Inspection**: HTTP headers, DOM hierarchy tree, JSON-LD Schema syntax validation, and side-by-side JavaScript SEO Diffing. - ---- - -## 2. Technical Stack Architecture - -```mermaid -flowchart TD - subgraph UI ["React 19 Frontend (ui/ directory)"] - ReactApp["React 19 + TypeScript"] - Tailwind["Tailwind CSS"] - TanStack["@tanstack/react-virtual (50k+ Rows)"] - Charts["Recharts / Lucide Icons"] - TauriAPI["@tauri-apps/api (IPC Client)"] - end - - subgraph TauriApp ["Tauri v2 Desktop Shell (src-tauri/)"] - WebView["OS Native Webview (WebView2 on Windows / WebKit on macOS)"] - IPCBridge["Tauri IPC Bridge (Zero Network Sockets)"] - DialogPlugin["tauri-plugin-dialog (Native File Picker)"] - NotifyPlugin["tauri-plugin-notification (OS Notifications)"] - end - - subgraph CoreEngine ["SEO Lens Rust Core Engine"] - Crawler["Crawler & AIMD Politeness"] - Rules["120 Rules Engine"] - SQLite["SQLite Storage (WAL Mode)"] - DiffEngine["JS SEO Diff Engine"] - EventBus["Tokio Broadcast Channel (Telemetry)"] - end - - ReactApp --> TauriAPI - TauriAPI <-->|Tauri IPC invoke()| IPCBridge - TauriAPI <--|Tauri listen() Events| IPCBridge - IPCBridge <--> CoreEngine - EventBus -->|app.emit()| IPCBridge - TauriApp --> DialogPlugin - TauriApp --> NotifyPlugin -``` - -### 2.1 Why Tauri v2 over Embedded Browser Server (`axum`) - -| Feature | Embedded Web Server (`axum` in browser) | Desktop App (Tauri v2) | -|---|---|---| -| **Launch UX** | Requires terminal or background script | Native app icon (Dock / Start Menu) | -| **Port Conflicts** | `localhost:8080` can collide with local dev servers | **Zero sockets**: Uses OS IPC directly | -| **File Export** | Browser download bar, fixed downloads directory | Native OS File Picker ("Save to Desktop...") | -| **OS Notifications** | Browser permissions prompt | Native system banner notifications | -| **Memory / Bundle** | Requires open browser tab + binary (~15MB) | Native OS Webview + binary (~12–18MB) | -| **Chromium Bloat** | N/A | **Zero Chromium**: Uses OS WebView2/WebKit | - ---- - -## 3. UI Screens & User Flows - ---- - -### Screen 1: Audits Overview & Launch Center -The primary workspace when opening the application. - -#### UI Elements: -- **Header Bar**: Native macOS/Windows window frame controls, Dark/Light mode toggle, Settings cog. -- **"New Audit" Action Card / Modal**: - - Target URL input (with URL auto-formatting). - - Max Pages selector (presets: 100, 500, 1,000, 10,000, Unlimited). - - Max Depth slider (1 to 10). - - JavaScript Rendering switch (`--render-js` via local Chrome CDP). - - Obey `robots.txt` switch. - - AI / GEO Readiness audit switch. -- **Audits History Table**: - - Domain, Date, Health Score badge, Total Pages, Duration. - - Actions: Open Audit, Re-crawl, Export Report, Delete. - ---- - -### Screen 2: Real-Time Live Crawl Telemetry -Displays live progress while a crawl is actively running. - -#### UI Elements: -- **Live Progress Ring**: Animated circular gauge displaying % complete. -- **Real-Time KPI Cards**: - - *Pages Crawled / Discovered*: e.g. `412 / 1,200`. - - *Current Speed*: Pages per second (e.g. `24 p/s`). - - *AIMD Delay*: Dynamic politeness backoff in ms. - - *p95 Response Latency*: TTFB in milliseconds. -- **Live Issues Ticker**: Real-time counter badges for Critical, Alert, and Warning issues found as they happen. -- **Live URL Activity Log**: Scrolling list showing currently fetching URLs. -- **Crawl Controls**: "Pause", "Resume", "Stop & Finalize". - ---- - -### Screen 3: Executive Scorecard & Overview -Comprehensive summary of completed crawl results. - -#### UI Elements: -- **Giant Health Score Gauge (0–100)**: Color-coded with plain-English health badge. -- **Status Code Breakdown**: Interactive Donut chart (2xx Green, 3xx Blue, 4xx Red, 5xx Purple). -- **Issue Priority Breakdown**: Stacked severity bar (Critical, Alert, Warning, Notice). -- **6-Pillar Radar Chart**: - - Technical & Transport - - On-Page & Headings - - Indexability & Canonicalization - - Security & HTTPS - - Mobile & UX Signals - - AI & GEO Readiness -- **Detected CMS Banner**: e.g., `"Detected CMS: WordPress 6.5 with Yoast SEO"`. -- **Top 5 Urgent Action Items**: Expandable cards with "Copy AI Fix Prompt" buttons. - ---- - -### Screen 4: All Pages Explorer (High-Performance Virtual Grid) -Primary inspection workspace for all discovered URLs. - -#### UI Elements: -- **Instant Search & Filter Bar**: Search by URL, title, or H1. Filter by status (`Broken`, `Redirects`, `Noindex`, `Canonicalized`, `Images`). -- **Virtualized Grid (`@tanstack/react-virtual`)**: - - Renders only visible DOM rows, supporting 50,000+ pages at 60 FPS without memory leaks. - - Columns: Status Code, URL, Page Title, Meta Description, H1, Canonical URL, Word Count, Inlinks, Outlinks, Depth, Response Time, Issues Badge. -- **Row Click**: Slides open the **Single Page Detail Drawer**. -- **Native Export**: "Export CSV" opens native OS file dialog. - ---- - -### Screen 5: Issues Explorer -Categorized, actionable list of all triggered SEO rules. - -#### UI Elements: -- **Sidebar Filters**: By Severity (Critical, Alert, Warning, Notice) and Category. -- **Issue Cards**: - - Rule Headline & Severity Badge. - - Number of affected pages badge. - - **"Why it matters"** (Business impact for clients). - - **"How to fix"** with platform tabs: - - *Code / Developer Fix*: Raw HTML / Nginx / Next.js code snippet. - - *WordPress Fix*: Step-by-step for Yoast / RankMath. - - *Webflow / Framer Fix*: Designer settings walkthrough. - - **"Copy AI Fix Prompt"**: Copies prompt ready to paste into Claude/Cursor/v0. - - **Affected URLs Table**: List of all pages triggering this issue. - ---- - -### Screen 6: Single Page Detail Drawer -Deep-dive drawer that slides out from the right: - -#### Tabs: -1. **Overview**: Status, headers, content length, depth, indexability verdict. -2. **Headings Tree**: Visual outline showing H1 through H6 hierarchy with character counts. -3. **Internal Links Graph**: Table of all incoming and outgoing links with anchor text. -4. **Structured Data (JSON-LD)**: Syntax-highlighted schema inspector with validation badges. -5. **JavaScript SEO Diff**: Side-by-side comparison of Server HTML vs. Client Rendered DOM. - ---- - -## 4. Tauri IPC Command & Event Specification - -Instead of HTTP REST and Server-Sent Events, the frontend communicates with the Rust engine via **Tauri IPC (`invoke`)** and the **Tauri Event System**. - -### 4.1 Commands (`src-tauri/src/commands.rs`) - -```rust -#[tauri::command] -async fn start_crawl(config: CrawlConfig) -> Result; - -#[tauri::command] -async fn stop_crawl(session_id: String) -> Result<(), String>; - -#[tauri::command] -async fn list_crawls() -> Result, String>; - -#[tauri::command] -async fn get_crawl_summary(session_id: String) -> Result; - -#[tauri::command] -async fn get_pages(session_id: String, filter: PageFilter) -> Result; - -#[tauri::command] -async fn get_page_details(page_id: i64) -> Result; - -#[tauri::command] -async fn get_issues(session_id: String, severity: Option) -> Result, String>; - -#[tauri::command] -async fn export_report(session_id: String, format: String, destination_path: String) -> Result<(), String>; -``` - -### 4.2 Real-Time Event Stream - -During an active crawl, the Rust engine emits real-time events to the Tauri webview: - -```typescript -// Frontend Listener (React 19 hook) -import { listen } from '@tauri-apps/api/event'; - -useEffect(() => { - const unlistenProgress = listen('crawl-progress', (event) => { - setProgress(event.payload); - }); - - const unlistenIssue = listen('crawl-issue-found', (event) => { - addLiveIssue(event.payload); - }); - - const unlistenDone = listen('crawl-completed', (event) => { - onCrawlComplete(event.payload); - }); - - return () => { - unlistenProgress.then(f => f()); - unlistenIssue.then(f => f()); - unlistenDone.then(f => f()); - }; -}, [sessionId]); -``` - ---- - -## 5. Project Directory Layout & Cargo Workspace - -The repository is structured as a **2-Member Cargo Workspace** (`.` for the engine/CLI and `src-tauri` for the desktop app). This shares the build cache, ensures a single `Cargo.lock`, and guarantees zero GUI bloat inside the headless CLI binary. - -``` -seo-lens/ -β”œβ”€β”€ Cargo.toml # Root workspace manifest & Member 1 (Engine + CLI) -β”œβ”€β”€ ui/ # React 19 Frontend (Vite + Tailwind) -β”‚ β”œβ”€β”€ package.json -β”‚ β”œβ”€β”€ vite.config.ts -β”‚ β”œβ”€β”€ tsconfig.json -β”‚ β”œβ”€β”€ tailwind.config.js -β”‚ β”œβ”€β”€ index.html -β”‚ └── src/ -β”‚ β”œβ”€β”€ components/ # Scorecards, VirtualGrid, IssueDrawer, CMSBadges -β”‚ β”œβ”€β”€ hooks/ # useTauriEvents, useCrawl -β”‚ β”œβ”€β”€ pages/ # Overview, LiveCrawl, Explorer, Issues -β”‚ └── App.tsx -β”œβ”€β”€ src-tauri/ # Member 2: Tauri v2 Desktop Wrapper -β”‚ β”œβ”€β”€ Cargo.toml # Declares: seo-lens = { path = ".." } -β”‚ β”œβ”€β”€ tauri.conf.json # Window settings, bundle IDs, icons, plugins -β”‚ β”œβ”€β”€ src/ -β”‚ β”‚ β”œβ”€β”€ main.rs # Tauri entry point -β”‚ β”‚ β”œβ”€β”€ commands.rs # #[tauri::command] IPC bindings -β”‚ β”‚ └── events.rs # Tokio -> Tauri event emitter bridge -β”‚ └── icons/ # Native app icons (.icns, .ico, .png) -β”œβ”€β”€ tests/ # Integration tests & test fixtures -└── src/ # Pure Rust Engine & CLI Implementation - β”œβ”€β”€ lib.rs # Re-usable core engine - β”œβ”€β”€ main.rs # Headless CLI (`audit`, `mcp`, `report`) - β”œβ”€β”€ core/ - β”œβ”€β”€ crawler/ - β”œβ”€β”€ parser/ - β”œβ”€β”€ rules/ - β”œβ”€β”€ storage/ - β”œβ”€β”€ graph/ - β”œβ”€β”€ mcp/ - └── report/ -``` - -### Workspace Manifest Snippets: - -**Root `Cargo.toml`**: -```toml -[workspace] -members = [ - ".", # Member 1: seo-lens core library + CLI binary - "src-tauri", # Member 2: Tauri desktop shell -] -resolver = "2" -``` - -**`src-tauri/Cargo.toml`**: -```toml -[dependencies] -seo-lens = { path = ".." } -tauri = { version = "2", features = [] } -``` - ---- - -## 6. Build & Packaging Pipeline - -### 6.1 Development Workflow -```bash -# In terminal: -cargo tauri dev -``` -- Automatically launches Vite dev server (`http://localhost:5173`) with HMR. -- Compiles the Tauri Rust desktop wrapper. -- Spawns the native desktop window with dev tools enabled. - -### 6.2 Production Release Packaging -```bash -cargo tauri build -``` -Generates native platform installers in `src-tauri/target/release/bundle/`: -- **macOS**: Universal binary `.dmg` and `.app` (supports Intel + Apple Silicon). -- **Windows**: `.msi` and `.exe` installers with desktop shortcut and uninstaller. -- **Linux**: `.AppImage` and `.deb` packages. - ---- - -## 7. Summary - -Replacing the browser-based web server with **Tauri v2** gives **SEO Lens**: -1. **Zero Terminal Barrier**: Non-technical users double-click an installer to run. -2. **Zero Port Conflicts**: No `localhost` collisions with developer environments. -3. **Audience-Specific Superpowers**: - - "Copy AI Fix Prompt" for vibe coders. - - CMS Auto-Detection & Platform Remediation for WordPress/Webflow designers. - - Executive Health Scorecard & 1-Click Exports for clients. - - High-performance virtualized grid for technical developers. -4. **Lightweight Native Desktop Footprint**: 12–18MB installer with zero Chromium bloat. diff --git a/docs/MCP_SPECIFICATION.md b/docs/MCP_SPECIFICATION.md deleted file mode 100644 index a5395d3..0000000 --- a/docs/MCP_SPECIFICATION.md +++ /dev/null @@ -1,474 +0,0 @@ -# SEO Lens: Model Context Protocol (MCP) Specification -**Document Status**: Permanent Technical Specification -**Scope**: MCP Server Architecture, Agent Tools, JSON Schemas, Resources, and Protocol Transports - ---- - -## 1. Architectural Role & Executive Overview - -The **Model Context Protocol (MCP)** integration turns `SEO Lens` from a passive reporting tool into an active, autonomous **AI Agent Assistant**. - -Rather than dumping monolithic 20MB JSON or CSV files that overwhelm LLM context windows, `SEO Lens` embeds an in-process MCP server exposing an asymmetric, asynchronous API tailored for agent workflows. - ---- - -## 2. Protocol Transports & Execution Lifecycle - -`SEO Lens` implements the standard JSON-RPC 2.0 Model Context Protocol via the `rmcp` Rust crate: - -```mermaid -flowchart TD - Agent["AI Agent (Claude / Cursor / Windsurf)"] - - subgraph Transports - Stdio["stdio Transport: seolens mcp"] - SSE["HTTP/SSE Transport: seolens mcp --transport sse --port 8080"] - end - - Agent <-->|JSON-RPC 2.0| Stdio - Agent <-->|JSON-RPC 2.0| SSE - - subgraph Server ["SEO Lens In-Binary MCP Server"] - Dispatcher["MCP Request Dispatcher & Validator"] - - subgraph ToolSet ["8 Autonomous Agent Tools"] - T1["seo_start_audit (Async, < 1s)"] - T2["seo_audit_status (Telemetry Polling)"] - T3["seo_get_markdown_report (LLM Context Format)"] - T4["seo_quick_page_check (Sync, < 500ms)"] - T5["seo_query_issues (Filtered Lookups)"] - T6["seo_check_ai_readiness (GEO & llms.txt)"] - T7["seo_validate_schema (Google Rich Results)"] - T8["seo_cleanup_session (Ephemeral Wipe)"] - end - - subgraph BackgroundTask ["Tokio Background Task"] - Engine["Dual-Engine Crawler & Rules Evaluator"] - State[(SQLite WAL State Store)] - end - end - - Dispatcher --> ToolSet - T1 -->|Spawn Task| BackgroundTask - T2 -->|Read State| State - T3 -->|Query & Format| State - T4 -->|Direct Stream| Engine -``` - -### The Non-Blocking Asynchronous Guarantee -Most LLM client applications impose a strict **30 to 60-second timeout** on tool execution. A crawl of thousands of pages can take several minutes. -- Under **no circumstances** does `seo_start_audit` block until the crawl finishes. -- It validates the target URL, spawns a background Tokio task, registers the `session_id` in SQLite, and returns in **under 1.0 second**. -- The AI agent subsequently polls `seo_audit_status(session_id)` every 15–20 seconds until `status == "completed"`, then calls `seo_get_markdown_report(session_id)`. - ---- - -## 3. The 8 Core MCP Tools Specification - -Below is the exhaustive specification of all 8 MCP tools, including input schemas, parameter constraints, and return payloads. - ---- - -### Tool 1: `seo_start_audit` -Kicks off a background crawl and technical audit for an entire website. Returns immediately with a session token. - -#### Input Schema: -```json -{ - "type": "object", - "properties": { - "url": { - "type": "string", - "format": "uri", - "description": "The root or starting URL to crawl (e.g. 'https://example.com')." - }, - "max_pages": { - "type": "integer", - "default": 500, - "minimum": 1, - "maximum": 50000, - "description": "Maximum number of pages to crawl." - }, - "max_depth": { - "type": "integer", - "default": 5, - "minimum": 1, - "maximum": 20, - "description": "Maximum click depth from the start URL." - }, - "render_js": { - "type": "boolean", - "default": false, - "description": "Enable headless Chrome CDP to render JavaScript and audit SPAs." - }, - "respect_robots": { - "type": "boolean", - "default": true, - "description": "Whether to fetch and obey /robots.txt rules." - }, - "ai_geo_audit": { - "type": "boolean", - "default": true, - "description": "Audit /llms.txt and AI bot crawler accessibility." - } - }, - "required": ["url"] -} -``` - -#### Return Payload: -```json -{ - "session_id": "c1f8a840-7e3f-42e5-a6e1-9257e84999ab", - "status": "queued", - "target_url": "https://example.com", - "message": "Audit started in background. Poll 'seo_audit_status' with session_id to monitor progress.", - "poll_interval_seconds": 15 -} -``` - ---- - -### Tool 2: `seo_audit_status` -Polls the live progress and operational telemetry of an active or finished audit. - -#### Input Schema: -```json -{ - "type": "object", - "properties": { - "session_id": { - "type": "string", - "description": "The unique audit session ID returned by seo_start_audit." - } - }, - "required": ["session_id"] -} -``` - -#### Return Payload: -```json -{ - "session_id": "c1f8a840-7e3f-42e5-a6e1-9257e84999ab", - "status": "crawling", // "queued" | "crawling" | "analyzing_graph" | "completed" | "failed" - "pages_crawled": 142, - "pages_discovered": 380, - "current_delay_ms": 125, - "p95_ttfb_ms": 340, - "error_rate_pct": 0.0, - "issues_count": { - "critical": 3, - "alert": 12, - "warning": 45, - "notice": 18 - }, - "elapsed_seconds": 45, - "is_complete": false -} -``` - ---- - -### Tool 3: `seo_get_markdown_report` -Generates a structured, concise Markdown report engineered specifically for LLM context windows. - -#### Input Schema: -```json -{ - "type": "object", - "properties": { - "session_id": { - "type": "string", - "description": "The audit session ID." - }, - "top_issues_limit": { - "type": "integer", - "default": 20, - "description": "Maximum number of distinct issue types to summarize." - }, - "include_urls": { - "type": "boolean", - "default": true, - "description": "Include sample affected URLs for each issue." - } - }, - "required": ["session_id"] -} -``` - -#### Return Payload: -Returns a pure Markdown document formatted with clear headings, severity indicators, and exact code fix instructions (see Section 4 for format). - ---- - -### Tool 4: `seo_quick_page_check` -Performs an instant, synchronous audit of a single URL in $< 500\text{ms}$. Ideal for testing a specific landing page or verifying a code fix immediately. - -#### Input Schema: -```json -{ - "type": "object", - "properties": { - "url": { - "type": "string", - "format": "uri", - "description": "Single URL to fetch and audit." - }, - "render_js": { - "type": "boolean", - "default": false, - "description": "Execute JavaScript via headless Chrome." - } - }, - "required": ["url"] -} -``` - -#### Return Payload: -```json -{ - "url": "https://example.com/pricing", - "status_code": 200, - "ttfb_ms": 145, - "title": "Pricing Plans | Example SaaS", - "meta_description": "Affordable plans for teams of all sizes.", - "h1": "Transparent Pricing", - "canonical_url": "https://example.com/pricing", - "word_count": 840, - "is_indexable": true, - "issues_detected": [ - { - "code": "WARN_IMG_MISSING_ALT", - "severity": "warning", - "message": "2 images missing alt attributes." - }, - { - "code": "WARN_SECURITY_MISSING_HSTS", - "severity": "warning", - "message": "Missing Strict-Transport-Security header." - } - ] -} -``` - ---- - -### Tool 5: `seo_query_issues` -Queries specific issues from an audit database, filtered by severity, category, or URL pattern. - -#### Input Schema: -```json -{ - "type": "object", - "properties": { - "session_id": { - "type": "string", - "description": "The audit session ID." - }, - "severity": { - "type": "string", - "enum": ["critical", "alert", "warning", "notice"], - "description": "Optional severity filter." - }, - "category": { - "type": "string", - "description": "Optional category filter (e.g. 'canonicalization', 'security', 'indexability')." - }, - "url_pattern": { - "type": "string", - "description": "Optional SQL LIKE or glob pattern (e.g. '%/blog/%')." - }, - "limit": { - "type": "integer", - "default": 50, - "maximum": 500, - "description": "Number of records to return." - } - }, - "required": ["session_id"] -} -``` - -#### Return Payload: -```json -{ - "total_matching": 4, - "issues": [ - { - "code": "ERR_CANONICAL_TO_4XX_5XX", - "severity": "critical", - "category": "canonicalization", - "target_url": "https://example.com/old-page", - "message": "Canonical points to broken URL returning 404: https://example.com/deleted", - "source_page_url": null - } - ] -} -``` - ---- - -### Tool 6: `seo_check_ai_readiness` -Audits whether a website is optimized for Generative Engine Optimization (GEO) and AI Search Engines (ChatGPT Search, Perplexity, Claude). - -#### Input Schema: -```json -{ - "type": "object", - "properties": { - "url": { - "type": "string", - "format": "uri", - "description": "The website base URL." - } - }, - "required": ["url"] -} -``` - -#### Return Payload: -```json -{ - "base_url": "https://example.com", - "llms_txt_found": true, - "llms_full_txt_found": false, - "ai_crawler_access": { - "retrieval_citation_bots": { - "OAI-SearchBot": "ALLOWED", - "ChatGPT-User": "ALLOWED", - "PerplexityBot": "DISALLOWED", - "Claude-User": "ALLOWED" - }, - "training_bots": { - "GPTBot": "DISALLOWED", - "ClaudeBot": "DISALLOWED", - "Google-Extended": "DISALLOWED" - } - }, - "citation_search_risk": "HIGH", - "recommendations": [ - "PerplexityBot is blocked in robots.txt. Your site will not be cited as a source in Perplexity answers. Allow PerplexityBot to regain visibility." - ] -} -``` - ---- - -### Tool 7: `seo_validate_schema` -Validates a raw JSON-LD or Schema.org block against Google Rich Results eligibility rules. - -#### Input Schema: -```json -{ - "type": "object", - "properties": { - "json_ld": { - "type": "string", - "description": "The raw JSON-LD string or object snippet." - }, - "target_type": { - "type": "string", - "description": "Optional expected @type (e.g. 'Product', 'Article', 'FAQPage')." - } - }, - "required": ["json_ld"] -} -``` - -#### Return Payload: -```json -{ - "is_valid_json": true, - "detected_type": "Product", - "is_rich_result_eligible": false, - "missing_required_fields": ["offers"], - "missing_recommended_fields": ["review", "aggregateRating"], - "error_message": "Missing required Google Rich Result property 'offers' (price and availability)." -} -``` - ---- - -### Tool 8: `seo_cleanup_session` -Deletes audit records and artifacts from disk for a session. - -#### Input Schema: -```json -{ - "type": "object", - "properties": { - "session_id": { - "type": "string", - "description": "The audit session ID to purge." - } - }, - "required": ["session_id"] -} -``` - -#### Return Payload: -```json -{ - "session_id": "c1f8a840-7e3f-42e5-a6e1-9257e84999ab", - "purged": true, - "bytes_freed": 1420800 -} -``` - ---- - -## 4. LLM-Optimized Markdown Report Template - -When an AI agent requests the report via `seo_get_markdown_report`, the output must be formatted for **maximum context efficiency** and **actionable code changes**: - -```markdown -# Technical SEO Audit: example.com -**Health Score**: 78/100 | **Pages Crawled**: 450 | **Duration**: 42s -**Summary**: 2 Critical Issues, 5 Alerts, 14 Warnings - ---- - -## 🚨 Critical Issues (Immediate Fix Required) - -### 1. `ERR_CANONICAL_TO_4XX_5XX` (Affects 4 pages) -- **Problem**: `` points to a 404 dead page. -- **Affected URLs**: - - `https://example.com/products/shoes` -> `https://example.com/shoes-old` (404) -- **Action for Agent**: Update canonical tag in `src/pages/products/[slug].tsx` to point to self or active product URL. - -### 2. `ERR_SECURITY_INSECURE_FORM` (Affects 1 page) -- **Problem**: Login `
` on HTTPS submits to `http://api.example.com/login`. -- **Affected URL**: `https://example.com/login` -- **Action for Agent**: Update form `action` to `https://api.example.com/login`. - ---- - -## ⚠️ High-Priority Alerts - -### 1. `ALERT_GEO_AI_RETRIEVAL_BOT_BLOCKED` -- **Problem**: `robots.txt` disallows `PerplexityBot`. Site is blocked from AI Search citations. -- **Action for Agent**: Edit `public/robots.txt` and remove `Disallow: /` under `User-agent: PerplexityBot`. - ---- - -## πŸ’‘ Quick Wins for Agent -1. Add missing `` to 12 landing pages. -2. Add `alt` attributes to 8 product thumbnails in `components/ProductCard.tsx`. -3. Add `Strict-Transport-Security: max-age=31536000` header in server config. -``` - ---- - -## 5. MCP Resources Specification - -In addition to tools, `SEO Lens` exposes read-only MCP URI resources: -- `seo://crawls`: Lists all recent crawl sessions in SQLite. -- `seo://crawls/{session_id}/summary`: JSON summary of crawl metrics and health score. -- `seo://crawls/{session_id}/report`: Full Markdown report. -- `seo://crawls/{session_id}/issues`: Raw JSON array of all issues. - ---- - -## 6. Summary - -This specification gives AI agents a complete, non-blocking interface to inspect, diagnose, and remediate technical SEO defects on any website. diff --git a/docs/PRODUCTION_SPEC.md b/docs/PRODUCTION_SPEC.md deleted file mode 100644 index 2c85805..0000000 --- a/docs/PRODUCTION_SPEC.md +++ /dev/null @@ -1,396 +0,0 @@ -# SEO Lens: Production Specification & Engineering Roadmap - -**Document Status**: Active / Long-Term Living Specification -**Architecture**: Unified 2-Member Cargo Workspace (`.` for Core Library & Headless CLI, `src-tauri` for Native Desktop App) -**Development Methodology**: Strict Test-Driven Development (TDD) across Micro-Phases - ---- - -## 1. System Overview & Production Goals - -`SEO Lens` is a high-performance, local-first website crawler, technical SEO audit engine, and AI-native auditor written in Rust. It is built to run reliably on client websites, local developer machines, production Linux servers, Docker containers, and autonomous AI agent environments. - -### Core Production Requirements - -1. **Resilience**: The crawler must never crash or panic on malformed HTML, circular redirects, connection resets, TLS handshake errors, or malicious payloads (e.g. decompression bombs or infinite calendar loops). -2. **Politeness & Safety**: The system must actively prevent origin overloading. An adaptive congestion controller must throttle requests dynamically based on origin latency and error spikes, strictly respecting `robots.txt` directives. -3. **Accuracy**: Audit rules must avoid false positives. Edge cases (e.g., `` tags inside ``, relative canonical URLs, soft 404s, WAF challenge pages) must be explicitly classified. -4. **AI-Agent Readiness (MCP)**: Native Model Context Protocol (MCP) server running in-process, exposing non-blocking asynchronous tools for AI coding assistants and autonomous agents. -5. **Zero-Bloat Multi-Target Deployment**: - - Headless CLI / MCP single executable (`seolens audit`, `seolens mcp`) for developers, CI/CD, and AI agents. - - Native Desktop Application powered by Tauri v2 (`.dmg`, `.msi`, `.AppImage`) for non-technical clients, WordPress/Webflow creators, and vibe coders. - - Zero external runtime dependencies (no Node.js or Python needed for the end user). - ---- - -## 2. Engineering Standards & TDD Protocol - -Every module in `SEO Lens` will be implemented strictly using **Test-Driven Development (TDD)**: - -``` -β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β” -β”‚ TDD Cycle β”‚ -β”‚ β”‚ -β”‚ [1. Write Failing Test] ──> [2. Minimal Implementation] β”‚ -β”‚ β–² β”‚ β”‚ -β”‚ β”‚ β–Ό β”‚ -β”‚ [4. Test Verification] <── [3. Refactor & Lint] β”‚ -β”‚β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜ -``` - -### Protocol Rules: - -1. **Tests First**: No implementation code is written without a corresponding automated test fixture defining the expected behavior. -2. **Automated + Manual Verification**: Each micro-phase must satisfy both: - - Automated unit and integration test passes (`cargo test`). - - Manual verification (CLI invocation, inspecting data output, edge-case testing) reviewed collaboratively before advancing. -3. **No Unverified Assumptions**: Performance, memory, and binary sizes are treated as empirical engineering metrics to be measured, profiled, and optimized during testingβ€”not assumed upfront. -4. **Typed Error Handling**: All library functions return structured `Result` using `thiserror`. Panics (`unwrap()`, `expect()`) are forbidden in library code paths. - ---- - -## 3. Modular Architecture Topology (2-Member Cargo Workspace) - -The project uses a **2-Member Cargo Workspace**. This architecture ensures: - -- **Unified Build Cache**: One root `target/` directory and one `Cargo.lock`. Shared dependencies (`tokio`, `reqwest`, `serde`, `rusqlite`) are compiled once, preventing 5–10 GB of duplicate build artifacts. -- **Zero GUI Bloat in CLI**: Desktop/Tauri dependencies (`tauri`, `webkit2gtk`) only exist in `src-tauri/Cargo.toml`. The CLI binary (`seolens`) remains ultra-lean (~15MB) with zero OS GUI dependencies for CI/CD and Docker. -- **Frictionless Development**: Phases 0 through 9 are built directly in `src/` and `tests/`. Phase 10 seamlessly connects `src-tauri` to `seo-lens` via `path = ".."`. - -### Directory Layout: - -``` -seo-lens/ -β”œβ”€β”€ Cargo.toml # Root workspace manifest & Member 1 (Engine + CLI) -β”œβ”€β”€ ui/ # React 19 + Tailwind + Vite Desktop UI -β”‚ β”œβ”€β”€ package.json -β”‚ β”œβ”€β”€ vite.config.ts -β”‚ β”œβ”€β”€ index.html -β”‚ └── src/ # Virtualized tables, charts, live telemetry UI -β”œβ”€β”€ src-tauri/ # Member 2: Tauri v2 Native Desktop Wrapper -β”‚ β”œβ”€β”€ Cargo.toml # Depends on: seo-lens = { path = ".." } -β”‚ β”œβ”€β”€ tauri.conf.json # Desktop window configuration & app icons -β”‚ └── src/ -β”‚ β”œβ”€β”€ main.rs # Tauri desktop entry point -β”‚ β”œβ”€β”€ commands.rs # #[tauri::command] IPC bindings to core engine -β”‚ └── events.rs # Real-time event bridge (app.emit) -β”œβ”€β”€ tests/ # Integration tests & fixtures -β”‚ β”œβ”€β”€ fixtures/ # Synthetic HTML, sitemaps, robots.txt files -β”‚ β”œβ”€β”€ crawl_tests.rs # End-to-end crawling test suite -β”‚ β”œβ”€β”€ rules_tests.rs # SEO rules validation suite -β”‚ └── mcp_tests.rs # MCP JSON-RPC protocol test suite -└── src/ # Core Library & CLI Implementation - β”œβ”€β”€ main.rs # Headless CLI entry point (`audit`, `mcp`, `report`) - β”œβ”€β”€ lib.rs # Library root exporting public API - β”œβ”€β”€ core/ # Domain models, URL normalization, configuration - β”œβ”€β”€ crawler/ # Async HTTP engine, AIMD politeness, frontier, browser CDP - β”œβ”€β”€ parser/ # lol_html streaming parser, metadata, schema, content - β”œβ”€β”€ rules/ # 120 technical SEO checks (single-page, graph, JS diff) - β”œβ”€β”€ graph/ # petgraph link topology and internal PageRank - β”œβ”€β”€ storage/ # SQLite WAL persistence and batch operations - β”œβ”€β”€ mcp/ # Model Context Protocol stdio & SSE server - └── report/ # Markdown (LLM), JSON, CSV, and HTML exporters -``` - -### Workspace Manifest Definition (`Cargo.toml`): - -```toml -[workspace] -members = [ - ".", # Member 1: Core engine library + CLI binary - "src-tauri", # Member 2: Tauri desktop application wrapper -] -resolver = "2" - -[package] -name = "seo-lens" -version = "0.1.0" -edition = "2021" - -[lib] -name = "seo_lens" -path = "src/lib.rs" - -[[bin]] -name = "seolens" -path = "src/main.rs" -``` - -And in `src-tauri/Cargo.toml`: - -```toml -[package] -name = "seo-lens-desktop" -version = "0.1.0" -edition = "2021" - -[dependencies] -seo-lens = { path = ".." } -tauri = { version = "2", features = [] } -``` - ---- - -## 4. Step-by-Step Implementation Roadmap (Micro-Phases) - ---- - -### Phase 0: Project Scaffolding, Tooling & Test Harness - -**Objective**: Set up the Rust 2-member workspace, linting rules, dependency baselines, and test fixture infrastructure. - -#### Tasks: - -1. Initialize root `Cargo.toml` with `[workspace] members = [".", "src-tauri"]` and core dependencies: `tokio`, `serde`, `serde_json`, `thiserror`, `anyhow`, `tracing`, `clap`, `compact_str`, `bitflags`. -2. Configure Cargo release profile (`opt-level = "z"`, `lto = true`, `strip = true`). -3. Set up test fixture directory (`tests/fixtures/`) and mock HTTP server harness (`wiremock`). -4. Establish `src/lib.rs` and `src/main.rs` with baseline logging. - -#### Verification Gate: - -- **Automated**: `cargo test` executes cleanly. `cargo clippy` passes with zero warnings. -- **Manual**: Run `seolens --help` and verify CLI help output. - ---- - -### Phase 1: Core Domain Models & URL Canonicalization - -**Objective**: Build reliable, high-performance URL normalization and domain modeling. - -#### Tasks: - -1. `core/url.rs`: URL normalizer (strip tracking params, normalize schemes, resolve relative paths, lowercase hostnames, handle trailing slash consistency). -2. `core/models.rs`: Core types: `PageReport`, `DiscoveredLink`, `Issue`, `Severity` (Critical, Alert, Warning), `IssueCategory`, `CrawlSummary`. -3. `core/config.rs`: Crawl parameters (depth, concurrency, delay, headers, user-agents, proxy). - -#### Verification Gate: - -- **Automated (TDD)**: Test suite with 40+ URL edge cases (query strings, port numbers, IPv4/IPv6, punycode, non-ASCII paths, fragments). -- **Manual**: CLI command testing URL normalization inputs and verifying deterministic outputs. - ---- - -### Phase 2: Streaming HTML Parsing Engine (`lol_html`) - -**Objective**: Zero-copy extraction of HTML tags, metadata, and body content without loading full DOM trees into memory. - -#### Tasks: - -1. `parser/streaming.rs`: `lol_html` streaming rewriter setup with pre-compiled CSS selectors for ``, `<meta>`, `<link>`, `<h1>`-`<h6>`, `<a>`, `<img>`, `<script>`. -2. `parser/metadata.rs`: Extraction of canonical URLs, OpenGraph, Twitter Cards, charset, viewport, robots directives. -3. `parser/content.rs`: Text token extraction isolating editorial content (ignoring `<nav>`, `<header>`, `<footer>`, `<script>`, `<style>`). -4. `parser/schema.rs`: Extraction of JSON-LD (`<script type="application/ld+json">`) and Microdata attributes. - -#### Verification Gate: - -- **Automated (TDD)**: Synthetic HTML test fixtures verifying correct extraction on malformed HTML, unclosed tags, and deeply nested structures. -- **Manual**: Pass a saved HTML page from a major website through the parser and verify complete extracted metadata JSON. - ---- - -### Phase 3: Asynchronous HTTP Fetcher & AIMD Politeness Controller - -**Objective**: Resilient network fetching engine with adaptive rate limiting to prevent origin overloading. - -#### Tasks: - -1. `crawler/client.rs`: `reqwest` HTTP/2 client wrapper with custom redirect policies, timeout handling, and transport error classification (DNS failure, SSL error, connection refused). -2. `crawler/aimd.rs`: Additive-Increase/Multiplicative-Decrease congestion controller (tunes delay based on error rates and latency percentiles). -3. `crawler/waf.rs`: Fingerprint detection for Cloudflare, Akamai, DataDome, and Imperva bot challenge screens. - -#### Verification Gate: - -- **Automated (TDD)**: Wiremock tests simulating 429 rate limits, 503 gateway timeouts, and response latency spikes to verify AIMD backoff and recovery. -- **Manual**: Run fetch against test endpoint with artificial rate limits and observe smooth adaptive throttling. - ---- - -### Phase 4: Frontier Queue, Depth Traversal & Robots/Sitemap Engine - -**Objective**: Robust crawl frontier managing discovery queues, depth boundaries, `robots.txt`, and XML sitemaps. - -#### Tasks: - -1. `crawler/frontier.rs`: Deduplication hash set (`hashbrown`), BFS/DFS queue management, max-pages and max-depth enforcement. -2. `crawler/robots.rs`: RFC 9309 compliant `robots.txt` parser with support for User-Agent matching and `Crawl-Delay`. -3. `crawler/sitemap.rs`: Streaming XML sitemap parser (`quick-xml`) supporting sitemap indexes, compressed `.xml.gz`, and alternate hreflang entries. - -#### Verification Gate: - -- **Automated (TDD)**: Complex `robots.txt` directive tests (wildcards, disallow vs allow precedence) and nested sitemap index parsing. -- **Manual**: Crawl a multi-level mock site; verify that max depth and URL exclusion rules are strictly respected. - ---- - -### Phase 5: Technical SEO Rules Engine β€” Phase 1 (Single-Page In-Flight Rules) - -**Objective**: Implement 80+ immediate document-level checks evaluated as pages are fetched. - -#### Tasks: - -1. `rules/catalog.rs`: Master issue dictionary with unique codes, severity tiers (Critical, Alert, Warning), and human-readable descriptions. -2. `rules/page/titles.rs` & `descriptions.rs`: Length, absence, multiple tags, whitespace irregularities. -3. `rules/page/headings.rs`: Missing H1, multiple H1, empty H1, heading hierarchy skipping. -4. `rules/page/status.rs`: HTTP 4xx, 5xx, 3xx redirects, timeouts, SSL handshake failures. -5. `rules/page/security.rs`: Missing HSTS, CSP, X-Frame-Options, mixed content resources, insecure form actions. -6. `rules/page/mobile.rs` & `images.rs`: Viewport presence, missing image `alt`, missing `width`/`height` dimensions (CLS). -7. `rules/page/schema_val.rs`: Validation of JSON-LD schemas against Google Rich Results guidelines (Article, Product, FAQ, LocalBusiness, Breadcrumb). -8. `rules/page/geo.rs`: Checking `/llms.txt` presence and evaluating `robots.txt` for AI Training Bots vs. AI Retrieval/Search Bots. - -#### Verification Gate: - -- **Automated (TDD)**: Unit tests for each rule module with positive (violating) and negative (passing) HTML fixtures. -- **Manual**: Run audit against an intentionally broken test site and verify that all intentional issues are detected. - ---- - -### Phase 6: Site Graph Topology & Phase 2 Rules (Multi-Page Graph Checks) - -**Objective**: Build a directed link graph to execute post-crawl, site-wide architectural analysis. - -#### Tasks: - -1. `graph/graph.rs`: Directed internal link graph (`petgraph`) mapping source pages to target pages with link attributes (nofollow, anchor text). -2. `graph/pagerank.rs`: Power-iteration internal link equity (PageRank) calculation. -3. `rules/graph/orphans.rs`: Detection of pages found in XML sitemaps with zero incoming internal links. -4. `rules/graph/duplicates.rs`: SHA256 exact duplicate content and SimHash/MinHash near-duplicate detection. -5. `rules/graph/canonicals.rs` & `redirects.rs`: Canonical chains (>1 hop), canonical loops, redirect chains, and circular redirect loops. -6. `rules/graph/hreflang.rs`: Reciprocal bidirectional return tag validation across languages. - -#### Verification Gate: - -- **Automated (TDD)**: Graph fixture tests verifying correct identification of orphan nodes, cycle detection in redirects, and broken hreflang pairs. -- **Manual**: Verify graph metrics on a simulated site with known orphan pages and redirect loops. - ---- - -### Phase 7: Headless Browser CDP Engine & JavaScript SEO Diffing (`--render-js`) - -**Objective**: Optional Chrome DevTools Protocol integration to audit client-rendered SPAs and compare raw HTML vs. rendered DOM. - -#### Tasks: - -1. `crawler/browser.rs`: `chromiumoxide` CDP manager behind `[features] js-render`. Auto-detects local Chrome/Brave/Edge or connects via `--chrome-ws`. -2. `crawler/diff.rs`: JavaScript SEO Diffing engine comparing initial server HTML with client-rendered DOM. -3. `rules/page/js_diff.rs`: Rules flagging client-side canonical modification, dynamic noindex injection, title/H1 overwriting, and hydration crashes. -4. SPA Heuristic warning in raw HTTP mode when unrendered SPAs are detected. - -#### Verification Gate: - -- **Automated (TDD)**: Test fixture comparing raw vs rendered HTML with simulated JS modifications. -- **Manual**: Audit a client-rendered React/Vue page with and without `--render-js` to observe the rendered DOM differences. - ---- - -### Phase 8: SQLite Persistence & Storage Layer - -**Objective**: High-throughput persistent storage for audit histories, client projects, and resume support. - -#### Tasks: - -1. `storage/schema.sql`: Relational tables for projects, audits, pages, issues, links, and resources with indexed URL hashes. -2. `storage/sqlite.rs`: SQLite connection pool in WAL mode (`rusqlite`) with batched transactions (250 pages/batch). -3. Query API for filtering issues by category, severity, or URL path. -4. Ephemeral mode support (auto-cleanup for temporary runs). - -#### Verification Gate: - -- **Automated (TDD)**: In-memory and on-disk SQLite migration and batch insert tests under load. -- **Manual**: Query the resulting SQLite database via CLI to verify relational integrity and query performance. - ---- - -### Phase 9: Model Context Protocol (MCP) Server Implementation - -**Objective**: Native in-process MCP server allowing AI agents to perform audits without external wrappers or timeouts. - -#### Tasks: - -1. `mcp/server.rs`: Stdio and HTTP/SSE JSON-RPC 2.0 transport using `rmcp`. -2. `mcp/tools.rs`: Implementation of non-blocking tools: - - `seo_start_audit`: Launches background audit, returns `session_id` immediately. - - `seo_audit_status`: Live progress polling. - - `seo_get_markdown_report`: Concise Markdown report for LLM ingestion. - - `seo_quick_page_check`: Synchronous single-URL check. - - `seo_query_issues`: Query issues with filters. - - `seo_check_ai_readiness`: Dedicated GEO `/llms.txt` and AI bot check. -3. `mcp/resources.rs`: MCP URI resources for audit summaries and issue logs. - -#### Verification Gate: - -- **Automated (TDD)**: End-to-end JSON-RPC test simulating an MCP client handshaking, tool listing, and non-blocking audit execution. -- **Manual**: Connect Claude Desktop or Cursor to `seolens mcp` via stdio and execute an audit via natural language. - ---- - -### Phase 10: Native Desktop Application (Tauri v2 + React 19 + Tailwind CSS) - -**Objective**: Build the cross-platform native desktop application providing a double-clickable GUI for non-technical users, WordPress/Webflow designers, vibe coders, and technical developers. - -#### Tasks: - -1. `ui/`: Initialize Vite + React 19 + TypeScript + Tailwind CSS application. -2. Build UI views tailored for all target personas: - - **Audits Overview & Launch Center**: URL input, preset selectors, AI audit toggle. - - **Live Telemetry Ring**: Animated % radial gauge, AIMD delay, live TTFB, real-time issue counters. - - **Executive Scorecard**: 0–100 health gauge, status breakdown donut, 6-pillar radar, and CMS banner. - - **All Pages Explorer**: High-performance 50,000+ row virtualized grid via `@tanstack/react-virtual`. - - **Issues Explorer**: Remediation tabs (Code vs. WordPress vs. Webflow/Framer) + **"Copy AI Fix Prompt"** button for vibe coders. - - **Single Page Detail Drawer**: Headings hierarchy tree, inlinks/outlinks graph, JSON-LD schema inspector, and side-by-side JS SEO Diff. -3. `src-tauri/`: Initialize Tauri v2 desktop wrapper with `tauri-plugin-dialog` (native file export) and `tauri-plugin-notification` (OS alerts). -4. `src-tauri/src/commands.rs`: Implement IPC handlers (`start_crawl`, `stop_crawl`, `list_crawls`, `get_pages`, `get_issues`, `export_report`) directly invoking `src/lib.rs`. -5. `src-tauri/src/events.rs`: Bridge real-time Tokio crawler telemetry to frontend via `app.emit("crawl-progress")`. - -#### Verification Gate: - -- **Automated (TDD)**: Tauri IPC command deserialization tests, live event emission unit tests, and virtualized table performance benchmarks. -- **Manual**: Run `cargo tauri dev`, launch an audit on a test domain, verify live progress events, test "Copy AI Fix Prompt", and export CSV via the native OS file picker. - ---- - -### Phase 11: CLI Reporting & Exporters - -**Objective**: Polished command-line user interfaces and multi-format export files. - -#### Tasks: - -1. `report/terminal.rs`: Live ANSI terminal dashboard (`indicatif`) with progress indicators and issue severity tables. -2. `report/markdown.rs`: Executive summary formatted for human reading and LLM context windows. -3. `report/json.rs`: Full structured data export. -4. `report/csv.rs`: Screaming Frog compatible CSV exports (`internal_all.csv`, `issues_all.csv`, `response_codes.csv`, `external_all.csv`). -5. `report/html.rs`: Standalone self-contained single-page offline HTML report. - -#### Verification Gate: - -- **Automated (TDD)**: CSV/JSON schema validation and markdown formatting tests. -- **Manual**: Run full CLI audit on a test domain, inspect terminal output, and verify generated CSVs open cleanly in Excel/Google Sheets. - ---- - -### Phase 12: Production Hardening, Cross-Platform Packaging & Docker - -**Objective**: Final production readiness, cross-compilation, desktop installers, and containerization. - -#### Tasks: - -1. Compiler optimization profile validation. -2. Static compilation verification (`x86_64-unknown-linux-musl`, macOS Apple Silicon/Intel, Windows `.exe`). -3. Tauri desktop packaging (`cargo tauri build`): - - macOS `.dmg` and `.app` bundle. - - Windows `.msi` and `.exe` installer. - - Linux `.AppImage` and `.deb`. -4. Multi-stage Dockerfile: - - `seolens:slim`: Scratch/Alpine base containing only the static headless CLI binary. - - `seolens:full`: Debian Slim with pre-installed Chromium for out-of-the-box JS rendering. -5. Final end-to-end integration test suite across CLI (`audit`), MCP (`stdio`), and Desktop (`Tauri`) modes. - -#### Verification Gate: - -- **Automated**: Full integration test suite passing on all target platforms in CI. -- **Manual**: Run headless CLI in Docker, test MCP server with Cursor/Claude Desktop, and launch the compiled native desktop installer (`.dmg`/`.msi`). - ---- - -## 5. Summary - -This document serves as our binding engineering contract. Each micro-phase must be built strictly test-first, verified through both automated tests and manual inspection, and signed off before proceeding to the next. diff --git a/docs/README.md b/docs/README.md new file mode 100644 index 0000000..e61d42a --- /dev/null +++ b/docs/README.md @@ -0,0 +1,50 @@ +# SEO Lens Documentation Hub + +Welcome to the technical documentation for **SEO Lens**β€”a high-performance, local-first website crawler, 120-rule technical SEO audit engine, and AI-native auditor written in Rust. + +Whether you are auditing a client website, connecting an AI coding agent via MCP, or contributing new features to the core engine, the guides below cover every aspect of the system. + +--- + +## Documentation Guides + +| Guide | Description | Target Audience | +| --- | --- | --- | +| [**Architecture & Roadmap**](./architecture.md) | 2-member workspace layout, asynchronous pipeline design, streaming parser (`lol_html`), and milestone progress. | Contributors, Systems Engineers | +| [**CLI Commands & Flags**](./cli.md) | Reference for all 10 CLI subcommands (`audit`, `inspect`, `mcp`, `report`, `issues`, etc.), flags, and exporters. | Users, DevOps, Automation | +| [**Crawler Engine & AIMD**](./crawler.md) | Asynchronous crawler mechanics, AIMD rate tuning, 8-stage URL normalization, RFC 9309 robots, and streaming XML sitemaps. | Contributors, Network Engineers | +| [**Model Context Protocol (MCP)**](./mcp.md) | Pure Rust stdio MCP server for AI coding agents (Claude, Cursor, Windsurf) with 8 tools and one-click setup prompt. | AI Engineers, Agent Developers | +| [**120 SEO Rules Catalog**](./rules.md) | Complete dictionary of all 120 technical SEO checks across 13 categories, detection heuristics, and remediation guidance. | SEO Specialists, Web Developers | +| [**Storage & SQLite Schema**](./storage.md) | Local-first persistence layer, asynchronous batch writer actor, 7 relational tables, and power-user SQL query cheatsheet. | Database Admins, Power Users | + +--- + +## Recommended Reading Pathways + +### "I want to audit a website from the command line" + +1. Read [CLI Commands & Flags](./cli.md) for usage examples and output options. +2. Review [120 SEO Rules Catalog](./rules.md) to interpret issue codes and remediation advice. + +### "I want my AI assistant (Claude, Cursor, Windsurf) to audit websites" + +1. Head to [Model Context Protocol (MCP)](./mcp.md). +2. Copy the prompt to instruct your agent to register `seolens mcp`. + +### "I want to contribute code or add a new SEO rule" + +1. Read [Architecture & Roadmap](./architecture.md) to understand module boundaries and data flow. +2. Read [CONTRIBUTING.md](../CONTRIBUTING.md) for workflow, testing (`cargo nextest`), and coding standards. +3. Check [120 SEO Rules Catalog](./rules.md) to see where your new rule fits into the catalog. + +### "I want to inspect or extract audit data directly" + +1. Open [Storage & SQLite Schema](./storage.md) to see table definitions. +2. Query `seolens.db` using the pre-built SQL examples or export to CSV with `seolens report <SESSION_ID> -f csv`. + +--- + +## Documentation Standards + +- All documentation in this folder uses lowercase filenames and relative links. +- Specifications reflect the current implementation in `src/` and are maintained as living documentation. diff --git a/docs/architecture.md b/docs/architecture.md new file mode 100644 index 0000000..6165f3d --- /dev/null +++ b/docs/architecture.md @@ -0,0 +1,175 @@ +# SEO Lens Architecture & Engineering Guide + +Welcome to the SEO Lens codebase! This document provides an architectural tour of SEO Lens: why it exists, how it is organized, how data flows through the system, and how you can contribute effectively. + +--- + +## 1. What is SEO Lens? + +**SEO Lens** is a high-performance, local-first website crawler, 120-rule technical SEO audit engine, and AI-native auditor written in Rust. + +### The Problem It Solves + +Traditional SEO crawlers (like Screaming Frog or Sitebulb) are heavy, memory-hungry desktop programs, while modern cloud crawlers are expensive and send your client data to third-party servers. On the developer side, open-source crawlers written in Python or Node.js are often slow, struggle on sites with thousands of URLs, and consume gigabytes of RAM. + +### The SEO Lens Approach + +- **Blazingly Fast & Lightweight**: Crawls ~500-800 pages/second on raw HTTP using under 50 MB of RAM. +- **Local-First & Private**: Runs entirely on your machine. Audits are stored in a local SQLite database in Write-Ahead Logging (WAL) mode. +- **Smart Dual-Engine**: Ultra-fast HTTP streaming by default, with opt-in Chrome DevTools Protocol (CDP) for JavaScript-heavy Single Page Applications (SPAs). +- **Two-Phase Audit Engine**: 120 rules split into immediate single-page checks (metadata, headings, status codes, schema) and post-crawl graph synthesis (orphan pages, canonical loops, internal PageRank). +- **AI-Native (MCP)**: Native Model Context Protocol server over `stdio`, allowing AI agents (Claude, Cursor, Windsurf) to launch audits, check progress, and query issues autonomously. +- **Zero Port Conflicts**: Distributed as a standalone CLI executable and an upcoming native desktop application shell. + +--- + +## 2. Workspace Structure (2-Member Cargo Workspace) + +SEO Lens is organized as a **2-member Cargo workspace**: + +```bash +seo-lens/ +β”œβ”€β”€ Cargo.toml # Workspace manifest + Member 1 (Core Engine & CLI) +β”œβ”€β”€ src/ # Core engine library & headless CLI +β”‚ β”œβ”€β”€ lib.rs # Library root (seo_lens crate) +β”‚ β”œβ”€β”€ main.rs # Headless CLI entry point (seolens binary) +β”‚ β”œβ”€β”€ cli/ # Command-line interface (10 subcommands via clap v4) +β”‚ β”œβ”€β”€ core/ # Domain models, URL normalization, crawl config +β”‚ β”œβ”€β”€ crawler/ # HTTP client, AIMD politeness, frontier, robots/sitemaps +β”‚ β”œβ”€β”€ parser/ # Streaming HTML tokenizer (lol_html), metadata, schema +β”‚ β”œβ”€β”€ rules/ # 120 Technical SEO audit rules (single-page & graph) +β”‚ β”œβ”€β”€ graph/ # petgraph internal link topology & PageRank +β”‚ β”œβ”€β”€ storage/ # SQLite WAL persistence layer +β”‚ β”œβ”€β”€ mcp/ # Native Model Context Protocol (MCP) stdio server +β”‚ └── report/ # Exporters (interactive TUI HTML, CSVs, Markdown, JSON) +β”œβ”€β”€ src-tauri/ # Member 2: Tauri v2 desktop application wrapper +β”‚ β”œβ”€β”€ Cargo.toml # Desktop shell manifest (depends on seo-lens = { path = ".." }) +β”‚ └── src/ # Desktop entry point and native OS window glue +β”œβ”€β”€ tests/ # Integration test suites and synthetic HTML fixtures +β”‚ β”œβ”€β”€ fixtures/ # Sample HTML pages, sitemaps, and robots.txt files +β”‚ β”œβ”€β”€ crawl_tests.rs # Mock HTTP crawling tests (wiremock) +β”‚ β”œβ”€β”€ rules_tests.rs # 120-rule validation test suite +β”‚ β”œβ”€β”€ mcp_tests.rs # MCP JSON-RPC protocol tests +β”‚ └── cli_tests.rs # CLI subcommand and reporting tests +└── docs/ # Specifications and architectural reference guides +``` + +### Why a 2-Member Workspace? + +1. **Zero GUI Bloat in the CLI**: OS desktop libraries (`tauri`, `webkit2gtk`) only live inside `src-tauri`. The CLI binary (`seolens`) compiles to a lean, standalone binary (~15 MB) ideal for servers, CI/CD, and Docker containers. +2. **Shared Build Cache**: Core dependencies (`tokio`, `reqwest`, `serde`, `rusqlite`) are compiled once in the shared `target/` directory, saving disk space and compile times. +3. **One Engine, Multiple Interfaces**: The core engine in `src/` can be invoked by the CLI (`src/main.rs`), an AI coding assistant via MCP (`src/mcp/`), or the native desktop GUI (`src-tauri`). + +--- + +## 3. High-Level System Architecture & Pipeline + +![SEO Lens Architecture & Pipeline](./assets/architecture_pipeline.png) + +The entire SEO Lens engine operates as an asynchronous, staged pipeline designed for maximum throughput and minimal memory consumption: + +1. **Interfaces & Clients**: Invocations arrive via the headless CLI (`seolens audit`, `inspect`), an AI coding agent via the Model Context Protocol (`seolens mcp`), or the desktop application shell. +2. **Concurrency & Crawler Engine**: Discovered URLs enter the **Frontier Queue**. Asynchronous Tokio worker tasks fetch targets using connection-pooled HTTP/2 while the **AIMD Politeness Controller** dynamically scales delays and concurrency to protect origin servers. +3. **Streaming Parser (`lol_html`)**: HTML streams are parsed on the fly using zero-copy CSS selector handlers. Tags, links, and schemas are extracted immediately without building heavy in-memory DOM trees. +4. **Two-Phase Rules Engine**: + - **Phase 1 (In-Flight)**: Instant document-level checks for titles, headings, directives, security headers, and schema completeness. + - **Phase 2 (Post-Crawl Graph)**: The full site topology is ingested into an in-memory directed graph (`petgraph`) to detect orphan pages, redirect loops, and calculate internal PageRank equity. +5. **Persistence & Multi-Format Reporting**: Audits are committed in atomic batches to SQLite WAL storage and exported as interactive TUI HTML reports, Screaming Frog CSVs, LLM-optimized Markdown, or structured JSON. + +--- + +## 4. Core Subsystems Explained + +### 4.1 URL Normalization Pipeline (`src/core/url.rs`) + +To prevent infinite crawl loops and duplicate fetching (e.g. `https://example.com`, `http://example.com/`, `https://example.com/?utm_source=fb`), all URLs pass through an 8-stage normalization pipeline: + +1. **Scheme Lowercasing**: Resolves protocol-relative URLs (`//cdn.example.com` $\rightarrow$ `https://cdn.example.com`). +2. **Host Normalization**: Lowercases hostnames, strips root trailing dots. +3. **Port Stripping**: Drops default `:80` and `:443`. +4. **Path Resolution**: Normalizes dot segments (`/a/b/../c` $\rightarrow$ `/a/c`). +5. **Trailing Slash Consistency**: Standardizes directory path rules. +6. **Fragment Removal**: Drops `#anchor` fragments. +7. **Tracking Param Stripping**: Removes UTM, `fbclid`, `gclid`, and advertising noise. +8. **Facet & Query Protection**: Configurable query parameter limits (`--max-query-params`) and facet stripping (`--ignore-sorting-facets`) to eliminate e-commerce spider traps. + +### 4.2 AIMD Politeness & Congestion Control (`src/crawler/aimd.rs`) + +Unlike basic crawlers that hammer servers at fixed concurrency or rely on arbitrary sleep delays, SEO Lens implements **Additive-Increase / Multiplicative-Decrease (AIMD)** congestion control (similar to TCP Reno): + +- **Healthy Operation**: Gradually ramps up concurrency and shaves delay by 25ms per successful sample window. +- **Server Distress (Errors or high latency)**: If the error rate exceeds 8% or p95 response time spikes, delay is immediately multiplied by 2x and concurrency is halved. +- **Respects Standards**: Honors `Crawl-Delay` directives in `/robots.txt` as a hard delay floor. + +### 4.3 Streaming HTML Parser (`src/parser/`) + +Rather than loading giant DOM trees into memory with standard parsing libraries, SEO Lens uses Cloudflare's **`lol_html`** streaming HTML rewriter: + +- Pre-compiled CSS selectors extract `<title>`, `<meta>`, `<link>`, `<h1>`–`<h6>`, `<a>`, `<img>`, and schema tags as bytes stream through the network. +- Memory consumption remains virtually flat regardless of page size. + +### 4.4 Two-Phase Rules Engine (`src/rules/`) + +Auditing is divided cleanly into two phases: + +- **Phase 1: In-Flight Document Rules**: Run immediately as each page arrives (HTTP status, missing H1, title lengths, security headers, Google Rich Results schema validation, mobile viewports). +- **Phase 2: Post-Crawl Graph Rules**: Run after the crawl finishes across the entire site graph (orphan pages with no inlinks, redirect chains and loops, duplicate content via SimHash, internal PageRank equity). + +### 4.5 SQLite Storage Layer (`src/storage/`) + +- All audit data is persisted into an embedded SQLite database using **Write-Ahead Logging (`PRAGMA journal_mode = WAL`)** and **`PRAGMA synchronous = NORMAL`**. +- Writes are batched in atomic chunks of 250 pages to ensure blazing disk throughput. +- By default, databases are stored in the user's OS data directory (`~/.local/share/seolens/seolens.db` on Linux), or locally in `./.seolens/seolens.db` when using the `-L` / `--local` flag. + +### 4.6 Native In-Process MCP Server (`src/mcp/`) + +SEO Lens embeds a native Model Context Protocol (MCP) server running JSON-RPC 2.0 over `stdio`: + +- Built in pure Rust with Tokio channels (zero extra MCP crate dependencies). +- Exposes 8 structured tools (`seo_start_audit`, `seo_audit_status`, `seo_get_markdown_report`, etc.). +- Guaranteed non-blocking: Background audits return a session token in under 1 second, avoiding AI agent tool timeout limits. + +--- + +## 5. Dual-Engine Strategy: HTTP vs. Headless Chrome + +SEO Lens uses an intelligent dual-engine architecture: + +1. **Engine 1: Turbo Raw HTTP (Default)**: + - Processes 80–85% of standard websites (WordPress, Shopify, Astro, Laravel, Next.js with SSR). + - Achieves 500–2,000 pages/second with $<50$ MB RAM. + - Includes automatic **SPA Heuristic Detection**: If a page has `<div id="root"></div>` with empty text and no links, it alerts the user to re-run with `--render-js`. + +2. **Engine 2: Headless Chrome CDP (`--render-js`) (Planned / Not Yet Completed)**: + - For JavaScript-rendered Single Page Applications (SPAs). + - **Status**: _Not yet completed_. Headless Chrome CDP integration is planned on the roadmap to support client-side rendered SPAs and raw HTML vs. rendered DOM diffing. + - **Decoupled Architecture**: When implemented, Chromium will _not_ be bundled into the binary. SEO Lens will auto-detect host browser installations or connect via `--chrome-ws`. + - **JavaScript SEO Diffing**: Will automatically compare raw server HTML with client-rendered DOM to detect canonical tampering, dynamically injected `noindex` tags, or hydration rendering failures. + +--- + +## 6. Project Status & Roadmap + +| Subsystem / Feature | Status | Notes | +| :-------------------------------------- | :------------------- | :---------------------------------------------------------------------------------------------- | +| **Core Crawl Engine (Raw HTTP)** | βœ… Complete | Async Tokio pipeline, AIMD politeness, robots.txt, sitemaps. | +| **120 SEO Rules Catalog** | βœ… Complete | Phase 1 in-flight + Phase 2 post-crawl graph checks. | +| **petgraph Topology & PageRank** | βœ… Complete | Link graph, orphan detection, internal PageRank calculation. | +| **SQLite WAL Persistence** | βœ… Complete | Batched atomic transactions, session history, session cleaning. | +| **CLI (10 Subcommands)** | βœ… Complete | `audit`, `inspect`, `mcp`, `report`, `list`, `issues`, `check-ai`, `delete`, `clean`, `schema`. | +| **Model Context Protocol (MCP)** | βœ… Complete | In-process stdio JSON-RPC server with 8 non-blocking agent tools. | +| **Multi-Format Exporters** | βœ… Complete | Interactive TUI HTML report, Screaming Frog CSVs, Markdown, JSON. | +| **Headless Chrome CDP (`--render-js`)** | ⏳ Not Yet Completed | Planned decoupled CDP engine for JavaScript SPAs and DOM diffing. | +| **Native Desktop App (Tauri v2)** | ⏳ Next Milestone | Shell placeholder ready in `src-tauri`; React 19 UI in progress. | +| **Binary Packaging & Docker** | ⏳ Next Milestone | Multi-stage Dockerfile (`seolens:slim` and `seolens:full`). | + +--- + +## 7. Engineering Standards for Contributors + +When writing code for SEO Lens, please follow these principles: + +1. **Zero Panics in Library Code**: Never use `unwrap()` or `expect()` in `src/` library modules. All fallible operations must return `Result<T, SeoError>` using `thiserror`. +2. **Memory First**: Use `compact_str::CompactString` for strings $\le 24$ bytes (URLs, tags, MIME types) to keep memory stack-inlined. Use `bitflags` for boolean flags. +3. **Test-Driven Development (TDD)**: Write failing tests in `tests/` before adding new features or SEO rules. Verify with `cargo nextest run` (or `cargo test`). +4. **Clean Code & Formatting**: Run `cargo fmt --all` and ensure `cargo clippy` passes cleanly before submitting PRs. diff --git a/docs/assets/architecture_pipeline.png b/docs/assets/architecture_pipeline.png new file mode 100644 index 0000000..4d89480 Binary files /dev/null and b/docs/assets/architecture_pipeline.png differ diff --git a/docs/assets/cli_terminal_audit.png b/docs/assets/cli_terminal_audit.png new file mode 100644 index 0000000..5b99f3b Binary files /dev/null and b/docs/assets/cli_terminal_audit.png differ diff --git a/docs/assets/html_report_dashboard.png b/docs/assets/html_report_dashboard.png new file mode 100644 index 0000000..900e8ec Binary files /dev/null and b/docs/assets/html_report_dashboard.png differ diff --git a/docs/cli.md b/docs/cli.md new file mode 100644 index 0000000..6ec4eeb --- /dev/null +++ b/docs/cli.md @@ -0,0 +1,237 @@ +# SEO Lens: CLI & Exporters Reference + +Welcome to the command-line interface (CLI) and reporting reference for **SEO Lens**! + +Whether you're running audits locally from your terminal, integrating SEO checks into your CI/CD pipelines, or exporting data for spreadsheets and dashboards, this guide covers every subcommand, option, exit code, and export format. + +--- + +## 1. Quick Tour & Subcommand Overview + +SEO Lens provides **10 purpose-built subcommands** via the `seolens` binary: + +```bash +seolens +β”œβ”€β”€ audit <url> # Run a full website crawl & technical audit +β”œβ”€β”€ inspect <url> # Instant X-ray for a single webpage (headers, DOM, schema) +β”œβ”€β”€ mcp # Start native Model Context Protocol server (stdio) +β”œβ”€β”€ report <session> # Re-export or inspect a past audit without recrawling +β”œβ”€β”€ list # List all historical audit sessions stored in SQLite +β”œβ”€β”€ issues <session> # Filter and drill down into findings for an audit +β”œβ”€β”€ check-ai <url> # Audit AI search readiness (/llms.txt, AI bot policies) +β”œβ”€β”€ schema <url> # Validate JSON-LD structured data against Google Rich Results +β”œβ”€β”€ delete <session> # Delete a specific crawl session and its records +└── clean # Reclaim disk space by cleaning old crawl sessions +``` + +> [!TIP] +> Run `seolens <subcommand> --help` anytime to view the built-in documentation and default values directly in your terminal. + +--- + +## 2. Command Details & Practical Examples + +### 2.1 `seolens audit <url>` (Full Website Crawl) + +The primary command for technical SEO auditing. It discovers URLs, parses pages with a zero-copy streaming parser, applies AIMD adaptive rate limiting, runs 120 SEO rules, and exports reports. + +```bash +# Basic crawl (defaults: 500 pages, max depth 5, concurrency 10) +seolens audit https://example.com + +# High-depth crawl with HTML, CSV, and Markdown exports +seolens audit https://example.com -p 2000 -d 8 -f html,csv,md -o ./my-reports + +# High-speed local audit with AIMD throttling disabled +seolens audit http://localhost:3000 --no-aimd -p 100 + +# CI/CD check: fail pipeline if any Critical issues are detected +seolens audit https://staging.example.com --fail-on critical +``` + +#### Flags & Options + +| Flag / Option | Short | Type | Default | What It Does | +| :------------------------ | :---- | :------- | :------------------- | :---------------------------------------------------------------- | +| `<url>` | | `String` | _(Required)_ | Root URL to crawl (e.g. `https://example.com`). | +| `--max-pages` | `-p` | `u32` | `500` | Maximum pages to crawl (`0` = unlimited). | +| `--max-depth` | `-d` | `u16` | `5` | Maximum click depth from start URL. | +| `--concurrency` | `-c` | `usize` | `10` | Number of concurrent network requests. | +| `--delay` | | `u64` | `0` | Delay between requests in ms (`0` = auto-AIMD). | +| `--no-aimd` | | `bool` | `false` | Disable adaptive AIMD throttling (ideal for local staging tests). | +| `--render-js` | | `bool` | `false` | Enable Headless Chrome CDP for JavaScript SPAs. | +| `--chrome-ws` | | `String` | `"auto"` | Custom Chrome WebSocket URL (e.g. `ws://127.0.0.1:9222`). | +| `--user-agent` | `-u` | `String` | `"SEOLens/1.0"` | Custom User-Agent header string. | +| `--format` | `-f` | `String` | `"terminal,json,md"` | Outputs: `terminal`, `json`, `md`, `html`, `csv`, or `all`. | +| `--output-dir` | `-o` | `Path` | `"./reports"` | Directory where export files will be saved. | +| `--fail-on` | | `String` | `"none"` | CI/CD gate: `critical`, `alert`, or `warning`. | +| `--no-robots` | | `bool` | `false` | Ignore `/robots.txt` disallow rules. | +| `--ephemeral` | | `bool` | `false` | Ephemeral run: auto-cleans SQLite state on finish. | +| `--max-query-params` | | `usize` | `2` | Max query parameters allowed before pruning spider traps. | +| `--ignore-sorting-facets` | | `bool` | `true` | Prunes faceted sorting parameters (`sort`, `order`, etc.). | +| `--db-path` | | `Path` | _(System default)_ | Custom path to SQLite persistence database. | +| `--local` | `-L` | `bool` | `false` | Persist database locally to `./.seolens/seolens.db`. | + +--- + +### 2.2 `seolens inspect <url>` (Single-Page X-Ray) + +Need to quickly inspect a single page without running a site crawl? `inspect` fetches the URL, runs document-level SEO checks, and outputs a complete technical X-ray in under 500 milliseconds. + +```bash +# Inspect a live page +seolens inspect https://example.com/about + +# Output as structured JSON for piping into jq +seolens inspect https://example.com/about -f json | jq '.headings' +``` + +--- + +### 2.3 `seolens mcp` (Model Context Protocol Server) + +Launches the native Model Context Protocol (MCP) server over `stdio`. This allows AI coding agents like **Claude Desktop**, **Cursor**, and **Windsurf** to communicate directly with SEO Lens. + +```bash +# Start MCP server over stdio +seolens mcp +``` + +_(See [`mcp.md`](./mcp.md) for tool definitions and agent configuration guides)._ + +--- + +### 2.4 `seolens report <session_id>` (Re-Export Existing Audits) + +Every crawl is saved in your local SQLite database. If you ran an audit yesterday and now want to generate an interactive HTML report or CSV files, `report` does this instantly without touching the network: + +```bash +# Generate HTML and CSV reports for session crawl_1788718395 +seolens report crawl_1788718395 -f html,csv -o ./exports +``` + +--- + +### 2.5 `seolens list` (Session History) + +Lists all audit sessions stored in your local database with target URLs, page counts, durations, and health scores. + +```bash +# List recent sessions +seolens list + +# Show up to 50 sessions in JSON format +seolens list -n 50 -f json +``` + +--- + +### 2.6 `seolens issues <session_id>` (Issue Drill-Down) + +Allows you to filter and inspect issues discovered during a crawl directly in your terminal. + +```bash +# View only Critical issues for an audit +seolens issues crawl_1788718395 -s critical + +# View Security category issues +seolens issues crawl_1788718395 -c security +``` + +--- + +### 2.7 `seolens check-ai <url>` (AI & GEO Readiness) + +Audits whether a website is ready for Generative Engine Optimization (GEO) and AI search engines: + +- Checks presence and structure of `/llms.txt` and `/llms-full.txt`. +- Inspects `/robots.txt` to see if AI Retrieval/Search Bots (e.g. `PerplexityBot`, `OAI-SearchBot`) or AI Training Crawlers (e.g. `GPTBot`, `ClaudeBot`) are blocked. + +```bash +seolens check-ai https://example.com +``` + +--- + +### 2.8 `seolens schema <url>` (Structured Data Validator) + +Extracts all JSON-LD scripts from a URL and validates them against Google Rich Results specifications (Product, Article, FAQ, LocalBusiness, Breadcrumbs, etc.). + +```bash +seolens schema https://example.com/products/headphones +``` + +--- + +### 2.9 `seolens delete <session_id>` & `seolens clean` (Storage Management) + +Manage your local SQLite storage footprint: + +```bash +# Delete a specific crawl session +seolens delete crawl_1788718395 -y + +# Preview what sessions would be cleaned (dry run) +seolens clean --keep 5 --dry-run + +# Reclaim space: keep only the 5 most recent crawls and delete the rest +seolens clean --keep 5 -y +``` + +--- + +## 3. Exit Codes (CI/CD Pipelines) + +SEO Lens uses standard UNIX exit codes so you can plug audits directly into GitHub Actions, GitLab CI, or pre-deployment hooks: + +| Exit Code | Meaning | Condition | +| :-------: | :-------------------------- | :-------------------------------------------------------------------------------------- | +| **`0`** | **Clean / Success** | Audit finished successfully and no issues violated the `--fail-on` threshold. | +| **`1`** | **Threshold Violation** | Found one or more issues matching or exceeding `--fail-on` (e.g. `--fail-on critical`). | +| **`2`** | **Runtime / Network Error** | Target URL unreachable, DNS failure, invalid arguments, or disk error. | +| **`130`** | **Interrupted (`Ctrl+C`)** | User cancelled the crawl gracefully. Partial results remain saved in SQLite. | + +--- + +## 4. Multi-Format Exporters + +SEO Lens supports 5 complementary export formats: + +### 4.1 Standalone Interactive HTML (`report.html`) + +- **Self-Contained**: 100% offline. CSS, SVG icons, and JavaScript are bundled directly into the single file. No external CDNs or Google Fonts. +- **Authentic Workstation Aesthetic**: Features an ASCII branding banner, high-contrast monospace typography, and retro CRT workstation styling. +- **Interactive Filtering**: + - Real-time search bar filtering across URLs, titles, error codes, and issue descriptions. + - Severity filter tabs (All, Critical, Alerts, Warnings, Notices). + - Expandable issue drawers with direct remediation advice. + - Dynamic ASCII-style progress bars and telemetry indicators. + +### 4.2 Screaming Frog Compatible CSV Suite (`reports/csv/`) + +Designed for agency teams and SEO consultants who work with spreadsheets: + +1. `internal_all.csv`: Full crawl inventory matching Screaming Frog columns (URL, Status, Title, Description, H1, Canonical, Inlinks, Outlinks, TTFB, Word Count). +2. `issues_all.csv`: Complete list of triggered audit findings with severity, categories, affected URLs, and remediation steps. +3. `response_codes.csv`: HTTP routing breakdown and redirect targets. +4. `external_all.csv`: External links found, anchor texts, and `rel="nofollow"` attributes. + +### 4.3 Executive Markdown (`report.md`) + +Formatted using clean GitHub-Flavored Markdown. Perfect for: + +- Pasting directly into client audit summaries or PR descriptions. +- Providing context directly to LLM prompt windows without token bloat. + +### 4.4 Machine-Readable JSON (`report.json`) + +Lossless structured dump of the entire audit: crawl summary, timing percentiles, every `PageReport`, triggered issues, and site link graph edges. + +### 4.5 Live Terminal UI + +During execution, `seolens audit` renders a live colored ANSI dashboard showing: + +- Real-time crawl rate (pages/sec) and progress bar. +- Dynamic AIMD delay and p95 server TTFB. +- Live issue counters (🚨 Critical, ⚠️ Alerts, ⚑ Warnings). +- Post-crawl executive scorecard with health score (0–100) and status breakdown. diff --git a/docs/crawler.md b/docs/crawler.md new file mode 100644 index 0000000..1965da3 --- /dev/null +++ b/docs/crawler.md @@ -0,0 +1,246 @@ +# SEO Lens: Crawler & Politeness Engine Guide + +Welcome to the **Crawler & Politeness Engine Guide**! + +This document provides a comprehensive tour of how SEO Lens navigates the web: how it discovers and queues URLs, respects web servers with adaptive congestion control, normalizes link targets to prevent duplicate visits, and parses robots and sitemaps with zero-copy efficiency. + +--- + +## 1. Concurrency Architecture & Task Flow + +SEO Lens uses an asynchronous producer-consumer pipeline built on the **Tokio** runtime: + +```mermaid +flowchart TD + subgraph Discovery ["1. URL Discovery & Frontier"] + Queue["URL Frontier (FIFO VecDeque)"] + VisitedSet["Visited Set (SwissTable 64-bit Hashes)"] + SitemapEngine["Sitemap Ingestion (quick-xml)"] + RobotsEngine["Robots.txt Parser (RFC 9309)"] + end + + SitemapEngine -->|Seed URLs| Queue + + subgraph ConcurrencyPipeline ["2. Concurrency & Politeness"] + Queue -->|"Pop URL & Depth"| Worker["Worker Task (tokio::spawn)"] + Worker -->|Check Permission| RobotsEngine + Worker --> RateGate["AIMD Rate Controller (Dynamic Delay)"] + RateGate --> HTTPClient["reqwest HTTP/2 Client Pool"] + end + + subgraph StreamProcessing ["3. Stream Ingestion & Feedback"] + HTTPClient --> Telemetry["Latency & Error Feedback Loop"] + Telemetry -->|Tune Delay & Concurrency| RateGate + HTTPClient --> Parser["Streaming lol_html Tokenizer"] + Parser --> LinkFilter["Domain Boundary & Link Filter"] + LinkFilter -->|Unseen URLs| VisitedSet + VisitedSet -->|New URLs| Queue + end +``` + +### The 3 Pipeline Stages Explained + +1. **Discovery & Frontier**: Seed URLs (or URLs discovered in sitemaps) are pushed into the Frontier Queue. Before any URL is enqueued, it is normalized and checked against the visited set using 64-bit fast hashing. +2. **Concurrency & Politeness Gate**: Worker tasks request permission from the RFC 9309 `robots.txt` engine and pass through the **AIMD rate controller**, which dynamically modulates delays based on server health. +3. **Stream Processing & Feedback**: As responses stream in, the client measures response latency (TTFB) and HTTP status codes, feeding this telemetry back to the AIMD controller to dynamically speed up or slow down the crawl rate. + +--- + +## 2. The AIMD Adaptive Politeness Engine + +### Why Politeness Matters + +Crawling websites with aggressive concurrency without rate limiting can easily crash origin databases, trigger 429 Too Many Requests errors, or prompt Cloudflare/WAF IP bans. + +To solve this, SEO Lens implements **Additive-Increase / Multiplicative-Decrease (AIMD)** congestion controlβ€”the same foundational algorithm that powers TCP Reno on the Internet: + +### The AIMD Decision Cycle + +```mermaid +flowchart TD + Window["Sample Window (N = 50 requests)"] + Compute["Compute: Error Rate (E) & p95 TTFB"] + Window --> Compute + + Compute -->|"Degradation Trigger<br/>(E > 8% OR p95 > 1.5x baseline)"| Backoff["Multiplicative Backoff (Load Halved)<br/>delay = min(delay * 2.0, 10,000ms)<br/>concurrency = max(floor(c * 0.5), 1)"] + Compute -->|"Healthy Recovery<br/>(E == 0% AND p95 < 500ms)"| StepUp["Additive Recovery (Gentle Ramp)<br/>delay = max(delay - 25ms, delay_floor)<br/>concurrency = min(c + 1, c_max)"] +``` + +### AIMD Control Parameters + +| Parameter | Symbol | Default Value | Engineering Rationale | +| :--- | :--- | :--- | :--- | +| **Sample Window** | $W$ | `50 requests` | Smooths statistical outliers while reacting to server distress in $<3$ seconds. | +| **Error Rate Ceiling** | $E_{\text{thresh}}$ | `0.08` (8%) | If $>4$ out of 50 requests fail (429, 500, 502, 503, 504), origin is in distress. | +| **Latency Multiplier** | $L_{\text{factor}}$ | `1.5` | If p95 TTFB exceeds $1.5\times$ baseline, the backend is experiencing query queueing. | +| **Multiplicative Backoff** | $\beta$ | `2.0` | Instantly halves origin load when strain or rate limiting is detected. | +| **Additive Recovery Step** | $\Delta_{\text{add}}$ | `25 ms` | Gently reduces delay by 25ms per successful window without shocking the origin. | +| **Max Delay Ceiling** | $\text{delay}_{\text{max}}$ | `10,000 ms` | Prevents infinite stalls by capping maximum request delay at 10 seconds. | +| **Delay Floor** | $\text{delay}_{\text{floor}}$ | $\max(\text{robots\_delay}, 0\text{ms})$ | Strictly honors `Crawl-Delay` directives in `/robots.txt`. | + +> [!TIP] +> When testing local staging environments or high-throughput benchmarks, you can bypass dynamic AIMD rate adjustments using the `--no-aimd` flag. + +--- + +## 3. URL Normalization Pipeline + +Discovered links often point to the same destination under different formats (e.g. `https://example.com`, `http://example.com/`, `https://example.com/?utm_source=twitter`). + +To guarantee that each unique page is crawled **exactly once**, all URLs pass through an 8-stage normalization pipeline in `src/core/url.rs`: + +```text +Raw Href ──> [1. Scheme] ──> [2. Hostname] ──> [3. Port] ──> [4. Path] + ──> [5. Trailing Slash] ──> [6. Strip Fragments] + ──> [7. Strip Tracking Query] ──> [8. Sort Query] ──> Normalized URL +``` + +### Step-by-Step Normalization Rules + +1. **Scheme Lowercasing**: Converts scheme to lowercase (`HTTP` $\rightarrow$ `http`). Resolves protocol-relative URLs (`//cdn.example.com` $\rightarrow$ `https://cdn.example.com`). +2. **Hostname Lowercasing & Root Dot Removal**: Lowercases domain names (`Example.COM` $\rightarrow$ `example.com`) and removes trailing root dots (`example.com.` $\rightarrow$ `example.com`). +3. **Default Port Stripping**: Drops standard ports (`:80` for HTTP, `:443` for HTTPS). +4. **Path Segment Resolution**: Resolves dot segments (`/a/b/../c` $\rightarrow$ `/a/c`) and collapses consecutive slashes (`/blog//post` $\rightarrow$ `/blog/post`). +5. **Root Path Enforcement**: If path is empty, ensures root `/` is present (`https://example.com` $\rightarrow$ `https://example.com/`). +6. **Fragment Removal**: Drops anchor fragments (`/page#reviews` $\rightarrow$ `/page`). +7. **Tracking Parameter Stripping**: Automatically strips marketing and analytics tracking noise: + - `utm_source`, `utm_medium`, `utm_campaign`, `utm_term`, `utm_content` + - `fbclid`, `gclid`, `msclkid`, `mc_eid`, `_ga`, `_gl`, `ref` +8. **Deterministic Query Sorting**: Lexicographically sorts legitimate query parameters (`?b=2&a=1` $\rightarrow$ `?a=1&b=2`). + +### Spider Trap & Facet Protection + +E-commerce and filtered catalog sites often generate infinite calendar loops or faceted navigation spider traps. SEO Lens provides two built-in guardrails: + +- **`--max-query-params <N>`** (default: `2`): Limits the number of permissible query parameters before flagging or pruning candidate URLs. +- **`--ignore-sorting-facets`** (default: `true`): Automatically strips sorting and display facets (e.g. `sort=price_desc`, `order=date`, `view=grid`) that produce duplicate content. + +### Hash-Based Deduplication + +After normalization, each URL string is converted into a **64-bit AHash (`u64`)**: + +- The `VisitedSet` in `src/crawler/frontier.rs` stores only `u64` values in a SwissTable (`hashbrown::HashSet<u64>`). +- Storing 50,000 URLs consumes **under 400 KB of RAM** (compared to $>10$ MB for raw string vectors). + +--- + +## 4. Frontier Queue & Depth Management + +The crawler frontier schedules pending URLs and enforces crawl boundaries: + +```rust +pub struct FrontierEntry { + pub url: String, + pub depth: u16, + pub source_url: Option<String>, +} +``` + +### Scheduling Policies + +1. **Breadth-First Search (BFS) (Default)**: + - Uses `VecDeque<FrontierEntry>` (FIFO queue). + - Ensures shallow, high-PageRank category pages are audited first before digging into deep pagination or leaf nodes. +2. **Depth Limiting**: + - When a link is discovered on a page at `depth = d`, candidate entries are assigned `depth = d + 1`. + - If `d + 1 > max_depth`, the URL is added to the site link graph as an edge but is **never enqueued for crawling**. +3. **Domain Boundary Control**: + - **Internal Domain**: Host matches the starting domain (or subdomains if enabled). Crawled recursively. + - **External Domain**: Outbound link. Validated with a lightweight status check, but outgoing links are not extracted. + +--- + +## 5. RFC 9309 Robots.txt & Sitemap Compliance + +### 5.1 Robots.txt Rules (`src/crawler/robots.rs`) + +SEO Lens adheres strictly to **RFC 9309 (Robots Exclusion Protocol)**: + +1. **User-Agent Matching**: + - Specific user-agent matches take precedence over wildcard `*`. + - Priority hierarchy: `SEOLens` $\rightarrow$ `Googlebot` $\rightarrow$ `*`. +2. **Longest Match Precedence**: + - When multiple directives match a path, the rule with the longest character pattern wins: + + ```text + Allow: /products/ + Disallow: /products/archived/ + # URL: /products/archived/item-1 -> DISALLOWED (20 chars vs 10 chars) + ``` + +3. **Allow Overrides Disallow on Equal Length**: + - If `Allow: /blog` and `Disallow: /blog` have identical pattern lengths, `Allow` takes precedence per RFC 9309. +4. **Wildcards**: Fully supports `*` (zero or more characters) and `$` (end of URL pattern). + +### 5.2 Streaming Sitemap Ingestion (`src/crawler/sitemap.rs`) + +Uses `quick-xml` for zero-allocation streaming XML parsing: + +1. **Auto-Discovery**: + - Checks `Sitemap:` declarations in `/robots.txt`. + - Probes standard paths: `/sitemap.xml`, `/sitemap_index.xml`, `/wp-sitemap.xml`. +2. **Sitemap Index Recursion**: + - Recursively expands nested `<sitemapindex>` entries up to 3 levels deep. +3. **Compressed Sitemaps (`.xml.gz`)**: + - Automatically detects compressed `.gz` sitemaps and streams decompression on the fly using `flate2`. +4. **Orphan Page Detection**: + - Sourced URLs are tagged with `is_sitemap_url = true`. + - If post-crawl analysis finds that a sitemap URL has 0 incoming internal links from crawled HTML pages, `ALERT_GRAPH_ORPHAN_PAGE` is triggered. + +--- + +## 6. WAF & Anti-Bot Fingerprint Detection (`src/crawler/waf.rs`) + +When crawling client sites protected by Cloudflare, Akamai, or DataDome, firewalls often return HTTP 200 or 403 with a JavaScript challenge screen. Parsing challenge pages blindly would flood audit reports with false alarms ("missing title", "zero words", "missing H1"). + +SEO Lens inspects response bodies for known challenge signatures: + +```rust +pub struct WafProbe { + pub provider: &'static str, + pub signatures: &'static [&'static str], +} + +pub const WAF_SIGNATURES: &[WafProbe] = &[ + WafProbe { + provider: "Cloudflare", + signatures: &[ + "cf-browser-verification", + "checking your browser before accessing", + "cloudflare ray id", + "/cdn-cgi/challenge-platform/", + ], + }, + WafProbe { + provider: "Akamai", + signatures: &[ + "akamai bot manager", + "_abck=", + "reference number:", + "bm-sz=", + ], + }, + WafProbe { + provider: "DataDome", + signatures: &[ + "geo.captcha-delivery.com", + "datadome", + "dd_cookie_test", + ], + }, + WafProbe { + provider: "Imperva", + signatures: &[ + "incapsula incident id", + "_incap_ses", + "visid_incap", + ], + }, +]; +``` + +When a challenge signature matches: + +1. Flags `ALERT_WAF_BOT_CHALLENGE` on the URL. +2. Suppresses false-positive warnings for missing titles, H1s, or content. +3. Advises the user in the report: *"Blocked by Cloudflare/Akamai WAF challenge. Whitelist the crawler IP or supply a session cookie."* diff --git a/docs/mcp.md b/docs/mcp.md new file mode 100644 index 0000000..43bd3a3 --- /dev/null +++ b/docs/mcp.md @@ -0,0 +1,272 @@ +# SEO Lens: Model Context Protocol (MCP) Guide + +Welcome to the **Model Context Protocol (MCP)** guide for **SEO Lens**! + +SEO Lens features a native, in-process MCP server that turns it into an autonomous AI assistant for technical SEO auditing. Instead of dumping massive 20 MB JSON or CSV files that blow past LLM context limits, SEO Lens gives AI agents (like **Claude Desktop**, **Cursor**, **Windsurf**, and **Devin**) clean, purpose-built tools to launch audits, check progress, and query actionable code remediation instructions. + +--- + +## 1. Quick Setup for AI Coding Agents + +Running `seolens mcp` starts the native MCP server over standard input/output (`stdio`). + +You can tell your AI agent to add the mcp server. Just copy and paste this prompt into your AI coding assistant (**Cursor**, **Claude Desktop**, **Windsurf**, **Cline**, or **Devin**): + +> _"Add SEO Lens to my MCP configuration so you can audit websites, check SEO health, and validate schemas. Command: `seolens`, args: `[\"mcp\"]`."_ + +### Standard MCP Server JSON + +```json +{ + "mcpServers": { + "seolens": { + "command": "seolens", + "args": ["mcp"] + } + } +} +``` + +--- + +## 2. Server Architecture & The Non-Blocking Guarantee + +```mermaid +flowchart TD + Agent["AI Coding Agent\n(Claude / Cursor / Windsurf)"] + + Agent <-->|"JSON-RPC 2.0 over stdio"| MCPCore["Native MCP Engine (src/mcp/)\n(Pure Rust + Tokio, zero extra crates)"] + + subgraph ToolSuite ["8 Structured Agent Tools"] + T1["seo_start_audit (Returns session token in < 1s)"] + T2["seo_audit_status (Live progress & telemetry)"] + T3["seo_get_markdown_report (LLM-optimized prompt summary)"] + T4["seo_quick_page_check (Instant single-page audit < 500ms)"] + T5["seo_query_issues (Filtered lookups by severity/category)"] + T6["seo_check_ai_readiness (/llms.txt & AI bot crawler rules)"] + T7["seo_validate_schema (Google Rich Results JSON-LD validator)"] + T8["seo_cleanup_session (Purge audit data & free disk space)"] + end + + subgraph BackgroundWorker ["Tokio Background Task"] + Crawler["Crawler & 120 SEO Rules"] + Storage[(SQLite WAL Database)] + end + + MCPCore --> ToolSuite + T1 -->|Spawn Task| BackgroundWorker + T2 -->|Read Progress| Storage + T3 -->|Format Actionable Report| Storage + T4 -->|Direct Stream| Crawler +``` + +### Why Non-Blocking Matters for AI Agents + +Most AI agent interfaces impose a strict **30 to 60-second timeout** on tool execution. A full website crawl of several hundred pages can take 1 to 2 minutes. + +- `seo_start_audit` **never blocks** waiting for the crawl to finish. +- It validates the URL, kicks off the crawl on a background Tokio task, registers the `session_id`, and returns in **under 1.0 second**. +- The agent can then poll `seo_audit_status` periodically until `status == "completed"`, and then retrieve a clean summary with `seo_get_markdown_report`. + +--- + +## 3. The 8 Core MCP Tools + +### Tool 1: `seo_start_audit` + +Starts an asynchronous background website crawl and technical SEO audit. + +**Input Parameters:** + +- `url` (_string_, required): Starting URL (e.g. `"https://example.com"`). +- `max_pages` (_integer_, optional, default: `500`): Maximum pages to crawl. +- `max_depth` (_integer_, optional, default: `5`): Maximum click depth. +- `render_js` (_boolean_, optional, default: `false`): Enable headless Chrome CDP for JavaScript SPAs. +- `respect_robots` (_boolean_, optional, default: `true`): Whether to obey `/robots.txt`. +- `ai_geo_audit` (_boolean_, optional, default: `true`): Check `/llms.txt` and AI bot crawler rules. + +**Sample Response:** + +```json +{ + "session_id": "crawl_1788718395", + "status": "queued", + "target_url": "https://example.com", + "message": "Audit started in background. Poll 'seo_audit_status' to monitor progress.", + "poll_interval_seconds": 15 +} +``` + +--- + +### Tool 2: `seo_audit_status` + +Polls the live progress and operational telemetry of an ongoing or completed audit. + +**Input Parameters:** + +- `session_id` (_string_, required): Session ID returned by `seo_start_audit`. + +**Sample Response:** + +```json +{ + "session_id": "crawl_1788718395", + "status": "crawling", + "pages_crawled": 142, + "pages_discovered": 380, + "current_delay_ms": 125, + "p95_ttfb_ms": 340, + "issues_count": { + "critical": 3, + "alert": 12, + "warning": 45, + "notice": 18 + }, + "elapsed_seconds": 45, + "is_complete": false +} +``` + +--- + +### Tool 3: `seo_get_markdown_report` + +Generates a structured, concise Markdown report engineered specifically for LLM context windows, highlighting the most critical issues and specific code remediation guidance. + +**Input Parameters:** + +- `session_id` (_string_, required): The audit session ID. +- `top_issues_limit` (_integer_, optional, default: `20`): Max issue types to summarize. +- `include_urls` (_boolean_, optional, default: `true`): Include sample affected URLs for each issue. + +--- + +### Tool 4: `seo_quick_page_check` + +Performs an instant, synchronous audit of a single URL in $< 500\text{ms}$. Great for testing a specific landing page or verifying a code fix immediately. + +**Input Parameters:** + +- `url` (_string_, required): Single URL to fetch and audit. +- `render_js` (_boolean_, optional, default: `false`): Execute JavaScript via headless Chrome. + +**Sample Response:** + +```json +{ + "url": "https://example.com/pricing", + "status_code": 200, + "ttfb_ms": 145, + "title": "Pricing Plans | Example SaaS", + "meta_description": "Affordable plans for teams of all sizes.", + "h1": "Transparent Pricing", + "canonical_url": "https://example.com/pricing", + "word_count": 840, + "is_indexable": true, + "issues_detected": [ + { + "code": "WARN_IMG_MISSING_ALT", + "severity": "warning", + "message": "2 images missing alt attributes." + } + ] +} +``` + +--- + +### Tool 5: `seo_query_issues` + +Queries specific issues from an audit database, filtered by severity, category, or URL pattern. + +**Input Parameters:** + +- `session_id` (_string_, required): The audit session ID. +- `severity` (_string_, optional): `"critical"`, `"alert"`, `"warning"`, or `"notice"`. +- `category` (_string_, optional): e.g. `"canonicalization"`, `"security"`, `"headings"`. +- `url_pattern` (_string_, optional): SQL LIKE or glob pattern (e.g. `"%blog%"`). +- `limit` (_integer_, optional, default: `50`): Max records to return. + +--- + +### Tool 6: `seo_check_ai_readiness` + +Audits whether a website is ready for Generative Engine Optimization (GEO) and AI Search Engines (ChatGPT Search, Perplexity, Claude). + +**Input Parameters:** + +- `url` (_string_, required): Website root URL. + +**Sample Response:** + +```json +{ + "base_url": "https://example.com", + "llms_txt_found": true, + "llms_full_txt_found": false, + "ai_crawler_access": { + "retrieval_citation_bots": { + "OAI-SearchBot": "ALLOWED", + "PerplexityBot": "DISALLOWED" + }, + "training_bots": { + "GPTBot": "DISALLOWED", + "ClaudeBot": "DISALLOWED" + } + }, + "citation_search_risk": "HIGH", + "recommendations": [ + "PerplexityBot is blocked in robots.txt. Your site will not be cited as a source in Perplexity answers. Allow PerplexityBot to regain visibility." + ] +} +``` + +--- + +### Tool 7: `seo_validate_schema` + +Validates a raw JSON-LD or Schema.org snippet against Google Rich Results guidelines. + +**Input Parameters:** + +- `json_ld` (_string_, required): Raw JSON-LD text or snippet. +- `target_type` (_string_, optional): Expected `@type` (e.g. `"Product"`, `"Article"`, `"FAQPage"`). + +--- + +### Tool 8: `seo_cleanup_session` + +Deletes audit records from SQLite for a session to free up disk space. + +**Input Parameters:** + +- `session_id` (_string_, required): Session ID to purge. + +--- + +## 4. MCP URI Resources + +In addition to tools, SEO Lens exposes read-only MCP URI resources: + +- `seo://crawls`: Lists all recent crawl sessions. +- `seo://crawls/{session_id}/summary`: JSON summary of crawl metrics and health score. +- `seo://crawls/{session_id}/report`: Full Markdown report. +- `seo://crawls/{session_id}/issues`: Raw JSON array of all issues. + +--- + +## 5. Testing the MCP Server Locally + +You can test the MCP server manually or using the MCP Inspector: + +```bash +# Test with npx @modelcontextprotocol/inspector +npx @modelcontextprotocol/inspector seolens mcp +``` + +Or run the automated integration tests: + +```bash +cargo nextest run --test mcp_tests +``` diff --git a/docs/SEO_RULES_CATALOG.md b/docs/rules.md similarity index 91% rename from docs/SEO_RULES_CATALOG.md rename to docs/rules.md index 9f4483a..a33bc04 100644 --- a/docs/SEO_RULES_CATALOG.md +++ b/docs/rules.md @@ -1,7 +1,6 @@ -# SEO Lens: Technical SEO Rules Catalog (120 Checks) +# Technical SEO Rules Catalog (120 Checks) -**Document Status**: Permanent Static Reference -**Purpose**: Authoritative specification of all 120 automated SEO checks, detection heuristics, severity ratings, and remediation guidelines for the `SEO Lens` engine. +This catalog is the authoritative specification for all 120 automated SEO checks in the `SEO Lens` engine. It details the detection heuristics, severity classification, search impact, and remediation guidance for each rule. --- @@ -16,22 +15,38 @@ ## Summary Matrix by Category -| Category | Checks | Phase | Focus | -| -------------------------------------- | ------- | ---------------------------- | -------------------------------------------------------------- | -| **1. HTTP & Network Transport** | 10 | Phase 1 (In-Flight) | HTTP codes, timeouts, DNS/TLS errors, WAF challenges | -| **2. Titles & Basic Metadata** | 10 | Phase 1 (In-Flight) | Title/description absence, lengths, multiplicity, keywords | -| **3. Headings & Document Structure** | 10 | Phase 1 (In-Flight) | H1 presence, hierarchy order, empty tags, DOM depth | -| **4. Indexability & Directives** | 10 | Phase 1 (In-Flight) | noindex, nofollow, robots.txt, pagination, SPA heuristics | -| **5. Canonicalization** | 10 | Phase 1 (In-Flight) | Missing, relative, conflicting, cross-domain, chain canonicals | -| **6. Links & Anchor Quality** | 10 | Phase 1 & 2 | Broken links, nofollow mix, empty anchors, bookmarks | -| **7. Security & Modern Transport** | 10 | Phase 1 (In-Flight) | HTTPS, mixed content, insecure forms, HSTS, CSP, headers | -| **8. Mobile, Performance & UX** | 10 | Phase 1 (In-Flight) | Viewports, TTFB, page size, image alt/dimensions (CLS) | -| **9. Internationalization (Hreflang)** | 8 | Phase 1 & 2 | Reciprocity graph, self-reference, lang codes, x-default | -| **10. Structured Data & Rich Results** | 8 | Phase 1 (In-Flight) | JSON-LD syntax, Google Rich Result required fields, OpenGraph | -| **11. GEO & AI Search Readiness** | 8 | Phase 1 (In-Flight) | /llms.txt, AI training vs retrieval bots, soft 404, AI tells | -| **12. Site Graph & Architecture** | 10 | Phase 2 (Post-Crawl) | Orphan pages, redirect/canonical loops, duplicates, PageRank | -| **13. JavaScript SEO Diffing** | 6 | Phase 1 (with `--render-js`) | Raw HTML vs Rendered DOM discrepancies | -| **Total** | **120** | | Comprehensive Technical Coverage | +> **Quick Jump**: +> +> 1. [HTTP & Network Transport](#category-1-http-status--network-transport-checks-110) +> 2. [Titles & Basic Metadata](#category-2-titles--basic-metadata-checks-1120) +> 3. [Headings & Document Structure](#category-3-headings--document-structure-checks-2130) +> 4. [Indexability & Directives](#category-4-indexability--robots-directives-checks-3140) +> 5. [Canonicalization](#category-5-canonicalization-checks-4150) +> 6. [Links & Anchor Quality](#category-6-links--anchor-quality-checks-5160) +> 7. [Security & Modern Transport](#category-7-security--modern-transport-checks-6170) +> 8. [Mobile, Performance & UX](#category-8-mobile-performance--ux-signals-checks-7180) +> 9. [Internationalization (Hreflang)](#category-9-internationalization--hreflang-checks-8188) +> 10. [Structured Data & Rich Results](#category-10-structured-data--google-rich-results-checks-8996) +> 11. [GEO & AI Search Readiness](#category-11-geo-generative-engine-optimization--ai-search-checks-97104) +> 12. [Site Graph & Architecture](#category-12-site-wide-graph--architecture-post-crawl-checks-105114) +> 13. [JavaScript SEO Diffing](#category-13-javascript-seo-diffing-checks-when---render-js-is-enabled-checks-115120) + +| Category | Checks | Phase | Focus | +| --- | --- | --- | --- | +| [**1. HTTP & Network Transport**](#category-1-http-status--network-transport-checks-110) | 10 | Phase 1 (In-Flight) | HTTP codes, timeouts, DNS/TLS errors, WAF challenges | +| [**2. Titles & Basic Metadata**](#category-2-titles--basic-metadata-checks-1120) | 10 | Phase 1 (In-Flight) | Title/description absence, lengths, multiplicity, keywords | +| [**3. Headings & Document Structure**](#category-3-headings--document-structure-checks-2130) | 10 | Phase 1 (In-Flight) | H1 presence, hierarchy order, empty tags, DOM depth | +| [**4. Indexability & Directives**](#category-4-indexability--robots-directives-checks-3140) | 10 | Phase 1 (In-Flight) | noindex, nofollow, robots.txt, pagination, SPA heuristics | +| [**5. Canonicalization**](#category-5-canonicalization-checks-4150) | 10 | Phase 1 (In-Flight) | Missing, relative, conflicting, cross-domain, chain canonicals | +| [**6. Links & Anchor Quality**](#category-6-links--anchor-quality-checks-5160) | 10 | Phase 1 & 2 | Broken links, nofollow mix, empty anchors, bookmarks | +| [**7. Security & Modern Transport**](#category-7-security--modern-transport-checks-6170) | 10 | Phase 1 (In-Flight) | HTTPS, mixed content, insecure forms, HSTS, CSP, headers | +| [**8. Mobile, Performance & UX**](#category-8-mobile-performance--ux-signals-checks-7180) | 10 | Phase 1 (In-Flight) | Viewports, TTFB, page size, image alt/dimensions (CLS) | +| [**9. Internationalization (Hreflang)**](#category-9-internationalization--hreflang-checks-8188) | 8 | Phase 1 & 2 | Reciprocity graph, self-reference, lang codes, x-default | +| [**10. Structured Data & Rich Results**](#category-10-structured-data--google-rich-results-checks-8996) | 8 | Phase 1 (In-Flight) | JSON-LD syntax, Google Rich Result required fields, OpenGraph | +| [**11. GEO & AI Search Readiness**](#category-11-geo-generative-engine-optimization--ai-search-checks-97104) | 8 | Phase 1 (In-Flight) | /llms.txt, AI training vs retrieval bots, soft 404, AI tells | +| [**12. Site Graph & Architecture**](#category-12-site-wide-graph--architecture-post-crawl-checks-105114) | 10 | Phase 2 (Post-Crawl) | Orphan pages, redirect/canonical loops, duplicates, PageRank | +| [**13. JavaScript SEO Diffing**](#category-13-javascript-seo-diffing-checks-when---render-js-is-enabled-checks-115120) | 6 | Phase 1 (with `--render-js`) | Raw HTML vs Rendered DOM discrepancies | +| **Total** | **120** | | Comprehensive Technical Coverage | --- diff --git a/docs/storage.md b/docs/storage.md new file mode 100644 index 0000000..12d958d --- /dev/null +++ b/docs/storage.md @@ -0,0 +1,162 @@ +# Storage & SQLite Architecture + +SEO Lens is **local-first**. All crawl sessions, discovered URLs, link topology, technical SEO issues, and structured data are persisted directly to a local SQLite database using Write-Ahead Logging (WAL). + +No external database servers (like PostgreSQL or MySQL) or cloud accounts are required. Your audit data remains private on your machine and can be queried in microseconds. + +--- + +## 1. Where the Database Lives + +By default, SEO Lens stores audits in your operating system's standard user data directory so that crawl sessions persist across terminal sessions. + +### Resolution Priority + +SEO Lens resolves the database file path in the following order: + +1. **CLI Flag (`--db-path <PATH>`)**: Explicit file path passed directly via the command line. +2. **Local Project Flag (`-L, --local`)**: Forces the database to `./.seolens/seolens.db` in your current working directory (ideal for project-specific audits and CI pipelines). +3. **Environment Variable (`SEOLENS_DB_PATH`)**: Overrides the default location system-wide. +4. **Operating System User Data Directory**: + - **Linux**: `~/.local/share/seolens/seolens.db` (or `$XDG_DATA_HOME/seolens/seolens.db`) + - **macOS**: `~/Library/Application Support/seolens/seolens.db` + - **Windows**: `%LOCALAPPDATA%\seolens\seolens.db` + +--- + +## 2. Storage Architecture + +Crawling thousands of pages generates an intense stream of database writes (pages, outlinks, images, and issues). Writing to disk synchronously on every HTTP response would throttle crawler speed. + +To achieve maximum throughput without blocking HTTP worker threads, SEO Lens uses an **Asynchronous Batch Writer Actor**: + +```mermaid +flowchart TD + Worker1["Tokio Worker Task #1"] -->|PageReport| Channel["tokio::sync::mpsc\n(Buffered Channel: 1,000)"] + Worker2["Tokio Worker Task #2"] -->|PageReport| Channel + WorkerN["Tokio Worker Task #N"] -->|PageReport| Channel + + Channel --> WriterActor["Background Writer Actor\n(src/storage/writer.rs)"] + + subgraph FlushTrigger ["Double-Trigger Flush"] + T1["Accumulated 250 records"] + T2["Timer tick: 2.0s elapsed"] + end + + FlushTrigger -.->|Triggers| WriterActor + WriterActor -->|"BEGIN TRANSACTION\nBulk Insert Pages & Links\nCOMMIT"| SQLite[("SQLite WAL Database\nseolens.db")] +``` + +### High-Throughput SQLite Tuning + +The connection pool initializes SQLite with production-grade WAL parameters: + +- **`PRAGMA journal_mode = WAL`**: Readers and writers never block each other. +- **`PRAGMA synchronous = NORMAL`**: Safe crash durability with minimal disk sync overhead. +- **`PRAGMA cache_size = -64000`**: Allocates 64 MB of RAM for the in-memory page cache. +- **`PRAGMA busy_timeout = 5000`**: Gracefully waits up to 5 seconds if a write transaction is busy. +- **`PRAGMA foreign_keys = ON`**: Enforces relational integrity across tables with cascade deletes. + +--- + +## 3. Database Schema Overview + +The definitive schema is maintained in [`../src/storage/schema.sql`](../src/storage/schema.sql), and the corresponding Rust domain models are defined in [`../src/core/models.rs`](../src/core/models.rs). + +Here is a summary of the 7 core tables: + +| Table | Description | Primary Key / Key Indexes | +| --- | --- | --- | +| **`crawls`** | Crawl sessions, target domain, crawl settings, overall health score, and status (`queued`, `crawling`, `completed`, `failed`). | `session_id TEXT` | +| **`pages`** | Comprehensive single-page audit reports: HTTP status, canonical, meta tags, TTFB, depth, word count, SimHash, and robots flags. | `id INTEGER`, indexed on `(crawl_id, url_hash)` | +| **`links`** | Directed edges in the site's link graph: source URL, target URL, anchor text, follow/nofollow, and target status. | `id INTEGER`, indexed on `(crawl_id, target_url_hash)` | +| **`issues`** | Specific technical SEO defects raised by the 120-rule audit engine. Severity levels: `1` (Critical), `2` (Alert), `3` (Warning), `4` (Notice). | `id INTEGER`, indexed on `(crawl_id, severity)`, `code` | +| **`schemas`** | Extracted JSON-LD and Microdata blocks with schema type and Google rich result eligibility. | `id INTEGER`, indexed on `crawl_id` | +| **`images`** | Image elements found on HTML pages with source URL, alt text, dimensions, and broken status. | `id INTEGER`, indexed on `crawl_id` | +| **`hreflangs`** | Internationalization alternate links with language codes and reciprocal validation status. | `id INTEGER`, indexed on `crawl_id` | + +--- + +## 4. Querying the Database Directly + +Because SEO Lens uses standard SQLite, you can query your crawl data directly using the `sqlite3` CLI, DB Browser for SQLite, or script it with Python/Node.js/Rust. + +### Opening the Database + +```bash +# Connect to the default user database +sqlite3 ~/.local/share/seolens/seolens.db + +# Or connect to a local project database +sqlite3 ./.seolens/seolens.db +``` + +### Useful SQL Queries + +#### 1. View recent crawl sessions + +```sql +SELECT session_id, target_url, status, total_pages, health_score, started_at, finished_at +FROM crawls +ORDER BY started_at DESC +LIMIT 5; +``` + +#### 2. Group issues by severity and rule code + +```sql +SELECT severity, code, title, COUNT(*) AS count +FROM issues +WHERE crawl_id = 'YOUR_SESSION_ID' +GROUP BY severity, code, title +ORDER BY severity ASC, count DESC; +``` + +#### 3. Find top broken internal links (404s) and where they are linked from + +```sql +SELECT l.source_url, l.target_url, l.anchor_text, p.status_code +FROM links l +JOIN pages p ON l.crawl_id = p.crawl_id AND l.target_url_hash = p.url_hash +WHERE l.crawl_id = 'YOUR_SESSION_ID' + AND p.status_code = 404 +ORDER BY l.source_url ASC +LIMIT 50; +``` + +#### 4. Find slowest pages by response time (TTFB) + +```sql +SELECT url, ttfb_ms, size_bytes, status_code +FROM pages +WHERE crawl_id = 'YOUR_SESSION_ID' +ORDER BY ttfb_ms DESC +LIMIT 20; +``` + +#### 5. Find missing `<title>` or meta descriptions + +```sql +SELECT url, title_length, meta_desc_length +FROM pages +WHERE crawl_id = 'YOUR_SESSION_ID' + AND (title IS NULL OR meta_description IS NULL) +LIMIT 50; +``` + +--- + +## 5. Cleaning Up Old Data + +You don't need to manually run SQL `DELETE` queries. SEO Lens includes built-in commands to manage disk usage: + +```bash +# Delete a specific crawl session (cascades across all tables) +seolens delete <SESSION_ID> + +# Delete sessions older than 30 days +seolens clean --older-than 30d + +# Delete all historical crawl sessions +seolens clean --all +``` diff --git a/src/cli/args.rs b/src/cli/args.rs index b8b1d83..129c910 100644 --- a/src/cli/args.rs +++ b/src/cli/args.rs @@ -3,12 +3,30 @@ //! Strongly-typed CLI options and subcommands for the `seolens` executable, //! parsed via `clap`. +use clap::builder::styling::{AnsiColor, Effects, Styles}; use clap::{Args, Parser, Subcommand}; use std::path::PathBuf; +/// Terminal styling for clap CLI outputs to match SEO Lens cyberpunk palette. +pub fn cli_styles() -> Styles { + Styles::styled() + .header(AnsiColor::BrightCyan.on_default() | Effects::BOLD) + .usage(AnsiColor::BrightCyan.on_default() | Effects::BOLD) + .literal(AnsiColor::BrightGreen.on_default() | Effects::BOLD) + .placeholder(AnsiColor::BrightCyan.on_default()) + .valid(AnsiColor::BrightGreen.on_default()) + .invalid(AnsiColor::BrightRed.on_default()) +} + /// SEO Lens - High-performance website crawler & technical SEO audit engine #[derive(Parser, Debug, Clone)] -#[command(name = "seolens", author, version, about, long_about = None)] +#[command( + name = "seolens", + author, + version, + about = "High-performance website crawler & AI-native technical SEO audit engine", + styles = cli_styles() +)] pub struct Cli { #[command(subcommand)] pub command: Commands, @@ -17,24 +35,46 @@ pub struct Cli { /// Available CLI subcommands. #[derive(Subcommand, Debug, Clone)] pub enum Commands { - /// Run a full or partial website crawl and audit + /// Run full website crawl with AIMD adaptive rate limiting & technical audit Audit(AuditArgs), - /// Inspect a single webpage: metadata, Open Graph, Twitter cards, JSON-LD, headings, and audit issues + /// Instant developer X-ray for a single webpage (DOM, tags, headers) Inspect(InspectArgs), - /// Start the native Model Context Protocol server (stdio for Cursor/Claude) + /// Start native Model Context Protocol server (stdio for Claude/Cursor) Mcp(McpArgs), /// Re-export or inspect an existing audit from the SQLite database Report(ReportArgs), - /// List all historical audit sessions stored locally + /// List all historical audit sessions stored locally in SQLite List(ListArgs), + /// Drill down and filter audit findings for a session + Issues(IssuesArgs), + /// Audit website readiness for AI search engines & /llms.txt + CheckAi(CheckAiArgs), + /// Delete a specific crawl session and its associated records + Delete(DeleteArgs), + /// Clean historical crawl sessions from the database + Clean(CleanArgs), + /// Validate JSON-LD / schema against Google Rich Results guidelines + Schema(SchemaArgs), } /// Command-line arguments for the `list` subcommand. #[derive(Args, Debug, Clone, Default)] pub struct ListArgs { + /// Maximum number of sessions to display + #[arg(short = 'n', long, default_value_t = 20)] + pub limit: usize, + + /// Output format: terminal (default) or json + #[arg(short = 'f', long, default_value = "terminal")] + pub format: String, + /// Custom path to SQLite persistence database #[arg(long)] pub db_path: Option<PathBuf>, + + /// Force database persistence to project-local `.seolens/seolens.db` + #[arg(short = 'L', long, default_value_t = false)] + pub local: bool, } /// Command-line arguments for the `audit` subcommand. @@ -103,9 +143,37 @@ pub struct AuditArgs { #[arg(long, default_value_t = true, action = clap::ArgAction::Set)] pub ignore_sorting_facets: bool, - /// Custom path to SQLite persistence database (default: .seolens/seolens.db) + /// Custom path to SQLite persistence database #[arg(long)] pub db_path: Option<PathBuf>, + + /// Force database persistence to project-local `.seolens/seolens.db` + #[arg(short = 'L', long, default_value_t = false)] + pub local: bool, + + /// Only crawl URLs matching this regex pattern + #[arg(short = 'i', long)] + pub include: Option<String>, + + /// Skip crawling URLs matching this regex pattern + #[arg(short = 'e', long)] + pub exclude: Option<String>, + + /// Custom HTTP request header(s) (e.g. -H "Authorization: Bearer xyz") + #[arg(short = 'H', long = "header")] + pub headers: Vec<String>, + + /// Explicit XML sitemap URL to crawl + #[arg(long)] + pub sitemap: Option<String>, + + /// Suppress progress bar output for clean CI/CD scripting + #[arg(short = 'q', long, default_value_t = false)] + pub quiet: bool, + + /// Optional audit or project name + #[arg(long)] + pub name: Option<String>, } /// Command-line arguments for the `inspect` subcommand. @@ -121,10 +189,18 @@ pub struct InspectArgs { /// Request timeout in seconds #[arg(long, default_value_t = 15)] pub timeout: u64, + + /// Output format: terminal (default), json, md + #[arg(short = 'f', long, default_value = "terminal")] + pub format: String, + + /// Custom HTTP request header(s) (e.g. -H "Authorization: Bearer xyz") + #[arg(short = 'H', long = "header")] + pub headers: Vec<String>, } /// Command-line arguments for the `mcp` subcommand. -#[derive(Args, Debug, Clone)] +#[derive(Args, Debug, Clone, Default)] pub struct McpArgs { /// Transport mechanism: stdio (local agents) or sse (remote HTTP) #[arg(long, default_value = "stdio")] @@ -133,10 +209,18 @@ pub struct McpArgs { /// Port to bind for HTTP/SSE transport (when --transport sse) #[arg(long, default_value_t = 8080)] pub port: u16, + + /// Custom path to SQLite persistence database + #[arg(long)] + pub db_path: Option<PathBuf>, + + /// Force database persistence to project-local `.seolens/seolens.db` + #[arg(short = 'L', long, default_value_t = false)] + pub local: bool, } /// Command-line arguments for the `report` subcommand. -#[derive(Args, Debug, Clone)] +#[derive(Args, Debug, Clone, Default)] pub struct ReportArgs { /// Session ID to inspect or re-export #[arg(short = 's', long)] @@ -157,4 +241,134 @@ pub struct ReportArgs { /// Custom path to SQLite persistence database #[arg(long)] pub db_path: Option<PathBuf>, + + /// Force database persistence to project-local `.seolens/seolens.db` + #[arg(short = 'L', long, default_value_t = false)] + pub local: bool, +} + +/// Command-line arguments for the `issues` subcommand. +#[derive(Args, Debug, Clone, Default)] +pub struct IssuesArgs { + /// Session ID to inspect + #[arg(short = 's', long)] + pub session: Option<String>, + + /// Session ID positional argument + #[arg(value_name = "SESSION_ID")] + pub session_pos: Option<String>, + + /// Filter by severity tier (critical, alert, warning, notice) + #[arg(long)] + pub severity: Option<String>, + + /// Filter by issue category (e.g. indexability, links, titles) + #[arg(short = 'c', long)] + pub category: Option<String>, + + /// Filter by rule ID code (e.g. ERR_HTTP_4XX_CLIENT_ERROR) + #[arg(long)] + pub code: Option<String>, + + /// Filter by URL substring (e.g. /blog/) + #[arg(long)] + pub url: Option<String>, + + /// Maximum number of issues to display + #[arg(short = 'n', long, default_value_t = 50)] + pub limit: usize, + + /// Pagination offset + #[arg(long, default_value_t = 0)] + pub offset: usize, + + /// Output format: terminal (default), json, md + #[arg(short = 'f', long, default_value = "terminal")] + pub format: String, + + /// Custom path to SQLite persistence database + #[arg(long)] + pub db_path: Option<PathBuf>, + + /// Force database persistence to project-local `.seolens/seolens.db` + #[arg(short = 'L', long, default_value_t = false)] + pub local: bool, +} + +/// Command-line arguments for the `check-ai` subcommand. +#[derive(Args, Debug, Clone)] +pub struct CheckAiArgs { + /// Website root URL to check (e.g. https://example.com) + pub url: String, + + /// Custom User-Agent string + #[arg(short = 'u', long, default_value = "SEOLens/1.0")] + pub user_agent: String, + + /// Request timeout in seconds + #[arg(long, default_value_t = 15)] + pub timeout: u64, + + /// Output format: terminal (default), json, md + #[arg(short = 'f', long, default_value = "terminal")] + pub format: String, +} + +/// Command-line arguments for the `delete` subcommand. +#[derive(Args, Debug, Clone, Default)] +pub struct DeleteArgs { + /// Session ID to delete + #[arg(short = 's', long)] + pub session: Option<String>, + + /// Session ID positional argument + #[arg(value_name = "SESSION_ID")] + pub session_pos: Option<String>, + + /// Custom path to SQLite persistence database + #[arg(long)] + pub db_path: Option<PathBuf>, + + /// Force database persistence to project-local `.seolens/seolens.db` + #[arg(short = 'L', long, default_value_t = false)] + pub local: bool, +} + +/// Command-line arguments for the `clean` subcommand. +#[derive(Args, Debug, Clone, Default)] +pub struct CleanArgs { + /// Purge all crawl sessions + #[arg(long, default_value_t = false)] + pub all: bool, + + /// Purge sessions started older than N days ago + #[arg(long)] + pub older_than: Option<u32>, + + /// Custom path to SQLite persistence database + #[arg(long)] + pub db_path: Option<PathBuf>, + + /// Force database persistence to project-local `.seolens/seolens.db` + #[arg(short = 'L', long, default_value_t = false)] + pub local: bool, +} + +/// Command-line arguments for the `schema` subcommand. +#[derive(Args, Debug, Clone)] +pub struct SchemaArgs { + /// URL or path to a local JSON/HTML file containing structured data + pub target: String, + + /// Expected schema @type (e.g. Product, Article, FAQPage) + #[arg(short = 't', long = "type")] + pub expected_type: Option<String>, + + /// Output format: terminal (default) or json + #[arg(short = 'f', long, default_value = "terminal")] + pub format: String, + + /// Custom User-Agent string (when target is a URL) + #[arg(short = 'u', long, default_value = "SEOLens/1.0")] + pub user_agent: String, } diff --git a/src/cli/commands.rs b/src/cli/commands.rs index f149e5f..51f3f72 100644 --- a/src/cli/commands.rs +++ b/src/cli/commands.rs @@ -1,23 +1,30 @@ //! # CLI Command Handlers //! -//! Implements execution logic for `audit`, `inspect`, `mcp`, `report`, and `list` commands. +//! Implements execution logic for `audit`, `inspect`, `mcp`, `report`, `list`, +//! `issues`, `check-ai`, `delete`, `clean`, and `schema` commands. -use crate::cli::args::{AuditArgs, Cli, Commands, InspectArgs, ListArgs, McpArgs, ReportArgs}; +use crate::cli::args::{ + AuditArgs, CheckAiArgs, CleanArgs, Cli, Commands, DeleteArgs, InspectArgs, IssuesArgs, + ListArgs, McpArgs, ReportArgs, SchemaArgs, +}; use crate::core::config::CrawlConfig; -use crate::core::models::Severity; +use crate::core::models::{IssueCategory, Severity}; +use crate::crawler::ai_check::audit_ai_readiness; +use crate::crawler::client::{FetchOptions, HttpClient}; use crate::crawler::engine::{run_crawl_with_options, CrawlResult, ProgressCallback}; -use crate::crawler::inspector::inspect_url; +use crate::crawler::inspector::inspect_url_with_options; use crate::graph::{compute_pagerank, SiteGraph}; use crate::report::{ - create_crawl_progress_bar, export_json_report, export_markdown_report, finish_crawl_progress, - print_audit_banner, print_executive_scorecard, print_historical_sessions, - print_page_inspection, update_crawl_progress, + create_crawl_progress_bar, export_csv_suite, export_html_report, export_json_report, + export_markdown_report, finish_crawl_progress, print_ai_readiness_scorecard, + print_audit_banner, print_executive_scorecard, print_historical_sessions, print_issues_matrix, + print_page_inspection, print_schema_outcome, update_crawl_progress, }; -use crate::storage::{default_db_path, CrawlSessionInit, Database}; +use crate::rules::page::schema_val::validate_raw_schema; +use crate::storage::{resolve_db_path, CrawlSessionInit, Database, IssueFilterCriteria}; use std::path::PathBuf; use std::sync::Arc; use std::time::Duration; -use tracing::info; /// Executes the parsed CLI command. pub async fn execute(cli: Cli) -> Result<(), Box<dyn std::error::Error>> { @@ -27,6 +34,11 @@ pub async fn execute(cli: Cli) -> Result<(), Box<dyn std::error::Error>> { Commands::Mcp(args) => handle_mcp(args).await, Commands::Report(args) => handle_report(args).await, Commands::List(args) => handle_list(args).await, + Commands::Issues(args) => handle_issues(args).await, + Commands::CheckAi(args) => handle_check_ai(args).await, + Commands::Delete(args) => handle_delete(args).await, + Commands::Clean(args) => handle_clean(args).await, + Commands::Schema(args) => handle_schema(args).await, } } @@ -43,8 +55,23 @@ async fn handle_audit(args: AuditArgs) -> Result<(), Box<dyn std::error::Error>> config.ephemeral = args.ephemeral; config.max_query_params = args.max_query_params; config.ignore_sorting_facets = args.ignore_sorting_facets; + config.include_regex = args.include; + config.exclude_regex = args.exclude; + config.quiet = args.quiet; + config.crawl_name = args.name; + + if let Some(sm) = args.sitemap { + config.explicit_sitemaps.push(sm); + } + for h in args.headers { + if let Some((k, v)) = h.split_once(':') { + config + .headers + .push((k.trim().to_string(), v.trim().to_string())); + } + } - let db_path = args.db_path.clone().unwrap_or_else(default_db_path); + let db_path = resolve_db_path(args.db_path.clone(), args.local); config.db_path = Some(db_path.clone()); let session_id = format!( @@ -72,23 +99,36 @@ async fn handle_audit(args: AuditArgs) -> Result<(), Box<dyn std::error::Error>> (None, None) }; - print_audit_banner( - &config.start_url, - config.max_pages, - config.concurrency, - !config.no_aimd, - ); + if !config.quiet { + print_audit_banner( + &config.start_url, + config.max_pages, + config.concurrency, + !config.no_aimd, + ); + } - let pb = create_crawl_progress_bar(config.max_pages); - let pb_clone = pb.clone(); + let pb = if !config.quiet { + Some(create_crawl_progress_bar(config.max_pages)) + } else { + None + }; - let progress_cb: ProgressCallback = Arc::new(move |update| { - update_crawl_progress(&pb_clone, &update); - }); + let progress_cb: Option<ProgressCallback> = if let Some(ref progress_bar) = pb { + let pb_clone = progress_bar.clone(); + Some(Arc::new(move |update| { + update_crawl_progress(&pb_clone, &update); + })) + } else { + None + }; let crawl_result = - run_crawl_with_options(&config, Some(progress_cb), writer_handle.clone(), None).await?; - finish_crawl_progress(&pb); + run_crawl_with_options(&config, progress_cb, writer_handle.clone(), None).await?; + + if let Some(ref progress_bar) = pb { + finish_crawl_progress(progress_bar); + } if let (Some(handle), Some(task)) = (writer_handle, writer_task) { let _ = handle.shutdown().await; @@ -121,8 +161,26 @@ async fn handle_audit(args: AuditArgs) -> Result<(), Box<dyn std::error::Error>> } } - // Always display executive terminal scorecard if requested - if formats.contains(&"terminal") || formats.contains(&"all") || formats.is_empty() { + if formats.contains(&"csv") || formats.contains(&"all") { + if let Ok(paths) = export_csv_suite(&crawl_result, &args.output_dir) { + if let Some(first) = paths.first() { + if let Some(parent) = first.parent() { + exported_artifacts.push(("CSV Suite", parent.to_path_buf())); + } + } + } + } + + if formats.contains(&"html") || formats.contains(&"all") { + if let Ok(path) = export_html_report(&crawl_result, &args.output_dir) { + exported_artifacts.push(("HTML", path)); + } + } + + // Always display executive terminal scorecard if requested and not in quiet mode + if (formats.contains(&"terminal") || formats.contains(&"all") || formats.is_empty()) + && !config.quiet + { let ref_paths: Vec<(&str, &std::path::Path)> = exported_artifacts .iter() .map(|(fmt, p)| (*fmt, p.as_path())) @@ -168,9 +226,52 @@ async fn handle_audit(args: AuditArgs) -> Result<(), Box<dyn std::error::Error>> /// Inspects a single page and prints its SEO metadata and audit issues. async fn handle_inspect(args: InspectArgs) -> Result<(), Box<dyn std::error::Error>> { let timeout = Duration::from_secs(args.timeout); - match inspect_url(&args.url, &args.user_agent, timeout).await { + let mut custom_headers = Vec::new(); + for h in args.headers { + if let Some((k, v)) = h.split_once(':') { + custom_headers.push((k.trim().to_string(), v.trim().to_string())); + } + } + + match inspect_url_with_options(&args.url, &args.user_agent, timeout, custom_headers).await { Ok((page, fetch, issues)) => { - print_page_inspection(&page, &fetch, &issues); + if args.format.eq_ignore_ascii_case("json") { + let json_output = serde_json::json!({ + "url": fetch.url, + "final_url": fetch.final_url, + "status_code": fetch.status_code, + "ttfb_ms": fetch.ttfb_ms, + "content_type": fetch.content_type, + "title": page.title, + "meta_description": page.meta_description, + "h1": page.h1_primary, + "canonical_url": page.canonical_url, + "word_count": page.word_count, + "issues": issues, + }); + println!("{}", serde_json::to_string_pretty(&json_output)?); + } else if args.format.eq_ignore_ascii_case("md") { + println!("# Page Inspection: {}\n", fetch.final_url); + println!("- **Status**: {}", fetch.status_code); + println!("- **TTFB**: {}ms", fetch.ttfb_ms); + println!("- **Title**: {}", page.title.as_deref().unwrap_or("None")); + println!("- **H1**: {}", page.h1_primary.as_deref().unwrap_or("None")); + println!( + "- **Canonical**: {}", + page.canonical_url.as_deref().unwrap_or("None") + ); + println!("\n## Detected Issues ({})", issues.len()); + for issue in &issues { + println!( + "- **[{:?}]** {}: {}", + issue.severity, + issue.code.as_str(), + issue.message + ); + } + } else { + print_page_inspection(&page, &fetch, &issues); + } Ok(()) } Err(err) => { @@ -182,17 +283,22 @@ async fn handle_inspect(args: InspectArgs) -> Result<(), Box<dyn std::error::Err /// Launches the native Model Context Protocol (MCP) server. async fn handle_mcp(args: McpArgs) -> Result<(), Box<dyn std::error::Error>> { - info!(transport = %args.transport, "Starting MCP server (scaffold)"); - println!( - "Starting SEO Lens MCP server on transport: {}", - args.transport - ); + let db_path = resolve_db_path(args.db_path, args.local); + if args.transport.eq_ignore_ascii_case("stdio") { + crate::mcp::run_mcp_server(Some(db_path)).await?; + } else { + eprintln!( + "❌ Transport '{}' is not currently supported. Please use '--transport stdio'.", + args.transport + ); + std::process::exit(1); + } Ok(()) } /// Inspects or re-exports an existing audit session from persistence. async fn handle_report(args: ReportArgs) -> Result<(), Box<dyn std::error::Error>> { - let db_path = args.db_path.unwrap_or_else(default_db_path); + let db_path = resolve_db_path(args.db_path, args.local); if !db_path.exists() { eprintln!( "❌ Persistence database not found at '{}'. Run an audit first.", @@ -267,6 +373,22 @@ async fn handle_report(args: ReportArgs) -> Result<(), Box<dyn std::error::Error } } + if formats.contains(&"csv") || formats.contains(&"all") { + if let Ok(paths) = export_csv_suite(&crawl_result, &output_dir) { + if let Some(first) = paths.first() { + if let Some(parent) = first.parent() { + exported_artifacts.push(("CSV Suite", parent.to_path_buf())); + } + } + } + } + + if formats.contains(&"html") || formats.contains(&"all") { + if let Ok(path) = export_html_report(&crawl_result, &output_dir) { + exported_artifacts.push(("HTML", path)); + } + } + if formats.contains(&"terminal") || formats.contains(&"all") || formats.is_empty() { let ref_paths: Vec<(&str, &std::path::Path)> = exported_artifacts .iter() @@ -280,16 +402,247 @@ async fn handle_report(args: ReportArgs) -> Result<(), Box<dyn std::error::Error /// Lists historical audit sessions stored locally. async fn handle_list(args: ListArgs) -> Result<(), Box<dyn std::error::Error>> { - let db_path = args.db_path.unwrap_or_else(default_db_path); + let db_path = resolve_db_path(args.db_path, args.local); if !db_path.exists() { - print_historical_sessions(&db_path, &[]); + if args.format.eq_ignore_ascii_case("json") { + println!("[]"); + } else { + print_historical_sessions(&db_path, &[]); + } return Ok(()); } let db = Database::open(&db_path)?; let crawls = db.list_crawls()?; - print_historical_sessions(&db_path, &crawls); + if args.format.eq_ignore_ascii_case("json") { + let displayed = if args.limit > 0 { + crawls.into_iter().take(args.limit).collect::<Vec<_>>() + } else { + crawls + }; + println!("{}", serde_json::to_string_pretty(&displayed)?); + } else { + let displayed = if args.limit > 0 { + crawls.into_iter().take(args.limit).collect::<Vec<_>>() + } else { + crawls + }; + print_historical_sessions(&db_path, &displayed); + } + + Ok(()) +} + +/// Drill down and filter audit findings for a session. +async fn handle_issues(args: IssuesArgs) -> Result<(), Box<dyn std::error::Error>> { + let session_id = match args.session.or(args.session_pos) { + Some(s) => s, + None => { + eprintln!("❌ Missing session ID. Usage: seolens issues <SESSION_ID> [OPTIONS]"); + std::process::exit(1); + } + }; + + let db_path = resolve_db_path(args.db_path, args.local); + if !db_path.exists() { + eprintln!( + "❌ Database not found at '{}'. Run an audit first.", + db_path.display() + ); + std::process::exit(1); + } + + let db = Database::open(&db_path)?; + let sev_filter = args + .severity + .as_deref() + .and_then(|s| match s.to_lowercase().as_str() { + "critical" => Some(Severity::Critical), + "alert" => Some(Severity::Alert), + "warning" => Some(Severity::Warning), + "notice" => Some(Severity::Notice), + _ => None, + }); + let cat_filter = args + .category + .as_deref() + .and_then(IssueCategory::from_str_name); + + let criteria = IssueFilterCriteria { + severity: sev_filter, + category: cat_filter, + code: args.code.as_deref(), + url_substring: args.url.as_deref(), + limit: args.limit, + offset: args.offset, + }; + + let total = db.count_issues_filtered(&session_id, &criteria)?; + let issues = db.query_issues_filtered(&session_id, &criteria)?; + + if args.format.eq_ignore_ascii_case("json") { + let out = serde_json::json!({ + "session_id": session_id, + "total_matching": total, + "offset": args.offset, + "limit": args.limit, + "issues": issues, + }); + println!("{}", serde_json::to_string_pretty(&out)?); + } else if args.format.eq_ignore_ascii_case("md") { + println!("# Audit Issues: {session_id}\n"); + println!( + "Found {total} matching issues (showing {}..{}):\n", + args.offset, + (args.offset + issues.len()).min(total) + ); + for (i, issue) in issues.iter().enumerate() { + println!( + "### {}. [{:?}] {}", + args.offset + i + 1, + issue.severity, + issue.code.as_str() + ); + println!("- **Target URL**: {}", issue.target_url); + println!("- **Category**: {:?}", issue.category); + println!("- **Message**: {}", issue.message); + if let Some(ref s) = issue.source_page_url { + println!("- **Source Page**: {}", s); + } + println!(); + } + } else { + print_issues_matrix(&session_id, &issues, total, args.offset, args.limit); + } + + Ok(()) +} + +/// Check website readiness for AI search engines (ChatGPT Search, Perplexity, Claude) and /llms.txt. +async fn handle_check_ai(args: CheckAiArgs) -> Result<(), Box<dyn std::error::Error>> { + let timeout = Duration::from_secs(args.timeout); + let report = audit_ai_readiness(&args.url, &args.user_agent, timeout).await?; + + if args.format.eq_ignore_ascii_case("json") { + println!("{}", serde_json::to_string_pretty(&report)?); + } else if args.format.eq_ignore_ascii_case("md") { + println!("# AI Search & GEO Readiness: {}\n", report.base_url); + println!("- **Citation Risk**: {}", report.citation_search_risk); + println!( + "- **/llms.txt**: {}", + if report.llms_txt_found { + "Found (200 OK)" + } else { + "Missing (404)" + } + ); + println!( + "- **/llms-full.txt**: {}", + if report.llms_full_txt_found { + "Found (200 OK)" + } else { + "Not published" + } + ); + println!("\n## Real-Time Search & Retrieval Bots"); + for (bot, status) in &report.retrieval_bots { + println!("- **{bot}**: {status}"); + } + println!("\n## AI Training Bots"); + for (bot, status) in &report.training_bots { + println!("- **{bot}**: {status}"); + } + if !report.recommendations.is_empty() { + println!("\n## Recommendations"); + for rec in &report.recommendations { + println!("- {rec}"); + } + } + } else { + print_ai_readiness_scorecard(&report); + } + + Ok(()) +} + +/// Delete a specific crawl session and its associated records. +async fn handle_delete(args: DeleteArgs) -> Result<(), Box<dyn std::error::Error>> { + let session_id = match args.session.or(args.session_pos) { + Some(s) => s, + None => { + eprintln!("❌ Missing session ID. Usage: seolens delete <SESSION_ID>"); + std::process::exit(1); + } + }; + + let db_path = resolve_db_path(args.db_path, args.local); + if !db_path.exists() { + eprintln!("❌ Database not found at '{}'.", db_path.display()); + std::process::exit(1); + } + + let db = Database::open(&db_path)?; + let deleted = db.delete_crawl(&session_id)?; + + if deleted { + println!("πŸ—‘οΈ Successfully deleted session '{session_id}' and all associated records."); + } else { + eprintln!("⚠️ Session '{session_id}' was not found in database."); + } + + Ok(()) +} + +/// Clean historical crawl sessions from the database. +async fn handle_clean(args: CleanArgs) -> Result<(), Box<dyn std::error::Error>> { + let db_path = resolve_db_path(args.db_path, args.local); + if !db_path.exists() { + println!( + "Database file '{}' does not exist. Nothing to clean.", + db_path.display() + ); + return Ok(()); + } + + let db = Database::open(&db_path)?; + + if args.all { + let count = db.clean_all_crawls()?; + println!("🧹 Successfully purged all {count} audit sessions from database."); + } else if let Some(days) = args.older_than { + let count = db.clean_crawls_older_than(days)?; + println!("🧹 Successfully purged {count} audit sessions older than {days} days."); + } else { + eprintln!("❌ Please specify either --older-than <days> or --all to clean sessions."); + std::process::exit(1); + } + + Ok(()) +} + +/// Validate JSON-LD / schema against Google Rich Results guidelines. +async fn handle_schema(args: SchemaArgs) -> Result<(), Box<dyn std::error::Error>> { + let raw_content = if args.target.starts_with("http://") || args.target.starts_with("https://") { + let client = HttpClient::new(FetchOptions { + user_agent: args.user_agent, + timeout: Duration::from_secs(15), + ..Default::default() + })?; + let res = client.fetch(&args.target).await?; + res.body + } else { + std::fs::read_to_string(&args.target) + .map_err(|e| format!("Failed to read schema file '{}': {e}", args.target))? + }; + + let outcome = validate_raw_schema(&raw_content, args.expected_type.as_deref())?; + + if args.format.eq_ignore_ascii_case("json") { + println!("{}", serde_json::to_string_pretty(&outcome)?); + } else { + print_schema_outcome(&outcome); + } Ok(()) } diff --git a/src/cli/mod.rs b/src/cli/mod.rs index 907afcc..4939852 100644 --- a/src/cli/mod.rs +++ b/src/cli/mod.rs @@ -8,10 +8,21 @@ pub mod commands; pub use args::{AuditArgs, Cli, Commands, InspectArgs, McpArgs, ReportArgs}; pub use commands::execute; +use crate::report::print_cli_help; use clap::Parser; /// Parses command-line arguments and dispatches execution to the corresponding command handler. pub async fn run() -> Result<(), Box<dyn std::error::Error>> { + let args: Vec<String> = std::env::args().collect(); + + // Intercept top-level help or no-arguments invocation to present the cyberpunk home/help TUI + if args.len() <= 1 + || (args.len() == 2 && (args[1] == "--help" || args[1] == "-h" || args[1] == "help")) + { + print_cli_help(); + return Ok(()); + } + let cli = Cli::parse(); execute(cli).await } diff --git a/src/core/config.rs b/src/core/config.rs index e39ea95..3ae5b7a 100644 --- a/src/core/config.rs +++ b/src/core/config.rs @@ -78,6 +78,16 @@ pub struct CrawlConfig { pub session_id: Option<String>, /// Optional path override for SQLite persistence database. pub db_path: Option<std::path::PathBuf>, + /// Optional regex pattern; only URLs matching this pattern will be crawled. + pub include_regex: Option<String>, + /// Optional regex pattern; URLs matching this pattern will be skipped. + pub exclude_regex: Option<String>, + /// Explicit sitemap XML URLs to seed or audit. + pub explicit_sitemaps: Vec<String>, + /// Suppress progress indicators for quiet/CI script execution. + pub quiet: bool, + /// Optional human-friendly audit name or project label. + pub crawl_name: Option<String>, } impl CrawlConfig { @@ -131,6 +141,11 @@ impl CrawlConfig { ignore_sorting_facets: true, session_id: None, db_path: None, + include_regex: None, + exclude_regex: None, + explicit_sitemaps: Vec::new(), + quiet: false, + crawl_name: None, }) } diff --git a/src/core/models.rs b/src/core/models.rs index 3ce4896..895345e 100644 --- a/src/core/models.rs +++ b/src/core/models.rs @@ -66,6 +66,22 @@ impl Severity { pub const fn as_u8(&self) -> u8 { *self as u8 } + + /// Returns the lowercase string identifier of the severity tier. + pub const fn as_str(&self) -> &'static str { + match self { + Self::Critical => "critical", + Self::Alert => "alert", + Self::Warning => "warning", + Self::Notice => "notice", + } + } +} + +impl std::fmt::Display for Severity { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "{}", self.as_str()) + } } /// Functional categories mapping to the 120-check technical SEO audit catalog. diff --git a/src/crawler/ai_check.rs b/src/crawler/ai_check.rs new file mode 100644 index 0000000..3dc237e --- /dev/null +++ b/src/crawler/ai_check.rs @@ -0,0 +1,220 @@ +//! # AI Search & GEO Readiness Auditor +//! +//! Evaluates a website's readiness for Generative Engine Optimization (GEO) and +//! citations by AI search models (ChatGPT Search, Perplexity, Claude, Gemini). +//! Inspects `/robots.txt` for retrieval and training crawlers and probes `/llms.txt`. + +use crate::crawler::client::{FetchOptions, HttpClient}; +use crate::crawler::robots::RobotsTxt; +use crate::error::{SeoError, SeoResult}; +use serde::{Deserialize, Serialize}; +use std::collections::HashMap; +use std::time::Duration; +use url::Url; + +/// Risk level indicating whether an origin is blocking or harming AI search engine citations. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "SCREAMING_SNAKE_CASE")] +pub enum AiSearchRisk { + /// All major search/retrieval bots are allowed and /llms.txt is present. + Low, + /// Missing /llms.txt or non-critical AI bots blocked. + Medium, + /// Critical AI citation/search engines (PerplexityBot, OAI-SearchBot) are blocked. + High, +} + +impl std::fmt::Display for AiSearchRisk { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + Self::Low => write!(f, "LOW"), + Self::Medium => write!(f, "MEDIUM"), + Self::High => write!(f, "HIGH"), + } + } +} + +/// Comprehensive readiness audit report for AI search engines. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct AiReadinessReport { + /// Base normalized domain root (e.g. `https://example.com`). + pub base_url: String, + /// Whether `/robots.txt` was reachable. + pub robots_found: bool, + /// Whether `/llms.txt` is published and returned HTTP 200 OK. + pub llms_txt_found: bool, + /// Initial summary snippet extracted from `/llms.txt`, if present. + pub llms_txt_summary: Option<String>, + /// Whether the extended `/llms-full.txt` file is published. + pub llms_full_txt_found: bool, + /// Status of AI search and real-time retrieval bots (e.g. PerplexityBot, OAI-SearchBot). + pub retrieval_bots: HashMap<String, String>, + /// Status of bulk AI foundation training bots (e.g. GPTBot, ClaudeBot, CCBot). + pub training_bots: HashMap<String, String>, + /// Overall citation risk rating. + pub citation_search_risk: AiSearchRisk, + /// Concrete, actionable recommendations to improve AI search visibility. + pub recommendations: Vec<String>, +} + +/// AI search crawlers that perform real-time retrieval and provide citation source links. +pub const RETRIEVAL_BOTS: &[&str] = &[ + "OAI-SearchBot", + "ChatGPT-User", + "PerplexityBot", + "Claude-User", + "Google-Extended", +]; + +/// AI crawlers that scrape content for model training data. +pub const TRAINING_BOTS: &[&str] = &[ + "GPTBot", + "ClaudeBot", + "CCBot", + "Bytespider", + "Meta-ExternalAgent", + "Amazonbot", + "Applebot-Extended", +]; + +/// Probes a website's `/robots.txt` and `/llms.txt` to produce an AI search readiness assessment. +/// +/// # Errors +/// +/// Returns [`SeoError::Url`] if the input URL is invalid or malformed. +pub async fn audit_ai_readiness( + target_url: &str, + user_agent: &str, + timeout: Duration, +) -> SeoResult<AiReadinessReport> { + let parsed = Url::parse(target_url) + .map_err(|e| SeoError::Url(format!("Invalid target URL '{target_url}': {e}")))?; + + let origin = format!("{}://{}", parsed.scheme(), parsed.authority()); + let robots_url = format!("{}/robots.txt", origin); + let llms_url = format!("{}/llms.txt", origin); + let llms_full_url = format!("{}/llms-full.txt", origin); + + let client = HttpClient::new(FetchOptions { + user_agent: user_agent.to_string(), + timeout, + connect_timeout: Duration::from_secs(5), + max_redirects: 5, + ..Default::default() + })?; + + // 1. Probe /robots.txt + let (robots_found, robots_txt) = match client.fetch(&robots_url).await { + Ok(res) if res.status_code == 200 => (true, Some(RobotsTxt::parse(&res.body))), + _ => (false, None), + }; + + let mut retrieval_bots = HashMap::new(); + for &bot in RETRIEVAL_BOTS { + let status = match &robots_txt { + Some(r) => { + if r.is_allowed(bot, "/") { + "ALLOWED".to_string() + } else { + "DISALLOWED".to_string() + } + } + None => "ALLOWED".to_string(), + }; + retrieval_bots.insert(bot.to_string(), status); + } + + let mut training_bots = HashMap::new(); + for &bot in TRAINING_BOTS { + let status = match &robots_txt { + Some(r) => { + if r.is_allowed(bot, "/") { + "ALLOWED".to_string() + } else { + "DISALLOWED".to_string() + } + } + None => "ALLOWED".to_string(), + }; + training_bots.insert(bot.to_string(), status); + } + + // 2. Probe /llms.txt + let (llms_txt_found, llms_txt_summary) = match client.fetch(&llms_url).await { + Ok(res) if res.status_code == 200 => { + let snippet = res + .body + .lines() + .filter(|l| !l.trim().is_empty()) + .take(3) + .collect::<Vec<_>>() + .join("\n"); + let summary = if snippet.is_empty() { + None + } else { + Some(snippet) + }; + (true, summary) + } + _ => (false, None), + }; + + // 3. Probe /llms-full.txt + let llms_full_txt_found = match client.fetch(&llms_full_url).await { + Ok(res) => res.status_code == 200, + Err(_) => false, + }; + + // 4. Assess Citation Search Risk & Formulate Recommendations + let mut recommendations = Vec::new(); + + let perplexity_disallowed = retrieval_bots + .get("PerplexityBot") + .map(|s| s == "DISALLOWED") + .unwrap_or(false); + let oai_search_disallowed = retrieval_bots + .get("OAI-SearchBot") + .map(|s| s == "DISALLOWED") + .unwrap_or(false); + + let citation_search_risk = if perplexity_disallowed || oai_search_disallowed { + AiSearchRisk::High + } else if !llms_txt_found { + AiSearchRisk::Medium + } else { + AiSearchRisk::Low + }; + + if perplexity_disallowed { + recommendations.push( + "PerplexityBot is disallowed in robots.txt. Your website cannot be cited as a reference in Perplexity search responses. Remove 'Disallow: /' for PerplexityBot.".to_string(), + ); + } + if oai_search_disallowed { + recommendations.push( + "OAI-SearchBot is disallowed in robots.txt. ChatGPT Search will not retrieve or cite your content in real-time answers. Allow OAI-SearchBot to regain visibility.".to_string(), + ); + } + if !llms_txt_found { + recommendations.push( + "Missing /llms.txt file. Publish a markdown summary at /llms.txt providing an authoritative index and technical overview for AI agents and LLMs.".to_string(), + ); + } + if !llms_full_txt_found && llms_txt_found { + recommendations.push( + "Publishing /llms-full.txt alongside /llms.txt allows AI models to consume complete documentation or product catalogues in a single high-density context file.".to_string(), + ); + } + + Ok(AiReadinessReport { + base_url: origin, + robots_found, + llms_txt_found, + llms_txt_summary, + llms_full_txt_found, + retrieval_bots, + training_bots, + citation_search_risk, + recommendations, + }) +} diff --git a/src/crawler/engine.rs b/src/crawler/engine.rs index 73c7d12..b9add82 100644 --- a/src/crawler/engine.rs +++ b/src/crawler/engine.rs @@ -176,12 +176,13 @@ fn build_page_report( /// 2. Audits AI search crawler disallows (`GPTBot`, `ClaudeBot`, etc.) per Rule 11.1. /// 3. Checks for presence of `/llms.txt` per Rule 11.2. /// 4. If no sitemaps are declared in robots.txt, falls back to probing convention paths -/// (`/sitemap.xml`, `/sitemap_index.xml`, `/wp-sitemap.xml`) per CRAWLER_SPEC Β§5.2. +/// (`/sitemap.xml`, `/sitemap_index.xml`, `/wp-sitemap.xml`) per docs/crawler.md Β§5.2. /// 5. Recursively resolves nested sitemap index feeds up to 3 levels deep. async fn discover_robots_and_sitemaps( client: &HttpClient, seed_url: &str, respect_robots: bool, + explicit_sitemaps: &[String], ) -> (Option<RobotsTxt>, Vec<String>, Vec<IssueFinding>) { let Ok(parsed_url) = url::Url::parse(seed_url) else { return (None, Vec::new(), Vec::new()); @@ -189,7 +190,7 @@ async fn discover_robots_and_sitemaps( let origin = format!("{}://{}", parsed_url.scheme(), parsed_url.authority()); let robots_url = format!("{}/robots.txt", origin); - let mut sitemap_feed_seeds = Vec::new(); + let mut sitemap_feed_seeds = explicit_sitemaps.to_vec(); let mut site_issues = Vec::new(); let fetched_robots = if let Ok(res) = client.fetch(&robots_url).await { @@ -250,7 +251,7 @@ async fn discover_robots_and_sitemaps( site_issues.push(rule.to_finding(&llms_url, Some(&msg))); } - // If robots.txt declared no sitemaps, probe standard conventions per CRAWLER_SPEC Β§5.2 + // If robots.txt declared no sitemaps, probe standard conventions per docs/crawler.md Β§5.2 if sitemap_feed_seeds.is_empty() { sitemap_feed_seeds.push(format!("{}/sitemap.xml", origin)); sitemap_feed_seeds.push(format!("{}/sitemap_index.xml", origin)); @@ -495,16 +496,32 @@ pub async fn run_crawl_with_options( user_agent: config.user_agent.clone(), timeout: Duration::from_secs(30), max_redirects: 10, + custom_headers: config.headers.clone(), + proxy: config.proxy.clone(), ..Default::default() })?); - let (robots_txt, sitemap_urls, site_issues) = - discover_robots_and_sitemaps(&client, &normalized_start, config.respect_robots).await; + let (robots_txt, sitemap_urls, site_issues) = discover_robots_and_sitemaps( + &client, + &normalized_start, + config.respect_robots, + &config.explicit_sitemaps, + ) + .await; if let Some(ref writer) = db_writer { let _ = writer.save_issues(site_issues.clone()).await; } + let include_re = config + .include_regex + .as_deref() + .and_then(|pat| regex::Regex::new(pat).ok()); + let exclude_re = config + .exclude_regex + .as_deref() + .and_then(|pat| regex::Regex::new(pat).ok()); + let frontier = Arc::new(Mutex::new(Frontier::new( config.max_pages, config.max_depth, @@ -512,7 +529,24 @@ pub async fn run_crawl_with_options( { let mut f = frontier.lock().await; - f.register_sitemap_urls(&sitemap_urls); + let filtered_sitemaps: Vec<String> = sitemap_urls + .iter() + .filter(|u| { + if let Some(ref inc) = include_re { + if !inc.is_match(u) { + return false; + } + } + if let Some(ref exc) = exclude_re { + if exc.is_match(u) { + return false; + } + } + true + }) + .cloned() + .collect(); + f.register_sitemap_urls(&filtered_sitemaps); f.push(&normalized_start, 0, None)?; } @@ -701,6 +735,20 @@ pub async fn run_crawl_with_options( continue; } + // Path filter: Include regex + if let Some(ref inc) = include_re { + if !inc.is_match(&link.target_url) { + continue; + } + } + + // Path filter: Exclude regex + if let Some(ref exc) = exclude_re { + if exc.is_match(&link.target_url) { + continue; + } + } + let _ = f.push(&link.target_url, outcome.depth + 1, Some(&outcome.report.url)); } f.enqueued_count() as usize diff --git a/src/crawler/inspector.rs b/src/crawler/inspector.rs index cfc610d..2ebfd7d 100644 --- a/src/crawler/inspector.rs +++ b/src/crawler/inspector.rs @@ -23,6 +23,16 @@ pub async fn inspect_url( url: &str, user_agent: &str, timeout: Duration, +) -> SeoResult<(ParsedPage, FetchResult, Vec<IssueFinding>)> { + inspect_url_with_options(url, user_agent, timeout, Vec::new()).await +} + +/// Fetches and analyzes a single webpage with custom HTTP request headers. +pub async fn inspect_url_with_options( + url: &str, + user_agent: &str, + timeout: Duration, + custom_headers: Vec<(String, String)>, ) -> SeoResult<(ParsedPage, FetchResult, Vec<IssueFinding>)> { let normalized = normalize_url(url)?; @@ -30,6 +40,7 @@ pub async fn inspect_url( user_agent: user_agent.to_string(), timeout, max_redirects: 10, + custom_headers, ..Default::default() })?; diff --git a/src/crawler/mod.rs b/src/crawler/mod.rs index 558ffce..e2b3302 100644 --- a/src/crawler/mod.rs +++ b/src/crawler/mod.rs @@ -12,6 +12,7 @@ //! - [`robots`]: RFC 9309 compliant `robots.txt` rule evaluator and crawl-delay parser. //! - [`sitemap`]: Streaming XML sitemap parser with alternates and gzip decompression. +pub mod ai_check; pub mod aimd; pub mod client; pub mod engine; @@ -22,13 +23,15 @@ pub mod robots; pub mod sitemap; pub mod waf; +pub use ai_check::{audit_ai_readiness, AiReadinessReport, AiSearchRisk}; + pub use aimd::AimdController; pub use client::{FetchOptions, FetchResult, HttpClient}; pub use engine::{ run_crawl, run_crawl_with_options, CrawlResult, ProgressCallback, ProgressUpdate, }; pub use frontier::{Frontier, FrontierEntry}; -pub use inspector::inspect_url; +pub use inspector::{inspect_url, inspect_url_with_options}; pub use priority::{calculate_url_importance, is_pagination_url, parse_url_segments}; pub use robots::RobotsTxt; pub use sitemap::{parse_sitemap, SitemapDocument, SitemapEntry, SitemapIndexEntry}; diff --git a/src/lib.rs b/src/lib.rs index 048f8a5..6cd90bc 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -74,6 +74,7 @@ pub mod core; pub mod crawler; pub mod error; pub mod graph; +pub mod mcp; pub mod parser; pub mod report; pub mod rules; diff --git a/src/mcp/formatter.rs b/src/mcp/formatter.rs new file mode 100644 index 0000000..facd591 --- /dev/null +++ b/src/mcp/formatter.rs @@ -0,0 +1,200 @@ +//! # LLM-Optimized Markdown Report Generator +//! +//! Generates dense, high-signal Markdown audit reports engineered specifically for +//! AI coding agents (Claude, Cursor, Windsurf, Antigravity) and LLM context windows: +//! - Strictly ZERO ANSI color escape sequences (`\x1b[`). +//! - Strictly ZERO decorative ASCII art banners (`β–ˆβ–ˆ`). +//! - Bounded URL sample listings (max 3–5 per rule) to conserve context tokens. +//! - Categorized issues with rule codes, problem statement, and concrete `Action for Agent` instructions. + +use crate::core::models::{CrawlSummary, IssueFinding, RuleId, Severity}; +use crate::rules::catalog::get_rule; +use hashbrown::HashMap; + +/// Formats a complete technical SEO audit into a token-efficient Markdown document for AI agents. +pub fn format_llm_markdown_report( + summary: &CrawlSummary, + issues: &[IssueFinding], + top_issues_limit: usize, + include_urls: bool, +) -> String { + let mut md = String::with_capacity(4096); + + let host = url::Url::parse(&summary.target_url) + .map(|u| u.host_str().unwrap_or("site").to_string()) + .unwrap_or_else(|_| summary.target_url.clone()); + + // 1. Executive Summary Header + md.push_str(&format!("# Technical SEO Audit: {}\n", host)); + md.push_str(&format!( + "**Health Score**: {}/100 | **Pages Crawled**: {}\n", + summary.health_score, summary.total_pages_crawled + )); + + let mut critical_count = 0; + let mut alert_count = 0; + let mut warning_count = 0; + let mut notice_count = 0; + + for issue in issues { + match issue.severity { + Severity::Critical => critical_count += 1, + Severity::Alert => alert_count += 1, + Severity::Warning => warning_count += 1, + Severity::Notice => notice_count += 1, + } + } + + md.push_str(&format!( + "**Summary**: {} Critical Issues, {} Alerts, {} Warnings, {} Notices\n\n---\n\n", + critical_count, alert_count, warning_count, notice_count + )); + + if issues.is_empty() { + md.push_str("βœ… **Zero technical SEO defects detected! All document, indexability, and graph checks passed.**\n"); + return md; + } + + // Group issues by RuleId + let mut grouped_issues: HashMap<RuleId, Vec<&IssueFinding>> = HashMap::new(); + for issue in issues { + grouped_issues.entry(issue.code).or_default().push(issue); + } + + let mut sorted_groups: Vec<_> = grouped_issues.into_iter().collect(); + sorted_groups.sort_by_key(|(_, list)| { + match list.first().map(|i| i.severity).unwrap_or(Severity::Notice) { + Severity::Critical => 0, + Severity::Alert => 1, + Severity::Warning => 2, + Severity::Notice => 3, + } + }); + + let limit = if top_issues_limit == 0 { + 20 + } else { + top_issues_limit + }; + + let mut critical_groups = Vec::new(); + let mut alert_groups = Vec::new(); + let mut warning_groups = Vec::new(); + let mut notice_groups = Vec::new(); + + for (rule_id, findings) in sorted_groups.into_iter().take(limit) { + let sev = findings + .first() + .map(|i| i.severity) + .unwrap_or(Severity::Notice); + match sev { + Severity::Critical => critical_groups.push((rule_id, findings)), + Severity::Alert => alert_groups.push((rule_id, findings)), + Severity::Warning => warning_groups.push((rule_id, findings)), + Severity::Notice => notice_groups.push((rule_id, findings)), + } + } + + // 2. Critical Issues Section + if !critical_groups.is_empty() { + md.push_str("## 🚨 Critical Issues (Immediate Fix Required)\n\n"); + for (idx, (rule_id, findings)) in critical_groups.iter().enumerate() { + append_issue_block(&mut md, idx + 1, *rule_id, findings, include_urls); + } + md.push_str("---\n\n"); + } + + // 3. High-Priority Alerts Section + if !alert_groups.is_empty() { + md.push_str("## ⚠️ High-Priority Alerts\n\n"); + for (idx, (rule_id, findings)) in alert_groups.iter().enumerate() { + append_issue_block(&mut md, idx + 1, *rule_id, findings, include_urls); + } + md.push_str("---\n\n"); + } + + // 4. Warnings Section + if !warning_groups.is_empty() { + md.push_str("## ⚑ Warnings\n\n"); + for (idx, (rule_id, findings)) in warning_groups.iter().enumerate() { + append_issue_block(&mut md, idx + 1, *rule_id, findings, include_urls); + } + md.push_str("---\n\n"); + } + + // 5. Notices Section + if !notice_groups.is_empty() { + md.push_str("## ℹ️ Informational Notices\n\n"); + for (idx, (rule_id, findings)) in notice_groups.iter().enumerate() { + append_issue_block(&mut md, idx + 1, *rule_id, findings, include_urls); + } + md.push_str("---\n\n"); + } + + // 6. Actionable Quick Wins for Agent + md.push_str("## πŸ’‘ Quick Wins for Agent\n"); + let mut win_idx = 1; + for (rule_id, findings) in critical_groups.iter().chain(alert_groups.iter()).take(5) { + let rule_def = get_rule(*rule_id); + if !rule_def.fix_advice.is_empty() { + md.push_str(&format!( + "{}. {} (Affects {} page{})\n", + win_idx, + rule_def.fix_advice, + findings.len(), + if findings.len() == 1 { "" } else { "s" } + )); + win_idx += 1; + } + } + + md +} + +fn append_issue_block( + md: &mut String, + num: usize, + rule_id: RuleId, + findings: &[&IssueFinding], + include_urls: bool, +) { + let rule_def = get_rule(rule_id); + let count = findings.len(); + let count_suffix = if count == 1 { "page" } else { "pages" }; + + md.push_str(&format!( + "### {}. `{}` (Affects {} {})\n", + num, + rule_id.as_str(), + count, + count_suffix + )); + + let problem_msg = findings + .first() + .map(|f| f.message.as_str()) + .unwrap_or(rule_def.title); + md.push_str(&format!("- **Problem**: {}\n", problem_msg)); + + if include_urls { + md.push_str("- **Affected URLs**:\n"); + // Limit sample URLs to at most 3 to conserve agent context window + for f in findings.iter().take(3) { + md.push_str(&format!(" - `{}`\n", f.target_url)); + } + if count > 3 { + md.push_str(&format!( + " - *...and {} more pages (query with `seo_query_issues`)*\n", + count - 3 + )); + } + } + + if !rule_def.fix_advice.is_empty() { + md.push_str(&format!( + "- **Action for Agent**: {}\n", + rule_def.fix_advice + )); + } + md.push('\n'); +} diff --git a/src/mcp/mod.rs b/src/mcp/mod.rs new file mode 100644 index 0000000..67654c3 --- /dev/null +++ b/src/mcp/mod.rs @@ -0,0 +1,16 @@ +//! # Model Context Protocol (MCP) Server +//! +//! Exposes SEO Lens inspection tools and crawl session resources to AI agents +//! via standard JSON-RPC 2.0 over `stdio`. + +pub mod formatter; +pub mod protocol; +pub mod resources; +pub mod server; +pub mod tools; +pub mod types; + +pub use protocol::{handle_jsonrpc_request, McpContext}; +pub use server::{run_mcp_server, run_mcp_server_io}; +pub use tools::get_tool_definitions; +pub use types::*; diff --git a/src/mcp/protocol.rs b/src/mcp/protocol.rs new file mode 100644 index 0000000..9502ed1 --- /dev/null +++ b/src/mcp/protocol.rs @@ -0,0 +1,156 @@ +//! # JSON-RPC 2.0 MCP Protocol Dispatcher +//! +//! Handles request routing, protocol handshake, tool invocations, and resource queries +//! conforming to the Model Context Protocol (2024-11-05). + +use crate::mcp::resources::{get_resource_definitions, read_resource}; +use crate::mcp::tools::{execute_tool, get_tool_definitions}; +use crate::mcp::types::{ + CallToolResult, JsonRpcRequest, JsonRpcResponse, INVALID_PARAMS, METHOD_NOT_FOUND, PARSE_ERROR, +}; +use crate::storage::{default_db_path, Database}; +use serde_json::json; +use std::path::PathBuf; + +/// Shared runtime context for the MCP server. +#[derive(Debug, Clone)] +pub struct McpContext { + /// SQLite database handle for state inspection and crawl records. + pub db: Database, +} + +impl McpContext { + /// Creates a new MCP context with an optional database path. + /// + /// Falls back to default `.seolens/seolens.db` or a system temporary directory + /// if creation fails, guaranteeing zero panics. + pub fn new(db_path: Option<PathBuf>) -> Self { + let path = db_path.unwrap_or_else(default_db_path); + let db = Database::open(&path).unwrap_or_else(|_| Database::from_path(path)); + Self { db } + } + + /// Creates a context with a pre-configured database instance. + pub fn with_database(db: Database) -> Self { + Self { db } + } +} + +/// Dispatches a single JSON-RPC 2.0 request string and returns the serialized JSON-RPC response. +pub async fn handle_jsonrpc_request(raw_json: &str, ctx: &McpContext) -> String { + let request: JsonRpcRequest = match serde_json::from_str(raw_json) { + Ok(req) => req, + Err(e) => { + let err_resp = JsonRpcResponse::error( + None, + PARSE_ERROR, + format!("Failed to parse JSON-RPC request: {e}"), + None, + ); + return serde_json::to_string(&err_resp).unwrap_or_else(|_| "{}".to_string()); + } + }; + + let id = request.id.clone(); + let method = request.method.as_str(); + + let response = match method { + "initialize" => { + let server_version = env!("CARGO_PKG_VERSION"); + JsonRpcResponse::success( + id, + json!({ + "protocolVersion": "2024-11-05", + "capabilities": { + "tools": {}, + "resources": {} + }, + "serverInfo": { + "name": "seolens", + "version": server_version + } + }), + ) + } + "notifications/initialized" => { + if id.is_some() { + JsonRpcResponse::success(id, json!({})) + } else { + return String::new(); + } + } + "ping" => JsonRpcResponse::success(id, json!({})), + "tools/list" => { + let tools = get_tool_definitions(); + JsonRpcResponse::success(id, json!({ "tools": tools })) + } + "tools/call" => { + let params = request.params.unwrap_or_else(|| json!({})); + let tool_name = match params["name"].as_str() { + Some(n) => n, + None => { + return serde_json::to_string(&JsonRpcResponse::error( + id, + INVALID_PARAMS, + "Missing required parameter 'name' in tools/call", + None, + )) + .unwrap_or_else(|_| "{}".to_string()); + } + }; + + let tool_args = params.get("arguments"); + match execute_tool(tool_name, tool_args, &ctx.db).await { + Ok(res) => { + let val = serde_json::to_value(&res).unwrap_or_else(|_| json!({})); + JsonRpcResponse::success(id, val) + } + Err(e) => { + let err_call = CallToolResult::error(format!("Tool error: {e}")); + let val = serde_json::to_value(&err_call).unwrap_or_else(|_| json!({})); + JsonRpcResponse::success(id, val) + } + } + } + "resources/list" => { + let resources = get_resource_definitions(&ctx.db); + JsonRpcResponse::success(id, json!({ "resources": resources })) + } + "resources/read" => { + let params = request.params.unwrap_or_else(|| json!({})); + let uri = match params["uri"].as_str() { + Some(u) => u, + None => { + return serde_json::to_string(&JsonRpcResponse::error( + id, + INVALID_PARAMS, + "Missing required parameter 'uri' in resources/read", + None, + )) + .unwrap_or_else(|_| "{}".to_string()); + } + }; + + match read_resource(uri, &ctx.db) { + Ok(res) => { + let val = serde_json::to_value(&res).unwrap_or_else(|_| json!({})); + JsonRpcResponse::success(id, val) + } + Err(e) => JsonRpcResponse::error( + id, + -32002, // Resource not found or error + format!("Failed to read resource '{uri}': {e}"), + None, + ), + } + } + _ => JsonRpcResponse::error( + id, + METHOD_NOT_FOUND, + format!("Method '{method}' not found"), + None, + ), + }; + + serde_json::to_string(&response).unwrap_or_else(|_| "{}".to_string()) +} diff --git a/src/mcp/resources.rs b/src/mcp/resources.rs new file mode 100644 index 0000000..ad7cb3d --- /dev/null +++ b/src/mcp/resources.rs @@ -0,0 +1,126 @@ +//! # MCP Resource Handlers +//! +//! Exposes read-only Model Context Protocol URI resources for LLM agents: +//! - `seo://crawls`: Lists all recent crawl sessions in SQLite. +//! - `seo://crawls/{session_id}/summary`: JSON summary of crawl metrics and health score. +//! - `seo://crawls/{session_id}/report`: Full token-efficient Markdown report. +//! - `seo://crawls/{session_id}/issues`: Raw JSON array of all issues. + +use crate::error::{SeoError, SeoResult}; +use crate::mcp::formatter::format_llm_markdown_report; +use crate::mcp::types::{ReadResourceResult, ResourceContents, ResourceDefinition}; +use crate::storage::Database; + +/// Returns definitions for all available MCP resources. +pub fn get_resource_definitions(db: &Database) -> Vec<ResourceDefinition> { + let mut defs = vec![ResourceDefinition { + uri: "seo://crawls".to_string(), + name: "Recent Crawl Sessions".to_string(), + description: Some("Lists all historical and active crawl sessions in SQLite".to_string()), + mime_type: Some("application/json".to_string()), + }]; + + // Also advertise resources for recent crawl sessions + if let Ok(crawls) = db.list_crawls() { + for crawl in crawls.iter().take(20) { + defs.push(ResourceDefinition { + uri: format!("seo://crawls/{}/summary", crawl.session_id), + name: format!("Crawl Summary: {}", crawl.session_id), + description: Some(format!("Summary metrics for {}", crawl.target_url)), + mime_type: Some("application/json".to_string()), + }); + defs.push(ResourceDefinition { + uri: format!("seo://crawls/{}/report", crawl.session_id), + name: format!("Markdown Audit Report: {}", crawl.session_id), + description: Some(format!( + "LLM-optimized audit report for {}", + crawl.target_url + )), + mime_type: Some("text/markdown".to_string()), + }); + defs.push(ResourceDefinition { + uri: format!("seo://crawls/{}/issues", crawl.session_id), + name: format!("Audit Issues: {}", crawl.session_id), + description: Some(format!( + "All detected issue findings for {}", + crawl.target_url + )), + mime_type: Some("application/json".to_string()), + }); + } + } + + defs +} + +/// Reads the resource content for a given URI. +pub fn read_resource(uri: &str, db: &Database) -> SeoResult<ReadResourceResult> { + let trimmed = uri.trim(); + + if trimmed == "seo://crawls" { + let crawls = db.list_crawls()?; + let text = serde_json::to_string_pretty(&crawls) + .map_err(|e| SeoError::Internal(format!("Failed to serialize crawl sessions: {e}")))?; + return Ok(ReadResourceResult { + contents: vec![ResourceContents { + uri: trimmed.to_string(), + mime_type: Some("application/json".to_string()), + text, + }], + }); + } + + if let Some(rest) = trimmed.strip_prefix("seo://crawls/") { + let parts: Vec<&str> = rest.split('/').collect(); + if parts.len() == 2 { + let session_id = parts[0]; + let sub_resource = parts[1]; + + let crawl = db.get_crawl(session_id)?.ok_or_else(|| { + SeoError::Storage(format!("Crawl session '{session_id}' not found")) + })?; + + match sub_resource { + "summary" => { + let text = serde_json::to_string_pretty(&crawl).map_err(|e| { + SeoError::Internal(format!("Failed to serialize crawl summary: {e}")) + })?; + return Ok(ReadResourceResult { + contents: vec![ResourceContents { + uri: trimmed.to_string(), + mime_type: Some("application/json".to_string()), + text, + }], + }); + } + "report" => { + let issues = db.get_crawl_issues(session_id, None, None)?; + let text = format_llm_markdown_report(&crawl, &issues, 50, true); + return Ok(ReadResourceResult { + contents: vec![ResourceContents { + uri: trimmed.to_string(), + mime_type: Some("text/markdown".to_string()), + text, + }], + }); + } + "issues" => { + let issues = db.get_crawl_issues(session_id, None, None)?; + let text = serde_json::to_string_pretty(&issues).map_err(|e| { + SeoError::Internal(format!("Failed to serialize crawl issues: {e}")) + })?; + return Ok(ReadResourceResult { + contents: vec![ResourceContents { + uri: trimmed.to_string(), + mime_type: Some("application/json".to_string()), + text, + }], + }); + } + _ => {} + } + } + } + + Err(SeoError::Internal(format!("Resource not found: '{uri}'"))) +} diff --git a/src/mcp/server.rs b/src/mcp/server.rs new file mode 100644 index 0000000..64be968 --- /dev/null +++ b/src/mcp/server.rs @@ -0,0 +1,49 @@ +//! # MCP Stdio Server Runner +//! +//! Provides the asynchronous stdio JSON-RPC 2.0 loop reading from `stdin` +//! and writing responses to `stdout`. + +use crate::error::SeoResult; +use crate::mcp::protocol::{handle_jsonrpc_request, McpContext}; +use std::path::PathBuf; +use tokio::io::{AsyncBufRead, AsyncBufReadExt, AsyncWrite, AsyncWriteExt}; + +/// Runs the MCP JSON-RPC 2.0 server over generic asynchronous reader and writer streams. +pub async fn run_mcp_server_io<R, W>( + reader: R, + mut writer: W, + db_path: Option<PathBuf>, +) -> SeoResult<()> +where + R: AsyncBufRead + Unpin, + W: AsyncWrite + Unpin, +{ + let ctx = McpContext::new(db_path); + let mut lines = reader.lines(); + + while let Some(line) = lines.next_line().await? { + let trimmed = line.trim(); + if trimmed.is_empty() { + continue; + } + + let response = handle_jsonrpc_request(trimmed, &ctx).await; + if !response.is_empty() { + writer.write_all(response.as_bytes()).await?; + writer.write_all(b"\n").await?; + writer.flush().await?; + } + } + + Ok(()) +} + +/// Runs the MCP server over standard input (`stdin`) and standard output (`stdout`). +/// +/// Log messages and diagnostics are sent to `stderr` so as not to corrupt JSON-RPC frames on `stdout`. +pub async fn run_mcp_server(db_path: Option<PathBuf>) -> SeoResult<()> { + let stdin = tokio::io::stdin(); + let stdout = tokio::io::stdout(); + let reader = tokio::io::BufReader::new(stdin); + run_mcp_server_io(reader, stdout, db_path).await +} diff --git a/src/mcp/tools.rs b/src/mcp/tools.rs new file mode 100644 index 0000000..b981adf --- /dev/null +++ b/src/mcp/tools.rs @@ -0,0 +1,599 @@ +//! # MCP Agent Tools +//! +//! Implements the 8 core Model Context Protocol tools for AI agent pair-programming: +//! 1. `seo_start_audit`: Non-blocking async crawl kickoff (< 1.0s). +//! 2. `seo_audit_status`: Live crawl telemetry & issue counter polling. +//! 3. `seo_get_markdown_report`: Token-efficient Markdown report generation (zero ANSI). +//! 4. `seo_quick_page_check`: Synchronous single-page audit (< 500ms). +//! 5. `seo_query_issues`: Filtered query over SQLite issues table. +//! 6. `seo_check_ai_readiness`: Standalone Generative Engine Optimization audit. +//! 7. `seo_validate_schema`: Google Rich Results structured data validation. +//! 8. `seo_cleanup_session`: Drops session and cascades records from SQLite. + +use crate::core::config::CrawlConfig; +use crate::core::models::{IssueCategory, Severity}; +use crate::core::url::normalize_url; +use crate::crawler::ai_check::audit_ai_readiness; +use crate::crawler::engine::run_crawl_with_options; +use crate::crawler::inspector::inspect_url_with_options; +use crate::error::SeoResult; +use crate::mcp::formatter::format_llm_markdown_report; +use crate::mcp::types::{CallToolResult, ToolDefinition}; +use crate::rules::page::schema_val::validate_raw_schema; +use crate::storage::{CrawlSessionInit, Database, IssueFilterCriteria}; +use serde_json::{json, Value}; +use std::time::{Duration, SystemTime, UNIX_EPOCH}; + +/// Returns the static catalog of all 8 MCP tool definitions and their JSON schemas. +pub fn get_tool_definitions() -> Vec<ToolDefinition> { + vec![ + ToolDefinition { + name: "seo_start_audit".to_string(), + description: "Kicks off a background crawl and technical SEO audit for an entire website. Returns immediately with a session_id in < 1.0s.".to_string(), + input_schema: json!({ + "type": "object", + "properties": { + "url": { + "type": "string", + "format": "uri", + "description": "The root or starting URL to crawl (e.g. 'https://example.com')." + }, + "max_pages": { + "type": "integer", + "default": 500, + "minimum": 1, + "maximum": 50000, + "description": "Maximum number of pages to crawl." + }, + "max_depth": { + "type": "integer", + "default": 5, + "minimum": 1, + "maximum": 20, + "description": "Maximum click depth from the start URL." + }, + "render_js": { + "type": "boolean", + "default": false, + "description": "Enable headless Chrome CDP to render JavaScript and audit SPAs." + }, + "respect_robots": { + "type": "boolean", + "default": true, + "description": "Whether to fetch and obey /robots.txt rules." + }, + "ai_geo_audit": { + "type": "boolean", + "default": true, + "description": "Audit /llms.txt and AI bot crawler accessibility." + } + }, + "required": ["url"] + }), + }, + ToolDefinition { + name: "seo_audit_status".to_string(), + description: "Polls the live progress and operational telemetry of an active or finished crawl session.".to_string(), + input_schema: json!({ + "type": "object", + "properties": { + "session_id": { + "type": "string", + "description": "The unique audit session ID returned by seo_start_audit." + } + }, + "required": ["session_id"] + }), + }, + ToolDefinition { + name: "seo_get_markdown_report".to_string(), + description: "Generates a structured, token-efficient Markdown audit report engineered specifically for LLM context windows (zero ANSI codes, zero ASCII art, high signal-to-noise ratio).".to_string(), + input_schema: json!({ + "type": "object", + "properties": { + "session_id": { + "type": "string", + "description": "The audit session ID." + }, + "top_issues_limit": { + "type": "integer", + "default": 20, + "description": "Maximum number of distinct issue types to summarize." + }, + "include_urls": { + "type": "boolean", + "default": true, + "description": "Include sample affected URLs for each issue." + } + }, + "required": ["session_id"] + }), + }, + ToolDefinition { + name: "seo_quick_page_check".to_string(), + description: "Performs an instant, synchronous audit of a single URL in < 500ms. Ideal for testing a specific landing page or verifying a code fix immediately.".to_string(), + input_schema: json!({ + "type": "object", + "properties": { + "url": { + "type": "string", + "format": "uri", + "description": "Single URL to fetch and audit." + }, + "render_js": { + "type": "boolean", + "default": false, + "description": "Execute JavaScript via headless Chrome." + } + }, + "required": ["url"] + }), + }, + ToolDefinition { + name: "seo_query_issues".to_string(), + description: "Queries specific issues from an audit database, filtered by severity, category, or URL pattern.".to_string(), + input_schema: json!({ + "type": "object", + "properties": { + "session_id": { + "type": "string", + "description": "The audit session ID." + }, + "severity": { + "type": "string", + "enum": ["critical", "alert", "warning", "notice"], + "description": "Optional severity filter." + }, + "category": { + "type": "string", + "description": "Optional category filter (e.g. 'canonicalization', 'security', 'indexability', 'titles')." + }, + "url_pattern": { + "type": "string", + "description": "Optional URL substring or glob pattern (e.g. '/blog/')." + }, + "limit": { + "type": "integer", + "default": 50, + "maximum": 500, + "description": "Number of records to return." + } + }, + "required": ["session_id"] + }), + }, + ToolDefinition { + name: "seo_check_ai_readiness".to_string(), + description: "Audits whether a website is optimized for Generative Engine Optimization (GEO) and AI Search Engines (ChatGPT Search, Perplexity, Claude) and checks /llms.txt.".to_string(), + input_schema: json!({ + "type": "object", + "properties": { + "url": { + "type": "string", + "format": "uri", + "description": "The website base URL." + } + }, + "required": ["url"] + }), + }, + ToolDefinition { + name: "seo_validate_schema".to_string(), + description: "Validates a raw JSON-LD or Schema.org block against Google Rich Results eligibility rules.".to_string(), + input_schema: json!({ + "type": "object", + "properties": { + "json_ld": { + "type": "string", + "description": "The raw JSON-LD string or object snippet." + }, + "target_type": { + "type": "string", + "description": "Optional expected @type (e.g. 'Product', 'Article', 'FAQPage', 'BreadcrumbList')." + } + }, + "required": ["json_ld"] + }), + }, + ToolDefinition { + name: "seo_cleanup_session".to_string(), + description: "Deletes audit records and artifacts from disk for a crawl session.".to_string(), + input_schema: json!({ + "type": "object", + "properties": { + "session_id": { + "type": "string", + "description": "The audit session ID to purge." + } + }, + "required": ["session_id"] + }), + }, + ] +} + +/// Dispatches an MCP tool call to its corresponding handler. +pub async fn execute_tool( + name: &str, + args: Option<&Value>, + db: &Database, +) -> SeoResult<CallToolResult> { + match name { + "seo_start_audit" => tool_start_audit(args, db).await, + "seo_audit_status" => tool_audit_status(args, db).await, + "seo_get_markdown_report" => tool_get_markdown_report(args, db).await, + "seo_quick_page_check" => tool_quick_page_check(args).await, + "seo_query_issues" => tool_query_issues(args, db).await, + "seo_check_ai_readiness" => tool_check_ai_readiness(args).await, + "seo_validate_schema" => tool_validate_schema(args).await, + "seo_cleanup_session" => tool_cleanup_session(args, db).await, + _ => Ok(CallToolResult::error(format!("Unknown tool: '{name}'"))), + } +} + +async fn tool_start_audit(args: Option<&Value>, db: &Database) -> SeoResult<CallToolResult> { + let args_val = args.cloned().unwrap_or_else(|| json!({})); + let raw_url = match args_val["url"].as_str() { + Some(u) if !u.trim().is_empty() => u.trim(), + _ => return Ok(CallToolResult::error("Missing required parameter 'url'")), + }; + + let target_url = match normalize_url(raw_url) { + Ok(u) => u, + Err(e) => return Ok(CallToolResult::error(format!("Invalid URL: {e}"))), + }; + + let max_pages = args_val["max_pages"].as_u64().unwrap_or(500) as u32; + let max_depth = args_val["max_depth"].as_u64().unwrap_or(5) as u16; + let respect_robots = args_val["respect_robots"].as_bool().unwrap_or(true); + let render_js = args_val["render_js"].as_bool().unwrap_or(false); + + let ts = SystemTime::now() + .duration_since(UNIX_EPOCH) + .map(|d| d.as_secs()) + .unwrap_or(0); + let session_id = format!("crawl_{ts}"); + + // Register initial session record in SQLite + let init = CrawlSessionInit { + session_id: session_id.clone(), + target_url: target_url.clone(), + max_pages, + max_depth, + respect_robots, + render_js, + }; + db.init_crawl_session(&init)?; + + // Spawn non-blocking background Tokio task + let background_db = db.clone(); + let background_url = target_url.clone(); + let background_session_id = session_id.clone(); + tokio::spawn(async move { + if let Ok(mut config) = CrawlConfig::new(&background_url) { + config.session_id = Some(background_session_id.clone()); + config.max_pages = max_pages; + config.max_depth = max_depth; + config.respect_robots = respect_robots; + config.render_js = render_js; + config.quiet = true; // headless background task + + if let Ok((writer_handle, writer_task)) = + background_db.spawn_writer(&background_session_id, 20, Duration::from_millis(500)) + { + let crawl_res = + run_crawl_with_options(&config, None, Some(writer_handle), None).await; + let _ = writer_task.await; + + if let Ok(res) = crawl_res { + let errors = res + .issues + .iter() + .filter(|i| i.severity == Severity::Critical) + .count() as u32; + let alerts = res + .issues + .iter() + .filter(|i| i.severity == Severity::Alert) + .count() as u32; + let warnings = res + .issues + .iter() + .filter(|i| i.severity == Severity::Warning) + .count() as u32; + + let _ = background_db.update_crawl_status( + &background_session_id, + "completed", + None, + res.pages.len() as u32, + errors, + alerts, + warnings, + Some(res.health_score), + ); + } + } + } + }); + + let res = json!({ + "session_id": session_id, + "status": "queued", + "target_url": raw_url, + "message": "Audit started in background. Poll 'seo_audit_status' with session_id to monitor progress.", + "poll_interval_seconds": 15 + }); + + Ok(CallToolResult::success_json(&res)) +} + +async fn tool_audit_status(args: Option<&Value>, db: &Database) -> SeoResult<CallToolResult> { + let args_val = args.cloned().unwrap_or_else(|| json!({})); + let session_id = match args_val["session_id"].as_str() { + Some(s) if !s.trim().is_empty() => s.trim(), + _ => { + return Ok(CallToolResult::error( + "Missing required parameter 'session_id'", + )) + } + }; + + let crawl = match db.get_crawl(session_id)? { + Some(c) => c, + None => { + return Ok(CallToolResult::error(format!( + "Session '{session_id}' not found in database" + ))) + } + }; + + let is_complete = + crawl.status == "completed" || crawl.status == "interrupted" || crawl.status == "failed"; + + let res = json!({ + "session_id": session_id, + "status": crawl.status, + "pages_crawled": crawl.total_pages_crawled, + "pages_discovered": crawl.total_links_discovered, + "current_delay_ms": 0, + "p95_ttfb_ms": crawl.p95_ttfb_ms, + "error_rate_pct": 0.0, + "issues_count": { + "critical": crawl.total_errors, + "alert": crawl.total_alerts, + "warning": crawl.total_warnings, + "notice": crawl.total_notices + }, + "health_score": crawl.health_score, + "is_complete": is_complete + }); + + Ok(CallToolResult::success_json(&res)) +} + +async fn tool_get_markdown_report( + args: Option<&Value>, + db: &Database, +) -> SeoResult<CallToolResult> { + let args_val = args.cloned().unwrap_or_else(|| json!({})); + let session_id = match args_val["session_id"].as_str() { + Some(s) if !s.trim().is_empty() => s.trim(), + _ => { + return Ok(CallToolResult::error( + "Missing required parameter 'session_id'", + )) + } + }; + + let crawl = match db.get_crawl(session_id)? { + Some(c) => c, + None => { + return Ok(CallToolResult::error(format!( + "Session '{session_id}' not found in database" + ))) + } + }; + + let top_limit = args_val["top_issues_limit"].as_u64().unwrap_or(20) as usize; + let include_urls = args_val["include_urls"].as_bool().unwrap_or(true); + + let issues = db.get_crawl_issues(session_id, None, None)?; + let report_md = format_llm_markdown_report(&crawl, &issues, top_limit, include_urls); + + Ok(CallToolResult::success_text(report_md)) +} + +async fn tool_quick_page_check(args: Option<&Value>) -> SeoResult<CallToolResult> { + let args_val = args.cloned().unwrap_or_else(|| json!({})); + let raw_url = match args_val["url"].as_str() { + Some(u) if !u.trim().is_empty() => u.trim(), + _ => return Ok(CallToolResult::error("Missing required parameter 'url'")), + }; + + let (parsed_page, fetch_result, issues) = + match inspect_url_with_options(raw_url, "SEOLensBot/1.0", Duration::from_secs(15), vec![]) + .await + { + Ok(t) => t, + Err(e) => { + return Ok(CallToolResult::error(format!( + "Failed to fetch '{raw_url}': {e}" + ))) + } + }; + + let issues_json: Vec<Value> = issues + .iter() + .map(|i| { + json!({ + "code": i.code.as_str(), + "severity": i.severity.as_str(), + "message": i.message + }) + }) + .collect(); + + let is_indexable = !parsed_page + .robots_flags + .contains(crate::core::models::RobotsFlags::NOINDEX) + && fetch_result.status_code == 200; + + let res = json!({ + "url": fetch_result.final_url, + "status_code": fetch_result.status_code, + "ttfb_ms": fetch_result.ttfb_ms, + "title": parsed_page.title.as_deref().unwrap_or(""), + "meta_description": parsed_page.meta_description.as_deref().unwrap_or(""), + "h1": parsed_page.h1_primary.as_deref().unwrap_or(""), + "canonical_url": parsed_page.canonical_url.as_deref().unwrap_or(""), + "word_count": parsed_page.word_count, + "is_indexable": is_indexable, + "issues_detected": issues_json + }); + + Ok(CallToolResult::success_json(&res)) +} + +async fn tool_query_issues(args: Option<&Value>, db: &Database) -> SeoResult<CallToolResult> { + let args_val = args.cloned().unwrap_or_else(|| json!({})); + let session_id = match args_val["session_id"].as_str() { + Some(s) if !s.trim().is_empty() => s.trim(), + _ => { + return Ok(CallToolResult::error( + "Missing required parameter 'session_id'", + )) + } + }; + + let sev = args_val["severity"] + .as_str() + .and_then(|s| match s.to_lowercase().as_str() { + "critical" => Some(Severity::Critical), + "alert" => Some(Severity::Alert), + "warning" => Some(Severity::Warning), + "notice" => Some(Severity::Notice), + _ => None, + }); + + let cat = args_val["category"] + .as_str() + .and_then(IssueCategory::from_str_name); + let url_pattern = args_val["url_pattern"].as_str(); + let limit = args_val["limit"].as_u64().unwrap_or(50) as usize; + + let criteria = IssueFilterCriteria { + severity: sev, + category: cat, + code: None, + url_substring: url_pattern, + limit, + offset: 0, + }; + + let total = db.count_issues_filtered(session_id, &criteria)?; + let issues = db.query_issues_filtered(session_id, &criteria)?; + + let issues_json: Vec<Value> = issues + .iter() + .map(|i| { + json!({ + "code": i.code.as_str(), + "severity": i.severity.as_str(), + "category": i.category.as_str(), + "target_url": i.target_url, + "message": i.message, + "source_page_url": i.source_page_url + }) + }) + .collect(); + + let res = json!({ + "total_matching": total, + "issues": issues_json + }); + + Ok(CallToolResult::success_json(&res)) +} + +async fn tool_check_ai_readiness(args: Option<&Value>) -> SeoResult<CallToolResult> { + let args_val = args.cloned().unwrap_or_else(|| json!({})); + let raw_url = match args_val["url"].as_str() { + Some(u) if !u.trim().is_empty() => u.trim(), + _ => return Ok(CallToolResult::error("Missing required parameter 'url'")), + }; + + let report = match audit_ai_readiness(raw_url, "SEOLensBot/1.0", Duration::from_secs(15)).await + { + Ok(r) => r, + Err(e) => { + return Ok(CallToolResult::error(format!( + "AI readiness check failed: {e}" + ))) + } + }; + + let res = json!({ + "target_url": report.base_url, + "robots_found": report.robots_found, + "llms_txt_found": report.llms_txt_found, + "llms_txt_summary": report.llms_txt_summary, + "llms_full_txt_found": report.llms_full_txt_found, + "ai_crawler_access": { + "retrieval_citation_bots": report.retrieval_bots, + "foundation_training_bots": report.training_bots, + }, + "citation_risk": report.citation_search_risk.to_string(), + "recommendations": report.recommendations, + }); + + Ok(CallToolResult::success_json(&res)) +} + +async fn tool_validate_schema(args: Option<&Value>) -> SeoResult<CallToolResult> { + let args_val = args.cloned().unwrap_or_else(|| json!({})); + let json_ld = match args_val["json_ld"].as_str() { + Some(j) if !j.trim().is_empty() => j.trim(), + _ => { + return Ok(CallToolResult::error( + "Missing required parameter 'json_ld'", + )) + } + }; + + let target_type = args_val["target_type"].as_str(); + let outcome = validate_raw_schema(json_ld, target_type)?; + + let res = json!({ + "is_valid_json": outcome.is_valid_json, + "detected_type": outcome.detected_type, + "is_rich_result_eligible": outcome.is_rich_result_eligible, + "missing_required_fields": outcome.missing_required_fields, + "missing_recommended_fields": outcome.missing_recommended_fields, + "error_message": outcome.error_message + }); + + Ok(CallToolResult::success_json(&res)) +} + +async fn tool_cleanup_session(args: Option<&Value>, db: &Database) -> SeoResult<CallToolResult> { + let args_val = args.cloned().unwrap_or_else(|| json!({})); + let session_id = match args_val["session_id"].as_str() { + Some(s) if !s.trim().is_empty() => s.trim(), + _ => { + return Ok(CallToolResult::error( + "Missing required parameter 'session_id'", + )) + } + }; + + let purged = db.delete_crawl(session_id)?; + + let res = json!({ + "session_id": session_id, + "purged": purged, + "records_freed": if purged { 1 } else { 0 } + }); + + Ok(CallToolResult::success_json(&res)) +} diff --git a/src/mcp/types.rs b/src/mcp/types.rs new file mode 100644 index 0000000..2285415 --- /dev/null +++ b/src/mcp/types.rs @@ -0,0 +1,163 @@ +//! # Model Context Protocol (MCP) JSON-RPC 2.0 Types +//! +//! Defines JSON-RPC 2.0 message envelopes, MCP protocol negotiation models, +//! tool schemas, and resource definitions per the Model Context Protocol specification. + +use serde::{Deserialize, Serialize}; +use serde_json::Value; + +/// JSON-RPC 2.0 Request envelope. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct JsonRpcRequest { + pub jsonrpc: String, + pub id: Option<Value>, + pub method: String, + #[serde(default)] + pub params: Option<Value>, +} + +/// JSON-RPC 2.0 Response envelope. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct JsonRpcResponse { + pub jsonrpc: String, + #[serde(skip_serializing_if = "Option::is_none")] + pub id: Option<Value>, + #[serde(skip_serializing_if = "Option::is_none")] + pub result: Option<Value>, + #[serde(skip_serializing_if = "Option::is_none")] + pub error: Option<JsonRpcError>, +} + +impl JsonRpcResponse { + /// Constructs a successful JSON-RPC response. + pub fn success(id: Option<Value>, result: Value) -> Self { + Self { + jsonrpc: "2.0".to_string(), + id, + result: Some(result), + error: None, + } + } + + /// Constructs an error JSON-RPC response. + pub fn error( + id: Option<Value>, + code: i64, + message: impl Into<String>, + data: Option<Value>, + ) -> Self { + Self { + jsonrpc: "2.0".to_string(), + id, + result: None, + error: Some(JsonRpcError { + code, + message: message.into(), + data, + }), + } + } +} + +/// JSON-RPC 2.0 Error payload. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct JsonRpcError { + pub code: i64, + pub message: String, + #[serde(skip_serializing_if = "Option::is_none")] + pub data: Option<Value>, +} + +// Standard JSON-RPC 2.0 error codes +pub const PARSE_ERROR: i64 = -32700; +pub const INVALID_REQUEST: i64 = -32600; +pub const METHOD_NOT_FOUND: i64 = -32601; +pub const INVALID_PARAMS: i64 = -32602; +pub const INTERNAL_ERROR: i64 = -32603; + +/// MCP Tool schema definition. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct ToolDefinition { + pub name: String, + pub description: String, + #[serde(rename = "inputSchema")] + pub input_schema: Value, +} + +/// MCP Text content returned from tool calls. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct TextContent { + #[serde(rename = "type")] + pub content_type: String, + pub text: String, +} + +impl TextContent { + /// Creates a new text content item. + pub fn new(text: impl Into<String>) -> Self { + Self { + content_type: "text".to_string(), + text: text.into(), + } + } +} + +/// Tool execution response envelope. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct CallToolResult { + pub content: Vec<TextContent>, + #[serde(rename = "isError", default)] + pub is_error: bool, +} + +impl CallToolResult { + /// Creates a success tool result with JSON text content. + pub fn success_json(value: &Value) -> Self { + Self { + content: vec![TextContent::new(value.to_string())], + is_error: false, + } + } + + /// Creates a success tool result with Markdown/plain text content. + pub fn success_text(text: impl Into<String>) -> Self { + Self { + content: vec![TextContent::new(text)], + is_error: false, + } + } + + /// Creates an error tool result. + pub fn error(message: impl Into<String>) -> Self { + Self { + content: vec![TextContent::new(message)], + is_error: true, + } + } +} + +/// MCP Resource metadata definition. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct ResourceDefinition { + pub uri: String, + pub name: String, + #[serde(skip_serializing_if = "Option::is_none")] + pub description: Option<String>, + #[serde(rename = "mimeType", skip_serializing_if = "Option::is_none")] + pub mime_type: Option<String>, +} + +/// Content payload of a read resource. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct ResourceContents { + pub uri: String, + #[serde(rename = "mimeType", skip_serializing_if = "Option::is_none")] + pub mime_type: Option<String>, + pub text: String, +} + +/// Result returned from `resources/read`. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct ReadResourceResult { + pub contents: Vec<ResourceContents>, +} diff --git a/src/report/csv.rs b/src/report/csv.rs new file mode 100644 index 0000000..6831505 --- /dev/null +++ b/src/report/csv.rs @@ -0,0 +1,338 @@ +//! # Screaming Frog Compatible CSV Report Exporter +//! +//! Generates four industry-standard CSV reports under `--format csv` for immediate +//! compatibility with SEO agency spreadsheet and data analysis workflows: +//! 1. `internal_all.csv`: Full internal crawled pages table with headings, metadata, and link counts. +//! 2. `issues_all.csv`: Defect triage log containing rule IDs, severities, affected URLs, and fix advice. +//! 3. `response_codes.csv`: HTTP status code distribution and redirect routing map. +//! 4. `external_all.csv`: Outbound hyperlink audit with anchor text, destination URLs, and nofollow directives. + +use crate::core::models::RobotsFlags; +use crate::crawler::engine::CrawlResult; +use crate::error::{SeoError, SeoResult}; +use crate::rules::catalog::get_rule; +use std::fs; +use std::path::{Path, PathBuf}; + +/// Maps standard HTTP status codes to conventional reason phrases. +pub fn http_status_reason(status_code: u16) -> &'static str { + match status_code { + 200 => "OK", + 201 => "Created", + 202 => "Accepted", + 204 => "No Content", + 206 => "Partial Content", + 301 => "Moved Permanently", + 302 => "Found", + 303 => "See Other", + 304 => "Not Modified", + 307 => "Temporary Redirect", + 308 => "Permanent Redirect", + 400 => "Bad Request", + 401 => "Unauthorized", + 403 => "Forbidden", + 404 => "Not Found", + 405 => "Method Not Allowed", + 410 => "Gone", + 429 => "Too Many Requests", + 500 => "Internal Server Error", + 502 => "Bad Gateway", + 503 => "Service Unavailable", + 504 => "Gateway Timeout", + _ => match status_code { + 200..=299 => "OK", + 300..=399 => "Redirect", + 400..=499 => "Client Error", + 500..=599 => "Server Error", + _ => "Unknown", + }, + } +} + +/// Evaluates the indexability status of a crawled page in compliance with Screaming Frog standards. +pub fn evaluate_indexability( + status_code: u16, + robots_flags: RobotsFlags, + url: &str, + canonical_url: Option<&str>, +) -> (&'static str, String) { + if !(200..=299).contains(&status_code) { + let reason = http_status_reason(status_code); + return ("Non-Indexable", format!("{reason} ({status_code})")); + } + + if robots_flags.contains(RobotsFlags::NOINDEX) { + return ("Non-Indexable", "noindex".to_string()); + } + + if let Some(canonical) = canonical_url { + let trimmed_canonical = canonical.trim(); + let trimmed_url = url.trim(); + if !trimmed_canonical.is_empty() && trimmed_canonical != trimmed_url { + return ("Non-Indexable", "Canonicalised".to_string()); + } + } + + ("Indexable", String::new()) +} + +/// Exports the complete suite of four Screaming Frog compatible CSV files into `<output_dir>/csv/`. +/// +/// Returns the list of generated file paths: +/// `[internal_all.csv, issues_all.csv, response_codes.csv, external_all.csv]`. +pub fn export_csv_suite(result: &CrawlResult, output_dir: &Path) -> SeoResult<Vec<PathBuf>> { + let csv_dir = output_dir.join("csv"); + fs::create_dir_all(&csv_dir) + .map_err(|e| SeoError::Internal(format!("Failed to create CSV export directory: {e}")))?; + + let internal_path = csv_dir.join("internal_all.csv"); + let issues_path = csv_dir.join("issues_all.csv"); + let response_codes_path = csv_dir.join("response_codes.csv"); + let external_path = csv_dir.join("external_all.csv"); + + export_internal_all(result, &internal_path)?; + export_issues_all(result, &issues_path)?; + export_response_codes(result, &response_codes_path)?; + export_external_all(result, &external_path)?; + + Ok(vec![ + internal_path, + issues_path, + response_codes_path, + external_path, + ]) +} + +/// Writes `internal_all.csv` containing all crawled pages and metadata. +fn export_internal_all(result: &CrawlResult, path: &Path) -> SeoResult<()> { + let mut writer = csv::Writer::from_path(path) + .map_err(|e| SeoError::Internal(format!("Failed to initialize internal_all.csv: {e}")))?; + + // 19 Screaming Frog standard columns + writer + .write_record([ + "Address", + "Status Code", + "Status", + "Content Type", + "Size (Bytes)", + "Word Count", + "Title 1", + "Title 1 Length", + "Meta Description 1", + "Meta Description 1 Length", + "H1-1", + "H1-1 Length", + "Canonical Link Element 1", + "Indexability", + "Indexability Status", + "Inlinks", + "Outlinks", + "Crawl Depth", + "Response Time (ms)", + ]) + .map_err(|e| { + SeoError::Internal(format!("Failed to write header to internal_all.csv: {e}")) + })?; + + for page in &result.pages { + let status_reason = http_status_reason(page.status_code); + let (indexability, indexability_status) = evaluate_indexability( + page.status_code, + page.robots_flags, + &page.url, + page.canonical_url.as_deref(), + ); + + let inlinks = result.graph.in_degree(&page.url); + let outlinks = result.graph.out_degree(&page.url); + let h1_length = page + .h1_primary + .as_ref() + .map(|h| h.chars().count()) + .unwrap_or(0); + + writer + .write_record([ + page.url.as_str(), + &page.status_code.to_string(), + status_reason, + page.content_type.as_str(), + &page.size_bytes.to_string(), + &page.word_count.to_string(), + page.title.as_deref().unwrap_or(""), + &page.title_length.to_string(), + page.meta_description.as_deref().unwrap_or(""), + &page.meta_desc_length.to_string(), + page.h1_primary.as_deref().unwrap_or(""), + &h1_length.to_string(), + page.canonical_url.as_deref().unwrap_or(""), + indexability, + &indexability_status, + &inlinks.to_string(), + &outlinks.to_string(), + &page.crawl_depth.to_string(), + &page.ttfb_ms.to_string(), + ]) + .map_err(|e| { + SeoError::Internal(format!("Failed to write row to internal_all.csv: {e}")) + })?; + } + + writer + .flush() + .map_err(|e| SeoError::Internal(format!("Failed to flush internal_all.csv: {e}")))?; + + Ok(()) +} + +/// Writes `issues_all.csv` summarizing all detected SEO findings. +fn export_issues_all(result: &CrawlResult, path: &Path) -> SeoResult<()> { + let mut writer = csv::Writer::from_path(path) + .map_err(|e| SeoError::Internal(format!("Failed to initialize issues_all.csv: {e}")))?; + + writer + .write_record([ + "Issue Code", + "Issue Name", + "Severity", + "Category", + "URL", + "Source URL", + "Details", + "Recommendation", + ]) + .map_err(|e| { + SeoError::Internal(format!("Failed to write header to issues_all.csv: {e}")) + })?; + + for issue in &result.issues { + let rule = get_rule(issue.code); + + writer + .write_record([ + issue.code.as_str(), + issue.title.as_str(), + issue.severity.as_str(), + issue.category.as_str(), + issue.target_url.as_str(), + issue.source_page_url.as_deref().unwrap_or(""), + issue.message.as_str(), + rule.fix_advice, + ]) + .map_err(|e| { + SeoError::Internal(format!("Failed to write row to issues_all.csv: {e}")) + })?; + } + + writer + .flush() + .map_err(|e| SeoError::Internal(format!("Failed to flush issues_all.csv: {e}")))?; + + Ok(()) +} + +/// Writes `response_codes.csv` listing status codes, redirect destinations, and inlinks. +fn export_response_codes(result: &CrawlResult, path: &Path) -> SeoResult<()> { + let mut writer = csv::Writer::from_path(path) + .map_err(|e| SeoError::Internal(format!("Failed to initialize response_codes.csv: {e}")))?; + + writer + .write_record([ + "URL", + "Status Code", + "Status", + "Redirect URL", + "Redirect Type", + "Inlinks Count", + ]) + .map_err(|e| { + SeoError::Internal(format!("Failed to write header to response_codes.csv: {e}")) + })?; + + for page in &result.pages { + let status_reason = http_status_reason(page.status_code); + let inlinks = result.graph.in_degree(&page.url); + + let (redirect_url, redirect_type) = match page.status_code { + 301 | 308 => ( + page.final_url + .as_deref() + .filter(|&u| u != page.url) + .unwrap_or(""), + "Permanent", + ), + 302 | 303 | 307 => ( + page.final_url + .as_deref() + .filter(|&u| u != page.url) + .unwrap_or(""), + "Temporary", + ), + _ => ("", ""), + }; + + writer + .write_record([ + page.url.as_str(), + &page.status_code.to_string(), + status_reason, + redirect_url, + redirect_type, + &inlinks.to_string(), + ]) + .map_err(|e| { + SeoError::Internal(format!("Failed to write row to response_codes.csv: {e}")) + })?; + } + + writer + .flush() + .map_err(|e| SeoError::Internal(format!("Failed to flush response_codes.csv: {e}")))?; + + Ok(()) +} + +/// Writes `external_all.csv` auditing all external hyperlinks across the site. +fn export_external_all(result: &CrawlResult, path: &Path) -> SeoResult<()> { + let mut writer = csv::Writer::from_path(path) + .map_err(|e| SeoError::Internal(format!("Failed to initialize external_all.csv: {e}")))?; + + writer + .write_record([ + "Source URL", + "Destination URL", + "Anchor Text", + "Status Code", + "Is Nofollow", + ]) + .map_err(|e| { + SeoError::Internal(format!("Failed to write header to external_all.csv: {e}")) + })?; + + for page in &result.pages { + for link in &page.links { + if !link.is_internal { + let status_str = link.status_code.map(|s| s.to_string()).unwrap_or_default(); + + writer + .write_record([ + link.source_url.as_str(), + link.target_url.as_str(), + link.anchor_text.as_str(), + &status_str, + if link.is_nofollow { "true" } else { "false" }, + ]) + .map_err(|e| { + SeoError::Internal(format!("Failed to write row to external_all.csv: {e}")) + })?; + } + } + } + + writer + .flush() + .map_err(|e| SeoError::Internal(format!("Failed to flush external_all.csv: {e}")))?; + + Ok(()) +} diff --git a/src/report/html.rs b/src/report/html.rs new file mode 100644 index 0000000..71ccb44 --- /dev/null +++ b/src/report/html.rs @@ -0,0 +1,1444 @@ +//! # Standalone Interactive Offline HTML Report Exporter +//! +//! Generates single-file self-contained HTML visual audit reports under `--format html`. +//! +//! Features: +//! - **100% Offline & Self-Contained**: All CSS and JavaScript are embedded directly in the HTML. +//! Zero external CDN dependencies, web fonts, or tracking scripts. +//! - **Interactive UI**: +//! - Live search filtering across pages and defect findings. +//! - Severity filter chips (Critical, Alert, Warning, Notice). +//! - Interactive tabbed navigation (Scorecard, Defect Triage, Pages Explorer, Authority Hubs). +//! - Expandable issue accordions with "[Copy Prompt]" remediation assistance. +//! - Crawl depth histogram and PageRank distribution. +//! - **Authentic Terminal Workstation / TUI Design**: +//! - Monospace typography, ASCII branding banner, TUI panels, and clean CLI flag chips. + +use crate::core::models::{PageReport, RobotsFlags, RuleId, Severity}; +use crate::crawler::engine::CrawlResult; +use crate::error::{SeoError, SeoResult}; +use crate::rules::catalog::get_rule; +use hashbrown::HashMap; +use std::fs; +use std::path::{Path, PathBuf}; + +/// Escapes HTML special characters to prevent document corruption or injection. +fn html_escape(s: &str) -> String { + s.replace('&', "&") + .replace('<', "<") + .replace('>', ">") + .replace('"', """) + .replace('\'', "'") +} + +/// Evaluates page indexability for display in the pages explorer table. +fn evaluate_page_indexability(page: &PageReport) -> (&'static str, &'static str) { + if !(200..=299).contains(&page.status_code) { + return ("Non-Indexable", "badge-red"); + } + if page.robots_flags.contains(RobotsFlags::NOINDEX) { + return ("Noindex", "badge-yellow"); + } + if let Some(canonical) = &page.canonical_url { + let trimmed_c = canonical.trim(); + let trimmed_u = page.url.trim(); + if !trimmed_c.is_empty() && trimmed_c != trimmed_u { + return ("Canonicalised", "badge-yellow"); + } + } + ("Indexable", "badge-green") +} + +/// Exports the crawl result into a standalone interactive offline HTML report file. +pub fn export_html_report(result: &CrawlResult, output_dir: &Path) -> SeoResult<PathBuf> { + fs::create_dir_all(output_dir) + .map_err(|e| SeoError::Internal(format!("Failed to create output directory: {e}")))?; + + let host = url::Url::parse(&result.target_url) + .map(|u| u.host_str().unwrap_or("site").replace('.', "_")) + .unwrap_or_else(|_| "site_audit".to_string()); + + let filename = format!("{host}_audit.html"); + let output_path = output_dir.join(filename); + + let html_content = render_html_report(result); + + fs::write(&output_path, html_content) + .map_err(|e| SeoError::Internal(format!("Failed to write HTML report: {e}")))?; + + Ok(output_path) +} + +/// Builds the complete HTML string containing styles, DOM, and interactive client scripts. +fn render_html_report(result: &CrawlResult) -> String { + let target_esc = html_escape(&result.target_url); + let session_id = result + .pages + .first() + .map(|p| p.crawl_id.as_str()) + .unwrap_or("session_audit"); + + // Dynamic ASCII track: 20 character block width + let filled_blocks = ((result.health_score as usize * 20) / 100).min(20); + let empty_blocks = 20 - filled_blocks; + let ascii_bar = format!( + "[{}{}] {}%", + "β–ˆ".repeat(filled_blocks), + "β–‘".repeat(empty_blocks), + result.health_score + ); + + // Score rating & color + let (score_color, score_badge, rating_text) = if result.health_score >= 90 { + ("col-green", "badge-green", "Excellent // Production Ready") + } else if result.health_score >= 75 { + ("col-yellow", "badge-yellow", "Good // Minor Defects") + } else if result.health_score >= 50 { + ("col-blue", "badge-cyan", "Needs Attention") + } else { + ("col-red", "badge-red", "Critical Defects Detected") + }; + + // Severity counts + let mut critical_count = 0; + let mut alert_count = 0; + let mut warning_count = 0; + let mut notice_count = 0; + for issue in &result.issues { + match issue.severity { + Severity::Critical => critical_count += 1, + Severity::Alert => alert_count += 1, + Severity::Warning => warning_count += 1, + Severity::Notice => notice_count += 1, + } + } + + // Status code breakdown + let mut status_counts: HashMap<u16, usize> = HashMap::new(); + for p in &result.pages { + *status_counts.entry(p.status_code).or_default() += 1; + } + let mut sorted_statuses: Vec<_> = status_counts.into_iter().collect(); + sorted_statuses.sort_by_key(|k| k.0); + + // Crawl depth histogram + let mut depth_counts: HashMap<u16, usize> = HashMap::new(); + for p in &result.pages { + *depth_counts.entry(p.crawl_depth).or_default() += 1; + } + let mut sorted_depths: Vec<_> = depth_counts.into_iter().collect(); + sorted_depths.sort_by_key(|k| k.0); + + // Average TTFB + let avg_ttfb = if !result.pages.is_empty() { + let sum: u64 = result.pages.iter().map(|p| p.ttfb_ms as u64).sum(); + sum / (result.pages.len() as u64) + } else { + 0 + }; + + // Render Status Code Rows + let mut status_rows = String::new(); + for (status, count) in &sorted_statuses { + let pct = if !result.pages.is_empty() { + (*count as f64 / result.pages.len() as f64) * 100.0 + } else { + 0.0 + }; + let badge_class = match status { + 200..=299 => "badge-green", + 300..=399 => "badge-cyan", + 400..=499 => "badge-red", + _ => "badge-red", + }; + status_rows.push_str(&format!( + r#"<tr> + <td><span class="badge {badge_class}">{status}</span></td> + <td>{count} pages</td> + <td>{pct:.1}%</td> + </tr>"# + )); + } + + // Group issues by RuleId + let mut grouped_issues: HashMap<RuleId, Vec<&crate::core::models::IssueFinding>> = + HashMap::new(); + for issue in &result.issues { + grouped_issues.entry(issue.code).or_default().push(issue); + } + let mut sorted_issue_groups: Vec<_> = grouped_issues.into_iter().collect(); + sorted_issue_groups.sort_by_key(|(_, list)| { + match list.first().map(|i| i.severity).unwrap_or(Severity::Notice) { + Severity::Critical => 0, + Severity::Alert => 1, + Severity::Warning => 2, + Severity::Notice => 3, + } + }); + + let mut issues_accordion_html = String::new(); + for (idx, (rule_id, findings)) in sorted_issue_groups.iter().enumerate() { + let first = findings.first(); + let sev = first.map(|i| i.severity).unwrap_or(Severity::Notice); + let sev_str = sev.as_str(); + let sev_display = match sev { + Severity::Critical => "Critical", + Severity::Alert => "Alert", + Severity::Warning => "Warning", + Severity::Notice => "Notice", + }; + let rule_info = get_rule(*rule_id); + let title_esc = html_escape(first.map(|i| i.title.as_str()).unwrap_or(rule_info.title)); + let category_esc = html_escape(rule_info.category.as_str()); + let advice_esc = html_escape(rule_info.fix_advice); + let count = findings.len(); + + let sev_badge = match sev { + Severity::Critical => "badge-red", + Severity::Alert => "badge-yellow", + Severity::Warning => "badge-yellow", + Severity::Notice => "badge-cyan", + }; + + let mut samples_html = String::new(); + for f in findings.iter().take(20) { + let target = html_escape(&f.target_url); + let msg = html_escape(&f.message); + samples_html.push_str(&format!( + r#"<li class="sample-item"> + <span class="sample-url">{target}</span> + <span class="sample-msg">{msg}</span> + </li>"# + )); + } + if count > 20 { + samples_html.push_str(&format!( + r#"<li class="sample-more">... and {} more affected targets</li>"#, + count - 20 + )); + } + + issues_accordion_html.push_str(&format!( + r#"<div class="issue-card" data-severity="{sev_str}" id="issue-{idx}"> + <div class="issue-header" onclick="toggleIssue({idx})"> + <div class="issue-meta"> + <span class="fold-caret">β–Ά</span> + <span class="badge {sev_badge}">{sev_display}</span> + <span class="issue-code">{}</span> + <span class="badge badge-dim">{category_esc}</span> + <span class="issue-title">{title_esc}</span> + </div> + <div class="issue-count-pill">{count} target{}</div> + </div> + <div class="issue-body" id="issue-body-{idx}"> + <div class="remediation-box"> + <div class="remediation-header"> + <span class="remediation-title">// Remediation Instruction</span> + <button class="copy-btn" onclick="copyRemedy(event, {idx})">[Copy Prompt]</button> + </div> + <div class="remediation-content" id="remedy-{idx}">{advice_esc}</div> + </div> + <div class="affected-pages-title">Affected Target URLs:</div> + <ul class="samples-list">{samples_html}</ul> + </div> + </div>"#, + rule_id.as_str(), + if count == 1 { "" } else { "s" } + )); + } + + // Render Pages Table Rows + let mut pages_table_rows = String::new(); + for (i, p) in result.pages.iter().enumerate() { + let url_esc = html_escape(&p.url); + let title_esc = html_escape(p.title.as_deref().unwrap_or("-")); + let h1_esc = html_escape(p.h1_primary.as_deref().unwrap_or("-")); + let (indexability, ind_badge) = evaluate_page_indexability(p); + let inlinks = result.graph.in_degree(&p.url); + let outlinks = result.graph.out_degree(&p.url); + + let status_badge = match p.status_code { + 200..=299 => "badge-green", + 300..=399 => "badge-cyan", + 400..=499 => "badge-red", + _ => "badge-red", + }; + + pages_table_rows.push_str(&format!( + r#"<tr class="page-row" data-url="{url_esc}" data-title="{title_esc}" data-status="{}"> + <td class="col-num">{}</td> + <td class="col-url"><a href="{url_esc}" target="_blank" rel="noopener noreferrer">{url_esc}</a></td> + <td><span class="badge {status_badge}">{}</span></td> + <td><span class="badge {ind_badge}">{indexability}</span></td> + <td class="col-text">{title_esc}</td> + <td class="col-text">{h1_esc}</td> + <td>{inlinks}</td> + <td>{outlinks}</td> + <td>{}ms</td> + <td>{}</td> + </tr>"#, + p.status_code, + i + 1, + p.status_code, + p.ttfb_ms, + p.word_count + )); + } + + // Render Authority Hubs (Top PageRank) + let mut ranked_pages: Vec<_> = result + .pages + .iter() + .map(|p| { + let pr = result.pagerank.get(&p.url_hash).copied().unwrap_or(0.0); + (p, pr) + }) + .collect(); + ranked_pages.sort_by(|a, b| b.1.partial_cmp(&a.1).unwrap_or(std::cmp::Ordering::Equal)); + + let mut hubs_rows = String::new(); + for (rank, (p, pr)) in ranked_pages.iter().take(15).enumerate() { + let url_esc = html_escape(&p.url); + let inlinks = result.graph.in_degree(&p.url); + let outlinks = result.graph.out_degree(&p.url); + let pct = pr * 100.0; + + hubs_rows.push_str(&format!( + r#"<tr> + <td class="col-num">#{}</td> + <td><span class="badge badge-green">{pct:.2}%</span></td> + <td>{inlinks} in / {outlinks} out</td> + <td class="col-url"><a href="{url_esc}" target="_blank" rel="noopener noreferrer">{url_esc}</a></td> + </tr>"#, + rank + 1 + )); + } + + // Render Depth Histogram + let max_depth_count = sorted_depths.iter().map(|(_, c)| *c).max().unwrap_or(1); + let mut depth_bars_html = String::new(); + for (depth, count) in &sorted_depths { + let width_pct = (*count as f64 / max_depth_count as f64) * 100.0; + depth_bars_html.push_str(&format!( + r#"<div class="depth-bar-row"> + <div class="depth-label">Depth {depth}:</div> + <div class="depth-track"> + <div class="depth-fill" style="width: {width_pct:.1}%;"></div> + </div> + <div class="depth-count">{count} pages</div> + </div>"# + )); + } + + // Assemble complete self-contained HTML + format!( + r#"<!DOCTYPE html> +<html lang="en"> +<head> + <meta charset="utf-8"> + <meta name="viewport" content="width=device-width, initial-scale=1.0"> + <title>SEO Lens Audit Report β€” {target_esc} + + + +
+ +
+
+ +
v0.1.0 β”‚ High-Performance Website Crawler & AI-Native Technical SEO Engine
+
+
+
+ $ Target: + {target_esc} +
+
+ session{session_id} + probed{} targets + findings{} + duration{:.1}s +
+
+
+ +
+ +
+
+
+ // Health Score + Rating: {}/100 +
+
+
+
+ {} + /100 +
+
{ascii_bar}
+
+ Verdict: {rating_text} +
+
+
+
+ +
+
+ // Crawl Telemetry + Session: {session_id} +
+
+
+
+ Targets Probed: + {} pages +
+
+ Internal Links: + {} edges +
+
+ Average TTFB: + {avg_ttfb}ms +
+
+ Crawl Duration: + {:.1}s +
+
+ Critical Defects: + {critical_count} +
+
+ Alerts & Warnings: + {} +
+
+
+
+
+ + + + + +
+
+
+ Severity: + + + + + +
+
+ grep: + +
+
+
+ {issues_accordion_html} +
+
+ + +
+
+
+ grep: + +
+
Showing {} / {} targets
+
+
+ + + + + + + + + + + + + + + + + {pages_table_rows} + +
#Page URLStatusIndexabilityPage TitlePrimary H1InlinksOutlinksTTFBWords
+
+
+ + +
+
+
+ // HTTP Status Code Distribution + {} codes +
+
+
+ + + + + + + + + + {status_rows} + +
StatusTargetsRatio
+
+
+
+
+ + +
+
+
+
+ // Top Authority Hubs (PageRank) + Top 15 Nodes +
+
+
+ + + + + + + + + + + {hubs_rows} + +
RankEquityIn / OutAuthority Target URL
+
+
+
+
+
+ // Crawl Depth Hierarchy + Max Depth: {} +
+
+
+ {depth_bars_html} +
+
+
+
+
+
+
+ + + +"#, + result.pages.len(), + result.issues.len(), + result.duration.as_secs_f64(), + result.health_score, + result.health_score, + result.pages.len(), + result.graph.edge_count(), + result.duration.as_secs_f64(), + alert_count + warning_count, + result.issues.len(), + result.pages.len(), + result.issues.len(), + result.pages.len(), + result.pages.len(), + sorted_statuses.len(), + sorted_depths.last().map(|(d, _)| *d).unwrap_or(0) + ) +} diff --git a/src/report/mod.rs b/src/report/mod.rs index b1e2e58..55d0271 100644 --- a/src/report/mod.rs +++ b/src/report/mod.rs @@ -6,17 +6,22 @@ //! - [`json`]: Complete structured JSON export for automated pipelines and data warehouses. //! - [`score`]: Normalized 0–100 technical SEO Health Score calculation. +pub mod csv; +pub mod html; pub mod inspector; pub mod json; pub mod markdown; pub mod score; pub mod terminal; +pub use csv::export_csv_suite; +pub use html::export_html_report; pub use inspector::{format_page_inspection, print_page_inspection}; pub use json::export_json_report; pub use markdown::export_markdown_report; pub use score::calculate_health_score; pub use terminal::{ - create_crawl_progress_bar, finish_crawl_progress, print_audit_banner, - print_executive_scorecard, print_historical_sessions, update_crawl_progress, + create_crawl_progress_bar, finish_crawl_progress, print_ai_readiness_scorecard, + print_audit_banner, print_cli_help, print_executive_scorecard, print_historical_sessions, + print_issues_matrix, print_schema_outcome, update_crawl_progress, }; diff --git a/src/report/terminal.rs b/src/report/terminal.rs index 239b2b4..3d48b07 100644 --- a/src/report/terminal.rs +++ b/src/report/terminal.rs @@ -6,8 +6,10 @@ //! - Deep audit scorecard with visual health gauges, protocol radar, defect triage trees, //! PageRank authority distribution tables, and artifact links. -use crate::core::models::{CrawlSummary, Severity}; +use crate::core::models::{CrawlSummary, IssueFinding, Severity}; +use crate::crawler::ai_check::{AiReadinessReport, AiSearchRisk}; use crate::crawler::engine::{CrawlResult, ProgressUpdate}; +use crate::rules::page::schema_val::SchemaValidationOutcome; use hashbrown::HashMap; use std::io::{self, Write}; use std::path::Path; @@ -25,8 +27,81 @@ const ANSI_BRIGHT_WHITE: &str = "\x1b[38;5;231m"; const ANSI_BOLD: &str = "\x1b[1m"; const ANSI_RESET: &str = "\x1b[0m"; +fn format_section_header(title: &str) -> String { + let pad = title.chars().count() + 4; + let top = format!( + " {ANSI_BOLD}{ANSI_CYAN}β”Œ{}┐{ANSI_RESET}\n", + "─".repeat(pad) + ); + let mid = format!( + " {ANSI_BOLD}{ANSI_CYAN}β”‚{ANSI_RESET} {ANSI_BOLD}{ANSI_BRIGHT_WHITE}{title}{ANSI_RESET} {ANSI_BOLD}{ANSI_CYAN}β”‚{ANSI_RESET}\n" + ); + let bot = format!( + " {ANSI_BOLD}{ANSI_CYAN}β””{}β”˜{ANSI_RESET}\n", + "─".repeat(pad) + ); + format!("{top}{mid}{bot}") +} + +/// Splits `text` into lines where each line does not exceed `max_width`. +fn wrap_text(text: &str, max_width: usize) -> Vec { + let mut lines = Vec::new(); + let mut current_line = String::new(); + + for word in text.split_whitespace() { + if current_line.is_empty() { + current_line.push_str(word); + } else if current_line.len() + 1 + word.len() <= max_width { + current_line.push(' '); + current_line.push_str(word); + } else { + lines.push(current_line); + current_line = word.to_string(); + } + } + + if !current_line.is_empty() { + lines.push(current_line); + } + + if lines.is_empty() { + lines.push(String::new()); + } + + lines +} + +fn status_text(code: u16) -> &'static str { + match code { + 200 => "OK", + 201 => "Created", + 202 => "Accepted", + 204 => "No Content", + 301 => "Moved Permanently", + 302 => "Found", + 304 => "Not Modified", + 307 => "Temporary Redirect", + 308 => "Permanent Redirect", + 400 => "Bad Request", + 401 => "Unauthorized", + 403 => "Forbidden", + 404 => "Not Found", + 405 => "Method Not Allowed", + 410 => "Gone", + 429 => "Too Many Requests", + 500 => "Internal Server Error", + 502 => "Bad Gateway", + 503 => "Service Unavailable", + 504 => "Gateway Timeout", + _ => "Unknown", + } +} + +static BANNER_PRINTED: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::new(false); + /// Prints the cyberpunk startup ASCII banner and mission parameters. pub fn print_audit_banner(target_url: &str, max_pages: u32, concurrency: usize, aimd: bool) { + BANNER_PRINTED.store(true, Ordering::SeqCst); let aimd_status = if aimd { format!("{ANSI_GREEN}ACTIVE (AIMD){ANSI_RESET}") } else { @@ -39,8 +114,11 @@ pub fn print_audit_banner(target_url: &str, max_pages: u32, concurrency: usize, }; println!( - "\n{ANSI_CYAN}{ANSI_BOLD} β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ•— β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ•—β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—\n β–ˆβ–ˆβ•”β•β•β•β•β•β–ˆβ–ˆβ•”β•β•β•β•β•β–ˆβ–ˆβ•”β•β•β•β–ˆβ–ˆβ•— β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•”β•β•β•β•β•β–ˆβ–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ•‘β–ˆβ–ˆβ•”β•β•β•β•β•\n β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ•”β–ˆβ–ˆβ•— β–ˆβ–ˆβ•‘β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—\n β•šβ•β•β•β•β–ˆβ–ˆβ•‘β–ˆβ–ˆβ•”β•β•β• β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•”β•β•β• β–ˆβ–ˆβ•‘β•šβ–ˆβ–ˆβ•—β–ˆβ–ˆβ•‘β•šβ•β•β•β•β–ˆβ–ˆβ•‘\n β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•‘β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β•šβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•”β• β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ•‘ β•šβ–ˆβ–ˆβ–ˆβ–ˆβ•‘β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•‘\n β•šβ•β•β•β•β•β•β•β•šβ•β•β•β•β•β•β• β•šβ•β•β•β•β•β• β•šβ•β•β•β•β•β•β•β•šβ•β•β•β•β•β•β•β•šβ•β• β•šβ•β•β•β•β•šβ•β•β•β•β•β•β•{ANSI_RESET}\n\n{ANSI_DIM}β”Œβ”€β”€[{ANSI_RESET} {ANSI_BOLD}TARGET TELEMETRY{ANSI_RESET} {ANSI_DIM}]────────────────────────────────────────────────────────{ANSI_RESET}\n{ANSI_DIM}β”‚{ANSI_RESET} {ANSI_BOLD}Target URL {ANSI_RESET} : {ANSI_CYAN}{target_url}{ANSI_RESET}\n{ANSI_DIM}β”‚{ANSI_RESET} {ANSI_BOLD}Parameters {ANSI_RESET} : {pages_limit} {ANSI_DIM}β”‚{ANSI_RESET} {concurrency} workers {ANSI_DIM}β”‚{ANSI_RESET} AIMD: {aimd_status}\n{ANSI_DIM}β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜{ANSI_RESET}\n" + "\n{ANSI_CYAN}{ANSI_BOLD} β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ•— β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ•—β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—\n β–ˆβ–ˆβ•”β•β•β•β•β•β–ˆβ–ˆβ•”β•β•β•β•β•β–ˆβ–ˆβ•”β•β•β•β–ˆβ–ˆβ•— β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•”β•β•β•β•β•β–ˆβ–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ•‘β–ˆβ–ˆβ•”β•β•β•β•β•\n β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ•”β–ˆβ–ˆβ•— β–ˆβ–ˆβ•‘β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—\n β•šβ•β•β•β•β–ˆβ–ˆβ•‘β–ˆβ–ˆβ•”β•β•β• β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•”β•β•β• β–ˆβ–ˆβ•‘β•šβ–ˆβ–ˆβ•—β–ˆβ–ˆβ•‘β•šβ•β•β•β•β–ˆβ–ˆβ•‘\n β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•‘β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β•šβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•”β• β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ•‘ β•šβ–ˆβ–ˆβ–ˆβ–ˆβ•‘β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•‘\n β•šβ•β•β•β•β•β•β•β•šβ•β•β•β•β•β•β• β•šβ•β•β•β•β•β• β•šβ•β•β•β•β•β•β•β•šβ•β•β•β•β•β•β•β•šβ•β• β•šβ•β•β•β•β•šβ•β•β•β•β•β•β•{ANSI_RESET}\n" ); + print!("{}", format_section_header("SEO LENS // DEEP AUDIT MATRIX")); + println!(" {ANSI_BOLD}Target URL {ANSI_RESET} : {ANSI_CYAN}{target_url}{ANSI_RESET}\n"); + println!(" {ANSI_BOLD}Parameters {ANSI_RESET} : {pages_limit} {ANSI_DIM}β”‚{ANSI_RESET} {concurrency} workers {ANSI_DIM}β”‚{ANSI_RESET} AIMD: {aimd_status}\n"); } /// Interactive single-line progress indicator for live crawl monitoring. @@ -148,17 +226,17 @@ pub fn finish_crawl_progress(_pb: &CrawlProgressBar) { /// Renders the comprehensive post-crawl executive scorecard in the terminal. pub fn print_executive_scorecard(result: &CrawlResult, exported_paths: &[(&str, &Path)]) { - println!( - "\n{ANSI_CYAN}╔══════════════════════════════════════════════════════════════════════════╗" - ); - println!("β•‘ SEO LENS // DEEP AUDIT MATRIX β•‘"); - println!( - "β•šβ•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•{ANSI_RESET}" - ); - println!( - " {ANSI_BOLD}Target{ANSI_RESET} : {ANSI_CYAN}{}{ANSI_RESET}", - result.target_url - ); + let already_bannered = BANNER_PRINTED.swap(false, Ordering::SeqCst); + if !already_bannered { + print!( + "\n{}", + format_section_header("SEO LENS // DEEP AUDIT MATRIX") + ); + println!( + " {ANSI_BOLD}Target URL {ANSI_RESET} : {ANSI_CYAN}{}{ANSI_RESET}\n", + result.target_url + ); + } // Health Score Visual Gauge let filled_bars = (result.health_score as usize * 20) / 100; @@ -177,7 +255,7 @@ pub fn print_executive_scorecard(result: &CrawlResult, exported_paths: &[(&str, let gauge = format!("{score_color}[{filled_str}{ANSI_DIM}{empty_str}{score_color}]{ANSI_RESET}"); println!( - " {ANSI_BOLD}Health Score{ANSI_RESET} : {gauge} {score_color}{ANSI_BOLD}{}/100 [{rating}]{ANSI_RESET}", + " {ANSI_BOLD}Health Score{ANSI_RESET} : {gauge} {score_color}{ANSI_BOLD}{}/100 [{rating}]{ANSI_RESET}\n", result.health_score ); @@ -201,7 +279,7 @@ pub fn print_executive_scorecard(result: &CrawlResult, exported_paths: &[(&str, }; println!( - " {ANSI_BOLD}Telemetry{ANSI_RESET} : {ANSI_GREEN}{:.1}s{ANSI_RESET} elapsed {ANSI_DIM}β”‚{ANSI_RESET} {ANSI_GREEN}{}{ANSI_RESET} pages probed {ANSI_DIM}β”‚{ANSI_RESET} {ANSI_GREEN}{}{ANSI_RESET} links indexed {ANSI_DIM}β”‚{ANSI_RESET} TTFB: {ANSI_GREEN}{avg_ttfb}{ANSI_RESET}\n", + " {ANSI_BOLD}Telemetry {ANSI_RESET} : {ANSI_GREEN}{:.1}s{ANSI_RESET} elapsed {ANSI_DIM}β”‚{ANSI_RESET} {ANSI_GREEN}{}{ANSI_RESET} pages probed {ANSI_DIM}β”‚{ANSI_RESET} {ANSI_GREEN}{}{ANSI_RESET} links indexed {ANSI_DIM}β”‚{ANSI_RESET} TTFB: {ANSI_GREEN}{avg_ttfb}{ANSI_RESET}\n", result.duration.as_secs_f64(), result.pages.len(), result.graph.edge_count() @@ -213,7 +291,7 @@ pub fn print_executive_scorecard(result: &CrawlResult, exported_paths: &[(&str, *status_counts.entry(page.status_code).or_default() += 1; } - println!("{ANSI_DIM}β”Œβ”€β”€[{ANSI_RESET} {ANSI_BOLD}PROTOCOL TELEMETRY{ANSI_RESET} {ANSI_DIM}]───────────────────────────────────────────────────{ANSI_RESET}"); + print!("{}", format_section_header("PROTOCOL TELEMETRY")); let mut sorted_statuses: Vec<_> = status_counts.into_iter().collect(); sorted_statuses.sort_by_key(|k| k.0); @@ -229,17 +307,19 @@ pub fn print_executive_scorecard(result: &CrawlResult, exported_paths: &[(&str, 400..=499 => ("βœ–", ANSI_RED), _ => ("βœ–", ANSI_RED), }; + let reason = status_text(status); + let status_desc = format!("{status} {reason}"); println!( - "{ANSI_DIM}β”‚{ANSI_RESET} {color}{icon} {:<4} OK{ANSI_RESET} : {:>5} pages {ANSI_DIM}({:.1}%){ANSI_RESET}", - status, count, pct + " {color}{icon} {:<18}{ANSI_RESET} : {:>5} pages {ANSI_DIM}({:.1}%){ANSI_RESET}", + status_desc, count, pct ); } - println!("{ANSI_DIM}β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜{ANSI_RESET}\n"); + println!(); // Top Priority Issues (Tree Format) - println!("{ANSI_DIM}β”Œβ”€β”€[{ANSI_RESET} {ANSI_BOLD}DEFECT TRIAGE MATRIX{ANSI_RESET} {ANSI_DIM}]─────────────────────────────────────────────────{ANSI_RESET}"); + print!("{}", format_section_header("DEFECT TRIAGE MATRIX")); if result.issues.is_empty() { - println!("{ANSI_DIM}β”‚{ANSI_RESET} {ANSI_GREEN}βœ” Zero technical SEO defects detected across all probed nodes.{ANSI_RESET}"); + println!(" {ANSI_GREEN}βœ” Zero technical SEO defects detected across all probed nodes.{ANSI_RESET}\n"); } else { let mut grouped: HashMap< crate::core::models::RuleId, @@ -283,10 +363,41 @@ pub fn print_executive_scorecard(result: &CrawlResult, exported_paths: &[(&str, findings.len(), if findings.len() == 1 { "" } else { "s" } ); - println!(" {ANSI_DIM}β”œβ”€β”€ Defect :{ANSI_RESET} {}", title); - if !remediation.is_empty() { - println!(" {ANSI_DIM}β”œβ”€β”€ Remedy :{ANSI_RESET} {}", remediation); + + let has_sample = !findings.is_empty(); + let has_remedy = !remediation.is_empty(); + + // Defect line (wrapped at 58 chars to keep total line <= 76 chars) + let defect_wrapped = wrap_text(title, 58); + let defect_branch = if has_remedy || has_sample { + "β”œβ”€β”€" + } else { + "└──" + }; + let defect_cont = if has_remedy || has_sample { "β”‚" } else { " " }; + println!( + " {ANSI_DIM}{defect_branch} Defect :{ANSI_RESET} {}", + defect_wrapped[0] + ); + for cont in &defect_wrapped[1..] { + println!(" {ANSI_DIM}{defect_cont} {ANSI_RESET}{cont}"); } + + // Remedy line (wrapped at 58 chars) + if has_remedy { + let remedy_wrapped = wrap_text(remediation, 58); + let remedy_branch = if has_sample { "β”œβ”€β”€" } else { "└──" }; + let remedy_cont = if has_sample { "β”‚" } else { " " }; + println!( + " {ANSI_DIM}{remedy_branch} Remedy :{ANSI_RESET} {}", + remedy_wrapped[0] + ); + for cont in &remedy_wrapped[1..] { + println!(" {ANSI_DIM}{remedy_cont} {ANSI_RESET}{cont}"); + } + } + + // Sample line if let Some(sample) = findings.first() { println!( " {ANSI_DIM}└── Sample :{ANSI_RESET} {ANSI_DIM}{}{ANSI_RESET}", @@ -298,7 +409,10 @@ pub fn print_executive_scorecard(result: &CrawlResult, exported_paths: &[(&str, } // Top Authority Hubs // PageRank Distribution Matrix - println!("{ANSI_DIM}β”Œβ”€β”€[{ANSI_RESET} {ANSI_BOLD}TOP AUTHORITY HUBS // PAGERANK DISTRIBUTION{ANSI_RESET} {ANSI_DIM}]─────────────────────────{ANSI_RESET}"); + print!( + "{}", + format_section_header("TOP AUTHORITY HUBS // PAGERANK DISTRIBUTION") + ); println!(" {ANSI_DIM}Rank Equity Inlinks / Outlinks Authority Node{ANSI_RESET}"); println!(" {ANSI_DIM}───── ────── ────────────────── ──────────────{ANSI_RESET}"); @@ -327,34 +441,14 @@ pub fn print_executive_scorecard(result: &CrawlResult, exported_paths: &[(&str, // Exported Mission Artifacts if !exported_paths.is_empty() { - println!("{ANSI_DIM}β”Œβ”€β”€[{ANSI_RESET} {ANSI_BOLD}GENERATED MISSION ARTIFACTS{ANSI_RESET} {ANSI_DIM}]─────────────────────────────────────────{ANSI_RESET}"); + print!("{}", format_section_header("GENERATED MISSION ARTIFACTS")); for (fmt, path) in exported_paths { - println!( - "{ANSI_DIM}β”‚{ANSI_RESET} {ANSI_CYAN}β—ˆ {:<9}{ANSI_RESET} : {}", - fmt, - path.display() - ); + println!(" {ANSI_CYAN}β—ˆ {:<9}{ANSI_RESET} : {}", fmt, path.display()); } - println!("{ANSI_DIM}β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜{ANSI_RESET}\n"); + println!(); } } -fn format_section_header(title: &str) -> String { - let pad = title.chars().count() + 4; - let top = format!( - " {ANSI_BOLD}{ANSI_CYAN}β”Œ{}┐{ANSI_RESET}\n", - "─".repeat(pad) - ); - let mid = format!( - " {ANSI_BOLD}{ANSI_CYAN}β”‚{ANSI_RESET} {ANSI_BOLD}{ANSI_BRIGHT_WHITE}{title}{ANSI_RESET} {ANSI_BOLD}{ANSI_CYAN}β”‚{ANSI_RESET}\n" - ); - let bot = format!( - " {ANSI_BOLD}{ANSI_CYAN}β””{}β”˜{ANSI_RESET}\n", - "─".repeat(pad) - ); - format!("{top}{mid}{bot}") -} - /// Renders the cyberpunk-styled historical crawl sessions in developer inspector aesthetic. pub fn print_historical_sessions(db_path: &Path, crawls: &[CrawlSummary]) { // 1. Big Cyberpunk ASCII Header (matching inspect command) @@ -460,3 +554,286 @@ pub fn print_historical_sessions(db_path: &Path, crawls: &[CrawlSummary]) { println!(" {ANSI_CYAN}β—ˆ Re-export artifacts{ANSI_RESET} : seolens report --format md,json"); println!(" {ANSI_CYAN}β—ˆ Start fresh crawl {ANSI_RESET} : seolens audit \n"); } + +/// Formats and prints a filtered issues matrix in the terminal. +pub fn print_issues_matrix( + session_id: &str, + issues: &[IssueFinding], + total_count: usize, + offset: usize, + limit: usize, +) { + println!( + "\n{ANSI_CYAN}{ANSI_BOLD} β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ•—β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ•— β–ˆβ–ˆβ•—β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—\n β–ˆβ–ˆβ•”β•β•β•β•β•β–ˆβ–ˆβ•”β•β•β•β•β•β–ˆβ–ˆβ•”β•β•β•β–ˆβ–ˆβ•— β–ˆβ–ˆβ•‘β–ˆβ–ˆβ•”β•β•β•β•β•β–ˆβ–ˆβ•”β•β•β•β•β•β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•‘β–ˆβ–ˆβ•”β•β•β•β•β•β–ˆβ–ˆβ•”β•β•β•β•β•\n β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•‘β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•‘β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—\n β•šβ•β•β•β•β–ˆβ–ˆβ•‘β–ˆβ–ˆβ•”β•β•β• β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•‘β•šβ•β•β•β•β–ˆβ–ˆβ•‘β•šβ•β•β•β•β–ˆβ–ˆβ•‘β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•‘β–ˆβ–ˆβ•”β•β•β• β•šβ•β•β•β•β–ˆβ–ˆβ•‘\n β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•‘β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β•šβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•”β• β–ˆβ–ˆβ•‘β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•‘β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•‘β•šβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•”β•β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•‘\n β•šβ•β•β•β•β•β•β•β•šβ•β•β•β•β•β•β• β•šβ•β•β•β•β•β• β•šβ•β•β•šβ•β•β•β•β•β•β•β•šβ•β•β•β•β•β•β• β•šβ•β•β•β•β•β• β•šβ•β•β•β•β•β•β•β•šβ•β•β•β•β•β•β•{ANSI_RESET}\n" + ); + + print!( + "{}", + format_section_header("AUDIT DEFECTS // FILTERED QUERY") + ); + println!(" {ANSI_BOLD}Session ID{ANSI_RESET} : {ANSI_CYAN}{session_id}{ANSI_RESET}"); + println!( + " {ANSI_BOLD}Matching {ANSI_RESET} : {total_count} issues found (showing {offset}..{})\n", + (offset + issues.len()).min(total_count) + ); + + if issues.is_empty() { + println!(" {ANSI_GREEN}βœ” No issues matched your filter criteria.{ANSI_RESET}\n"); + return; + } + + for (i, issue) in issues.iter().enumerate() { + let (badge, color) = match issue.severity { + Severity::Critical => ( + format!("{ANSI_RED}{ANSI_BOLD}[🚨 CRITICAL]{ANSI_RESET}"), + ANSI_RED, + ), + Severity::Alert => ( + format!("{ANSI_YELLOW}{ANSI_BOLD}[⚠️ ALERT]{ANSI_RESET}"), + ANSI_YELLOW, + ), + Severity::Warning => ( + format!("{ANSI_YELLOW}[⚑ WARNING]{ANSI_RESET}"), + ANSI_YELLOW, + ), + Severity::Notice => (format!("{ANSI_CYAN}[β„Ή NOTICE]{ANSI_RESET}"), ANSI_CYAN), + }; + + println!( + " {ANSI_BOLD}#{:03}{ANSI_RESET} {badge} {color}{ANSI_BOLD}{}{ANSI_RESET}", + offset + i + 1, + issue.code.as_str() + ); + println!( + " {ANSI_BOLD}Target URL {ANSI_RESET}: {ANSI_CYAN}{}{ANSI_RESET}", + issue.target_url + ); + let diag_wrapped = wrap_text(&issue.message, 58); + println!( + " {ANSI_BOLD}Diagnosis {ANSI_RESET}: {}", + diag_wrapped[0] + ); + for cont in &diag_wrapped[1..] { + println!(" {cont}"); + } + if let Some(ref src) = issue.source_page_url { + println!(" {ANSI_BOLD}Source Page{ANSI_RESET}: {ANSI_DIM}{src}{ANSI_RESET}"); + } + println!(); + } + + if total_count > offset + issues.len() { + let next_offset = offset + limit; + println!( + " {ANSI_DIM}β—ˆ To view more: seolens issues {session_id} --offset {next_offset} --limit {limit}{ANSI_RESET}\n" + ); + } +} + +/// Prints a cyberpunk AI search & GEO readiness assessment scorecard. +pub fn print_ai_readiness_scorecard(report: &AiReadinessReport) { + println!( + "\n{ANSI_CYAN}{ANSI_BOLD} β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ•— β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•— \n β–ˆβ–ˆβ•”β•β•β•β•β•β–ˆβ–ˆβ•”β•β•β•β•β•β–ˆβ–ˆβ•”β•β•β•β–ˆβ–ˆβ•— β–ˆβ–ˆβ•”β•β•β–ˆβ–ˆβ•—β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•”β•β•β•β•β•β–ˆβ–ˆβ•”β•β•β•β•β•β–ˆβ–ˆβ•”β•β•β•β–ˆβ–ˆβ•—\n β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•‘β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•‘\n β•šβ•β•β•β•β–ˆβ–ˆβ•‘β–ˆβ–ˆβ•”β•β•β• β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•”β•β•β–ˆβ–ˆβ•‘β–ˆβ–ˆβ•‘ β•šβ•β•β•β•β–ˆβ–ˆβ•‘β–ˆβ–ˆβ•”β•β•β• β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•‘\n β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•‘β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β•šβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•”β• β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•‘β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•‘β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β•šβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•”β•\n β•šβ•β•β•β•β•β•β•β•šβ•β•β•β•β•β•β• β•šβ•β•β•β•β•β• β•šβ•β• β•šβ•β•β•šβ•β• β•šβ•β•β•β•β•β•β•β•šβ•β•β•β•β•β•β• β•šβ•β•β•β•β•β• {ANSI_RESET}\n" + ); + + let (risk_badge, risk_desc) = match report.citation_search_risk { + AiSearchRisk::Low => ( + format!("{ANSI_GREEN}{ANSI_BOLD}[ LOW CITATION RISK ]{ANSI_RESET}"), + format!("{ANSI_GREEN}Optimized for ChatGPT Search, Perplexity, and Claude{ANSI_RESET}"), + ), + AiSearchRisk::Medium => ( + format!("{ANSI_YELLOW}{ANSI_BOLD}[ MEDIUM CITATION RISK ]{ANSI_RESET}"), + format!("{ANSI_YELLOW}Missing structured documentation (/llms.txt) or partial bot limits{ANSI_RESET}"), + ), + AiSearchRisk::High => ( + format!("{ANSI_RED}{ANSI_BOLD}[ HIGH CITATION RISK ]{ANSI_RESET}"), + format!("{ANSI_RED}Critical AI search and retrieval engines blocked in robots.txt{ANSI_RESET}"), + ), + }; + + print!( + "{}", + format_section_header("GENERATIVE ENGINE OPTIMIZATION (GEO)") + ); + println!( + " {ANSI_BOLD}Target Domain{ANSI_RESET} : {ANSI_CYAN}{}{ANSI_RESET}", + report.base_url + ); + println!(" {ANSI_BOLD}Citation Risk{ANSI_RESET} : {risk_badge} - {risk_desc}\n"); + + // 1. LLMS.TXT Artifacts + print!("{}", format_section_header("LLMS.TXT PROTOCOL READINESS")); + let llms_status = if report.llms_txt_found { + format!("{ANSI_GREEN}βœ” FOUND (200 OK){ANSI_RESET}") + } else { + format!("{ANSI_RED}βœ– MISSING (404 NOT FOUND){ANSI_RESET}") + }; + let llms_full_status = if report.llms_full_txt_found { + format!("{ANSI_GREEN}βœ” FOUND (200 OK){ANSI_RESET}") + } else { + format!("{ANSI_DIM}β—‹ NOT PUBLISHED{ANSI_RESET}") + }; + + println!(" {ANSI_BOLD}/llms.txt {ANSI_RESET} : {llms_status}"); + println!(" {ANSI_BOLD}/llms-full.txt{ANSI_RESET} : {llms_full_status}"); + if let Some(ref summary) = report.llms_txt_summary { + println!("\n {ANSI_DIM}Preview:{ANSI_RESET}"); + for line in summary.lines() { + println!(" {ANSI_CYAN}{line}{ANSI_RESET}"); + } + } + println!(); + + // 2. Real-Time Search & Retrieval Bots + print!( + "{}", + format_section_header("REAL-TIME AI SEARCH & CITATION CRAWLERS") + ); + for (bot, status) in &report.retrieval_bots { + let status_str = if status == "ALLOWED" { + format!("{ANSI_GREEN}ALLOWED βœ”{ANSI_RESET}") + } else { + format!("{ANSI_RED}{ANSI_BOLD}DISALLOWED βœ–{ANSI_RESET}") + }; + println!(" {ANSI_BOLD}{:<18}{ANSI_RESET} : {status_str}", bot); + } + println!(); + + // 3. AI Training Crawlers + print!( + "{}", + format_section_header("AI FOUNDATION MODEL TRAINING BOTS") + ); + for (bot, status) in &report.training_bots { + let status_str = if status == "ALLOWED" { + format!("{ANSI_GREEN}ALLOWED βœ”{ANSI_RESET}") + } else { + format!("{ANSI_YELLOW}DISALLOWED βœ–{ANSI_RESET}") + }; + println!(" {ANSI_BOLD}{:<18}{ANSI_RESET} : {status_str}", bot); + } + println!(); + + // 4. Actionable Recommendations + if !report.recommendations.is_empty() { + print!( + "{}", + format_section_header("GEO REMEDIATION // ACTION ITEMS") + ); + for (idx, rec) in report.recommendations.iter().enumerate() { + println!(" {ANSI_CYAN}β—ˆ #{:02}{ANSI_RESET} : {rec}", idx + 1); + } + println!(); + } +} + +/// Prints a schema validation assessment in the terminal. +pub fn print_schema_outcome(outcome: &SchemaValidationOutcome) { + println!( + "\n{ANSI_CYAN}{ANSI_BOLD} β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ•— β–ˆβ–ˆβ•—β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•— \n β–ˆβ–ˆβ•”β•β•β•β•β•β–ˆβ–ˆβ•”β•β•β•β•β•β–ˆβ–ˆβ•”β•β•β•β–ˆβ–ˆβ•— β–ˆβ–ˆβ•”β•β•β•β•β•β–ˆβ–ˆβ•”β•β•β•β•β•β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•‘β–ˆβ–ˆβ•”β•β•β•β•β•β–ˆβ–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ–ˆβ–ˆβ•‘β–ˆβ–ˆβ•”β•β•β–ˆβ–ˆβ•—\n β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•‘β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ•”β–ˆβ–ˆβ–ˆβ–ˆβ•”β–ˆβ–ˆβ•‘β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•‘\n β•šβ•β•β•β•β–ˆβ–ˆβ•‘β–ˆβ–ˆβ•”β•β•β• β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•‘ β•šβ•β•β•β•β–ˆβ–ˆβ•‘β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•”β•β•β–ˆβ–ˆβ•‘β–ˆβ–ˆβ•”β•β•β• β–ˆβ–ˆβ•‘β•šβ–ˆβ–ˆβ•”β•β–ˆβ–ˆβ•‘β–ˆβ–ˆβ•”β•β•β–ˆβ–ˆβ•‘\n β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•‘β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β•šβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•”β• β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•‘β•šβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•‘β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ•‘ β•šβ•β• β–ˆβ–ˆβ•‘β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•‘\n β•šβ•β•β•β•β•β•β•β•šβ•β•β•β•β•β•β• β•šβ•β•β•β•β•β• β•šβ•β•β•β•β•β•β• β•šβ•β•β•β•β•β•β•šβ•β• β•šβ•β•β•šβ•β•β•β•β•β•β•β•šβ•β• β•šβ•β•β•šβ•β• β•šβ•β•{ANSI_RESET}\n" + ); + + let (badge, color) = if outcome.is_rich_result_eligible { + ( + format!("{ANSI_GREEN}{ANSI_BOLD}[ GOOGLE RICH RESULTS ELIGIBLE βœ” ]{ANSI_RESET}"), + ANSI_GREEN, + ) + } else { + ( + format!("{ANSI_RED}{ANSI_BOLD}[ NOT RICH RESULTS ELIGIBLE βœ– ]{ANSI_RESET}"), + ANSI_RED, + ) + }; + + print!( + "{}", + format_section_header("SCHEMA.ORG STRUCTURED DATA AUDIT") + ); + println!( + " {ANSI_BOLD}Syntax Status{ANSI_RESET} : {}", + if outcome.is_valid_json { + format!("{ANSI_GREEN}Valid JSON-LD βœ”{ANSI_RESET}") + } else { + format!("{ANSI_RED}Invalid JSON βœ–{ANSI_RESET}") + } + ); + println!( + " {ANSI_BOLD}Detected @type{ANSI_RESET}: {ANSI_CYAN}{}{ANSI_RESET}", + outcome.detected_type.as_deref().unwrap_or("Unknown") + ); + println!(" {ANSI_BOLD}Eligibility {ANSI_RESET} : {badge}\n"); + + if !outcome.missing_required_fields.is_empty() { + println!(" {ANSI_RED}{ANSI_BOLD}🚨 Missing Required Properties (Blocks Rich Results):{ANSI_RESET}"); + for field in &outcome.missing_required_fields { + println!(" {ANSI_RED}βœ– {field}{ANSI_RESET}"); + } + println!(); + } + + if !outcome.missing_recommended_fields.is_empty() { + println!(" {ANSI_YELLOW}⚑ Missing Recommended Properties (Enhances SERP Snippets):{ANSI_RESET}"); + for field in &outcome.missing_recommended_fields { + println!(" {ANSI_YELLOW}β—‹ {field}{ANSI_RESET}"); + } + println!(); + } + + if outcome.is_rich_result_eligible && outcome.missing_recommended_fields.is_empty() { + println!(" {color}βœ” Schema passes all Google Rich Results and schema.org guidelines.{ANSI_RESET}\n"); + } +} + +/// Prints the Cyberpunk / Matrix-style help and home screen for SEO Lens CLI. +pub fn print_cli_help() { + println!( + "\n{ANSI_CYAN}{ANSI_BOLD} β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ•— β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ•—β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—\n β–ˆβ–ˆβ•”β•β•β•β•β•β–ˆβ–ˆβ•”β•β•β•β•β•β–ˆβ–ˆβ•”β•β•β•β–ˆβ–ˆβ•— β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•”β•β•β•β•β•β–ˆβ–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ•‘β–ˆβ–ˆβ•”β•β•β•β•β•\n β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•— β–ˆβ–ˆβ•”β–ˆβ–ˆβ•— β–ˆβ–ˆβ•‘β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—\n β•šβ•β•β•β•β–ˆβ–ˆβ•‘β–ˆβ–ˆβ•”β•β•β• β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•‘ β–ˆβ–ˆβ•”β•β•β• β–ˆβ–ˆβ•‘β•šβ–ˆβ–ˆβ•—β–ˆβ–ˆβ•‘β•šβ•β•β•β•β–ˆβ–ˆβ•‘\n β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•‘β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β•šβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•”β• β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•—β–ˆβ–ˆβ•‘ β•šβ–ˆβ–ˆβ–ˆβ–ˆβ•‘β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ•‘\n β•šβ•β•β•β•β•β•β•β•šβ•β•β•β•β•β•β• β•šβ•β•β•β•β•β• β•šβ•β•β•β•β•β•β•β•šβ•β•β•β•β•β•β•β•šβ•β• β•šβ•β•β•β•β•šβ•β•β•β•β•β•β•{ANSI_RESET}\n {ANSI_DIM}v{} β”‚ High-Performance Website Crawler & AI-Native Technical SEO Engine{ANSI_RESET}\n", + env!("CARGO_PKG_VERSION") + ); + + let print_badge = |title: &str| { + let pad = title.chars().count() + 4; + println!(" {ANSI_BOLD}{ANSI_CYAN}β”Œ{}┐{ANSI_RESET}", "─".repeat(pad)); + println!(" {ANSI_BOLD}{ANSI_CYAN}β”‚{ANSI_RESET} {ANSI_BOLD}{ANSI_BRIGHT_WHITE}{title}{ANSI_RESET} {ANSI_BOLD}{ANSI_CYAN}β”‚{ANSI_RESET}"); + println!(" {ANSI_BOLD}{ANSI_CYAN}β””{}β”˜{ANSI_RESET}", "─".repeat(pad)); + }; + + print_badge("USAGE"); + println!(" {ANSI_BOLD}seolens{ANSI_RESET} {ANSI_CYAN}{ANSI_RESET} {ANSI_YELLOW}[FLAGS]{ANSI_RESET} {ANSI_DIM}[OPTIONS]{ANSI_RESET}\n"); + + print_badge("AUDIT & CRAWL COMMANDS"); + println!(" {ANSI_GREEN}{ANSI_BOLD}audit{ANSI_RESET} {ANSI_CYAN}{ANSI_RESET} Run full website crawl with AIMD adaptive congestion control"); + println!(" {ANSI_GREEN}{ANSI_BOLD}inspect{ANSI_RESET} {ANSI_CYAN}{ANSI_RESET} Instant developer X-ray for a single webpage (DOM, tags, headers)"); + println!(" {ANSI_GREEN}{ANSI_BOLD}check-ai{ANSI_RESET} {ANSI_CYAN}{ANSI_RESET} Audit AI search bot readiness (Perplexity, ChatGPT) & /llms.txt"); + println!(" {ANSI_GREEN}{ANSI_BOLD}schema{ANSI_RESET} {ANSI_CYAN}{ANSI_RESET} Validate JSON-LD against Google Rich Results rules (file or URL)\n"); + + print_badge("DATABASE & REPORTS"); + println!(" {ANSI_GREEN}{ANSI_BOLD}list{ANSI_RESET} List all historical crawl sessions stored in local SQLite DB"); + println!(" {ANSI_GREEN}{ANSI_BOLD}report{ANSI_RESET} {ANSI_CYAN}{ANSI_RESET} Re-export or inspect an existing audit (terminal, md, json)"); + println!(" {ANSI_GREEN}{ANSI_BOLD}issues{ANSI_RESET} {ANSI_CYAN}{ANSI_RESET} Drill down and filter audit findings by severity or category"); + println!(" {ANSI_GREEN}{ANSI_BOLD}delete{ANSI_RESET} {ANSI_CYAN}{ANSI_RESET} Purge a specific crawl session and cascading records"); + println!(" {ANSI_GREEN}{ANSI_BOLD}clean{ANSI_RESET} Clean historical crawl sessions older than N days\n"); + + print_badge("AI AGENT PROTOCOL"); + println!(" {ANSI_GREEN}{ANSI_BOLD}mcp{ANSI_RESET} Start native Model Context Protocol server (stdio for AI agents)\n"); + + print_badge("GLOBAL FLAGS"); + println!(" {ANSI_YELLOW}-h, --help{ANSI_RESET} Print this help guide (or use: seolens --help)"); + println!(" {ANSI_YELLOW}-V, --version{ANSI_RESET} Print version information\n"); + + print_badge("QUICKSTART EXAMPLES"); + println!( + " {ANSI_DIM}# Audit whole website with 500 pages limit & AIMD rate limiting:{ANSI_RESET}" + ); + println!(" seolens audit https://example.com --max-pages 500\n"); + println!(" {ANSI_DIM}# Fast single-page inspection with JSON output:{ANSI_RESET}"); + println!(" seolens inspect https://example.com/pricing --format json\n"); + println!( + " {ANSI_DIM}# Check whether AI bots (Perplexity, ChatGPT) can cite your site:{ANSI_RESET}" + ); + println!(" seolens check-ai https://example.com\n"); + println!(" {ANSI_DIM}# Filter critical issues from a previous crawl session:{ANSI_RESET}"); + println!(" seolens issues --severity critical\n"); + println!(" {ANSI_DIM}# Start MCP server for AI coding agents:{ANSI_RESET}"); + println!(" seolens mcp\n"); +} diff --git a/src/rules/catalog.rs b/src/rules/catalog.rs index a83f54e..e6d5a22 100644 --- a/src/rules/catalog.rs +++ b/src/rules/catalog.rs @@ -3,7 +3,7 @@ //! Authoritative dictionary defining unique error codes, severity ratings, //! audit categories, human-readable descriptions, and remediation guidance. //! -//! Complies with the 120-check technical SEO audit specification in `docs/SEO_RULES_CATALOG.md`. +//! Complies with the 120-check technical SEO audit specification in `docs/rules.md`. //! //! ## Examples //! diff --git a/src/rules/page/schema_val.rs b/src/rules/page/schema_val.rs index 3f3ee41..34e4611 100644 --- a/src/rules/page/schema_val.rs +++ b/src/rules/page/schema_val.rs @@ -5,8 +5,10 @@ //! based on detected page intent. use crate::core::models::{IssueFinding, PageArchetype}; +use crate::error::SeoResult; use crate::parser::ParsedPage; use crate::rules::catalog::{get_rule, RuleId}; +use serde::{Deserialize, Serialize}; use serde_json::Value; /// Checks if a date string conforms to standard ISO 8601 (YYYY-MM-DD or YYYY-MM-DDTHH:MM:SS...). @@ -312,3 +314,134 @@ fn json_has_nested_field(val: &Value, parent: &str, child: &str) -> bool { _ => false, } } + +/// Result of validating a raw schema block against Google Rich Results guidelines. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct SchemaValidationOutcome { + /// Whether the input string parsed successfully as valid JSON. + pub is_valid_json: bool, + /// Schema @type extracted from top-level or @graph container. + pub detected_type: Option, + /// Whether the schema fulfills all required properties for Google Rich Results. + pub is_rich_result_eligible: bool, + /// Missing required fields that completely block Rich Results eligibility. + pub missing_required_fields: Vec, + /// Missing recommended fields that enhance SERP appearance. + pub missing_recommended_fields: Vec, + /// Descriptive error or guidance message. + pub error_message: Option, +} + +/// Validates a raw JSON-LD snippet or HTML block against Google Rich Results eligibility rules. +pub fn validate_raw_schema( + raw: &str, + expected_type: Option<&str>, +) -> SeoResult { + let trimmed = raw.trim(); + let json_text = if let Some(start) = trimmed.find("') { + let rest = &trimmed[start + content_start + 1..]; + if let Some(end) = rest.find("") { + rest[..end].trim() + } else { + trimmed + } + } else { + trimmed + } + } else { + trimmed + }; + + let val: Value = match serde_json::from_str(json_text) { + Ok(v) => v, + Err(e) => { + return Ok(SchemaValidationOutcome { + is_valid_json: false, + detected_type: None, + is_rich_result_eligible: false, + missing_required_fields: Vec::new(), + missing_recommended_fields: Vec::new(), + error_message: Some(format!("Invalid JSON syntax: {e}")), + }); + } + }; + + let detected_type = if let Some(t) = val.get("@type").and_then(|v| v.as_str()) { + Some(t.to_string()) + } else if let Some(Value::Array(graph)) = val.get("@graph") { + graph + .first() + .and_then(|item| item.get("@type")) + .and_then(|v| v.as_str()) + .map(|s| s.to_string()) + } else { + None + }; + + let target = expected_type + .map(|s| s.to_string()) + .or_else(|| detected_type.clone()); + let mut missing_required = Vec::new(); + let mut missing_recommended = Vec::new(); + + if let Some(ref t) = target { + if t.eq_ignore_ascii_case("product") { + if !json_has_field(&val, "name") { + missing_required.push("name".to_string()); + } + if !json_has_field(&val, "image") { + missing_recommended.push("image".to_string()); + } + if !json_has_field(&val, "offers") { + missing_required.push("offers".to_string()); + } + if !json_has_field(&val, "aggregateRating") && !json_has_field(&val, "review") { + missing_recommended.push("aggregateRating".to_string()); + } + } else if t.eq_ignore_ascii_case("article") + || t.eq_ignore_ascii_case("newsarticle") + || t.eq_ignore_ascii_case("blogposting") + { + if !json_has_field(&val, "headline") { + missing_required.push("headline".to_string()); + } + if !json_has_field(&val, "author") { + missing_recommended.push("author".to_string()); + } + if !json_has_field(&val, "datePublished") { + missing_recommended.push("datePublished".to_string()); + } + if !json_has_field(&val, "image") { + missing_recommended.push("image".to_string()); + } + } else if t.eq_ignore_ascii_case("faqpage") { + if !json_has_field(&val, "mainEntity") { + missing_required.push("mainEntity".to_string()); + } + } else if t.eq_ignore_ascii_case("breadcrumblist") + && !json_has_field(&val, "itemListElement") + { + missing_required.push("itemListElement".to_string()); + } + } + + let is_eligible = detected_type.is_some() && missing_required.is_empty(); + let err_msg = if !missing_required.is_empty() { + Some(format!( + "Missing required Google Rich Result property '{}'", + missing_required.join("', '") + )) + } else { + None + }; + + Ok(SchemaValidationOutcome { + is_valid_json: true, + detected_type, + is_rich_result_eligible: is_eligible, + missing_required_fields: missing_required, + missing_recommended_fields: missing_recommended, + error_message: err_msg, + }) +} diff --git a/src/storage/mod.rs b/src/storage/mod.rs index a0881c3..280b9cb 100644 --- a/src/storage/mod.rs +++ b/src/storage/mod.rs @@ -8,10 +8,11 @@ pub mod sqlite; pub mod writer; pub use queries::{ - get_crawl, get_crawl_issues, get_crawl_pages, init_crawl_session, list_crawls, - update_crawl_status, CrawlSessionInit, + clean_all_crawls, clean_crawls_older_than, count_issues_filtered, delete_crawl, get_crawl, + get_crawl_issues, get_crawl_pages, init_crawl_session, list_crawls, query_issues_filtered, + update_crawl_status, CrawlSessionInit, IssueFilterCriteria, }; -pub use sqlite::{default_db_path, open_connection, SCHEMA}; +pub use sqlite::{default_db_path, local_db_path, open_connection, resolve_db_path, SCHEMA}; pub use writer::{spawn_db_writer, DbMessage, DbWriterHandle}; use crate::core::models::{CrawlSummary, IssueCategory, IssueFinding, PageReport, Severity}; @@ -33,6 +34,13 @@ impl Database { Ok(Self { path: path_buf }) } + /// Creates a Database handle pointing to the specified path without opening a connection immediately. + pub fn from_path(path: impl AsRef) -> Self { + Self { + path: path.as_ref().to_path_buf(), + } + } + /// Returns the database file path on disk. pub fn path(&self) -> &Path { &self.path @@ -124,4 +132,63 @@ impl Database { flush_interval, ) } + + /// Alias for initializing a new crawl session. + pub fn init_crawl(&self, init: &CrawlSessionInit) -> SeoResult<()> { + self.init_crawl_session(init) + } + + /// Deletes a specific crawl session and its cascading child records. + pub fn delete_crawl(&self, session_id: &str) -> SeoResult { + let conn = self.connect()?; + queries::delete_crawl(&conn, session_id) + } + + /// Purges crawl sessions older than `days` days. + pub fn clean_crawls_older_than(&self, days: u32) -> SeoResult { + let conn = self.connect()?; + queries::clean_crawls_older_than(&conn, days) + } + + /// Purges all historical crawl sessions from the database. + pub fn clean_all_crawls(&self) -> SeoResult { + let conn = self.connect()?; + queries::clean_all_crawls(&conn) + } + + /// Queries issues with advanced filters (severity, category, code, url substring, and pagination). + pub fn query_issues_filtered( + &self, + session_id: &str, + criteria: &IssueFilterCriteria, + ) -> SeoResult> { + let conn = self.connect()?; + queries::query_issues_filtered(&conn, session_id, criteria) + } + + /// Counts issues matching the specified filters. + pub fn count_issues_filtered( + &self, + session_id: &str, + criteria: &IssueFilterCriteria, + ) -> SeoResult { + let conn = self.connect()?; + queries::count_issues_filtered(&conn, session_id, criteria) + } + + /// Directly persists a batch of page reports into SQLite within a transaction. + pub fn save_page_batch(&self, session_id: &str, pages: &[PageReport]) -> SeoResult<()> { + let mut conn = self.connect()?; + let mut pages_vec = pages.to_vec(); + let mut issues_vec = Vec::new(); + writer::flush_to_db(&mut conn, session_id, &mut pages_vec, &mut issues_vec) + } + + /// Directly persists a batch of issue findings into SQLite within a transaction. + pub fn save_issue_batch(&self, session_id: &str, issues: &[IssueFinding]) -> SeoResult<()> { + let mut conn = self.connect()?; + let mut pages_vec = Vec::new(); + let mut issues_vec = issues.to_vec(); + writer::flush_to_db(&mut conn, session_id, &mut pages_vec, &mut issues_vec) + } } diff --git a/src/storage/queries.rs b/src/storage/queries.rs index 239e9a7..0ad20c7 100644 --- a/src/storage/queries.rs +++ b/src/storage/queries.rs @@ -582,3 +582,144 @@ pub fn get_crawl_issues( Ok(issues) } + +/// Deletes a specific crawl session from SQLite (cascading to pages, issues, links, etc.). +pub fn delete_crawl(conn: &Connection, session_id: &str) -> SeoResult { + let rows = conn.execute( + "DELETE FROM crawls WHERE session_id = ?1", + params![session_id], + )?; + Ok(rows > 0) +} + +/// Cleans/purges crawl sessions started older than `days` days ago. +pub fn clean_crawls_older_than(conn: &Connection, days: u32) -> SeoResult { + let rows = conn.execute( + "DELETE FROM crawls WHERE started_at < datetime('now', '-' || ?1 || ' days')", + params![days], + )?; + Ok(rows) +} + +/// Purges all historical crawl sessions from SQLite. +pub fn clean_all_crawls(conn: &Connection) -> SeoResult { + let rows = conn.execute("DELETE FROM crawls", [])?; + Ok(rows) +} + +/// Criteria for filtering issues during queries and counting. +#[derive(Debug, Clone, Default)] +pub struct IssueFilterCriteria<'a> { + pub severity: Option, + pub category: Option, + pub code: Option<&'a str>, + pub url_substring: Option<&'a str>, + pub limit: usize, + pub offset: usize, +} + +/// Advanced query for crawl issues with optional filters for severity, category, rule code, URL substring, and pagination. +pub fn query_issues_filtered( + conn: &Connection, + session_id: &str, + criteria: &IssueFilterCriteria, +) -> SeoResult> { + let mut sql = "SELECT target_url, code, category, severity, title, message, source_page_url + FROM issues WHERE crawl_id = ?1" + .to_string(); + let mut params_vec: Vec> = vec![Box::new(session_id.to_string())]; + + if let Some(sev) = criteria.severity { + params_vec.push(Box::new(sev.as_u8())); + sql.push_str(&format!(" AND severity = ?{}", params_vec.len())); + } + if let Some(cat) = criteria.category { + params_vec.push(Box::new(cat.as_str().to_string())); + sql.push_str(&format!(" AND category = ?{}", params_vec.len())); + } + if let Some(code) = criteria.code { + params_vec.push(Box::new(code.to_string())); + sql.push_str(&format!(" AND code = ?{}", params_vec.len())); + } + if let Some(sub) = criteria.url_substring { + params_vec.push(Box::new(format!("%{sub}%"))); + sql.push_str(&format!(" AND target_url LIKE ?{}", params_vec.len())); + } + + sql.push_str(" ORDER BY severity ASC, id ASC"); + let limit = if criteria.limit == 0 { + 50 + } else { + criteria.limit + }; + params_vec.push(Box::new(limit as i64)); + sql.push_str(&format!(" LIMIT ?{}", params_vec.len())); + params_vec.push(Box::new(criteria.offset as i64)); + sql.push_str(&format!(" OFFSET ?{}", params_vec.len())); + + let mut stmt = conn.prepare(&sql)?; + let param_refs: Vec<&dyn rusqlite::ToSql> = params_vec.iter().map(|b| b.as_ref()).collect(); + + let rows = stmt.query_map(param_refs.as_slice(), |row| { + let target_url: String = row.get(0)?; + let code_str: String = row.get(1)?; + let cat_str: String = row.get(2)?; + let sev_u8: u8 = row.get(3)?; + let title: String = row.get(4)?; + let message: String = row.get(5)?; + let source_page_url: Option = row.get(6)?; + + let code = RuleId::from_code(&code_str).unwrap_or(RuleId::ErrHttp5xxServerError); + let category = + IssueCategory::from_str_name(&cat_str).unwrap_or(IssueCategory::HttpTransport); + let severity = Severity::from_u8(sev_u8).unwrap_or(Severity::Warning); + + Ok(IssueFinding { + code, + category, + severity, + title: CompactString::new(&title), + message, + target_url, + source_page_url, + }) + })?; + + let mut list = Vec::new(); + for r in rows { + list.push(r?); + } + Ok(list) +} + +/// Counts total issues matching filters for pagination. +pub fn count_issues_filtered( + conn: &Connection, + session_id: &str, + criteria: &IssueFilterCriteria, +) -> SeoResult { + let mut sql = "SELECT COUNT(*) FROM issues WHERE crawl_id = ?1".to_string(); + let mut params_vec: Vec> = vec![Box::new(session_id.to_string())]; + + if let Some(sev) = criteria.severity { + params_vec.push(Box::new(sev.as_u8())); + sql.push_str(&format!(" AND severity = ?{}", params_vec.len())); + } + if let Some(cat) = criteria.category { + params_vec.push(Box::new(cat.as_str().to_string())); + sql.push_str(&format!(" AND category = ?{}", params_vec.len())); + } + if let Some(code) = criteria.code { + params_vec.push(Box::new(code.to_string())); + sql.push_str(&format!(" AND code = ?{}", params_vec.len())); + } + if let Some(sub) = criteria.url_substring { + params_vec.push(Box::new(format!("%{sub}%"))); + sql.push_str(&format!(" AND target_url LIKE ?{}", params_vec.len())); + } + + let mut stmt = conn.prepare(&sql)?; + let param_refs: Vec<&dyn rusqlite::ToSql> = params_vec.iter().map(|b| b.as_ref()).collect(); + let count: usize = stmt.query_row(param_refs.as_slice(), |r| r.get(0))?; + Ok(count) +} diff --git a/src/storage/sqlite.rs b/src/storage/sqlite.rs index d274ea7..334a074 100644 --- a/src/storage/sqlite.rs +++ b/src/storage/sqlite.rs @@ -10,11 +10,67 @@ use std::path::{Path, PathBuf}; /// Embedded authoritative SQLite DDL schema. pub const SCHEMA: &str = include_str!("schema.sql"); -/// Returns the standard default path for the SQLite database: `.seolens/seolens.db`. -pub fn default_db_path() -> PathBuf { +/// Returns the local database path in the current working directory: `.seolens/seolens.db`. +pub fn local_db_path() -> PathBuf { PathBuf::from(".seolens").join("seolens.db") } +/// Returns the standard default path for the SQLite persistence database. +/// +/// Resolution precedence: +/// 1. `SEOLENS_DB_PATH` environment variable (if set and non-empty). +/// 2. Local `./.seolens/seolens.db` in current working directory if it already exists. +/// 3. OS standard user data directory: +/// - Linux: `$XDG_DATA_HOME/seolens/seolens.db` (defaults to `~/.local/share/seolens/seolens.db`) +/// - macOS: `~/Library/Application Support/seolens/seolens.db` +/// - Windows: `%LOCALAPPDATA%\seolens\seolens.db` +/// 4. Fallback to `./.seolens/seolens.db` if the OS data directory cannot be determined. +pub fn default_db_path() -> PathBuf { + // 1. Environment variable override + if let Ok(env_path) = std::env::var("SEOLENS_DB_PATH") { + let trimmed = env_path.trim(); + if !trimmed.is_empty() { + return PathBuf::from(trimmed); + } + } + + // 2. Existing local .seolens/seolens.db in current working directory + let local = local_db_path(); + if local.exists() { + return local; + } + + // 3. Standard modern OS user data directory (XDG on Linux, App Support on macOS, AppData on Windows) + if let Some(mut data_dir) = dirs::data_dir() { + data_dir.push("seolens"); + data_dir.push("seolens.db"); + return data_dir; + } + + // 4. Fallback + local +} + +/// Resolves the database path based on explicit CLI arguments, the local flag, and environment/OS defaults. +/// +/// Precedence: +/// 1. Explicit path (`--db-path `) if provided. +/// 2. `local` (`-L, --local`) flag if true: returns `./.seolens/seolens.db`. +/// 3. Standard resolution via [`default_db_path`]: +/// - `SEOLENS_DB_PATH` environment variable +/// - Local `./.seolens/seolens.db` if it already exists +/// - Standard OS user data directory (`~/.local/share/seolens/seolens.db` on Linux, etc.) +/// - Fallback `./.seolens/seolens.db` +pub fn resolve_db_path(explicit: Option, local: bool) -> PathBuf { + if let Some(p) = explicit { + return p; + } + if local { + return local_db_path(); + } + default_db_path() +} + /// Opens a SQLite connection to the specified path and applies WAL mode and schema. pub fn open_connection(path: &Path) -> SeoResult { if let Some(parent) = path.parent() { diff --git a/src/storage/writer.rs b/src/storage/writer.rs index da0f246..028472c 100644 --- a/src/storage/writer.rs +++ b/src/storage/writer.rs @@ -207,7 +207,7 @@ async fn run_writer_loop( Ok(()) } -fn flush_to_db( +pub(crate) fn flush_to_db( conn: &mut Connection, crawl_id: &str, pages: &mut Vec, diff --git a/tests/cli_capabilities_tests.rs b/tests/cli_capabilities_tests.rs new file mode 100644 index 0000000..480100c --- /dev/null +++ b/tests/cli_capabilities_tests.rs @@ -0,0 +1,634 @@ +//! # CLI Capabilities & Extended Commands Integration Tests (Micro-Phase 01) +//! +//! Validates new subcommands (`issues`, `check-ai`, `delete`, `clean`, `schema`), +//! enriched flags (`--include`, `--exclude`, `--header`, `--quiet`, `--format`), +//! database cascade deletions, and AI readiness auditing. + +use clap::Parser; +use seo_lens::cli::args::{Cli, Commands}; +use seo_lens::core::config::CrawlConfig; +use seo_lens::core::models::{ + IssueCategory, IssueFinding, PageReport, RobotsFlags, RuleId, Severity, +}; +use seo_lens::crawler::ai_check::{audit_ai_readiness, AiSearchRisk}; +use seo_lens::crawler::engine::run_crawl; +use seo_lens::rules::page::schema_val::validate_raw_schema; +use seo_lens::storage::{CrawlSessionInit, Database, IssueFilterCriteria}; +use std::path::PathBuf; +use std::sync::atomic::{AtomicUsize, Ordering}; +use wiremock::matchers::{method, path}; +use wiremock::{Mock, MockServer, ResponseTemplate}; + +static TEST_COUNTER: AtomicUsize = AtomicUsize::new(100); + +fn unique_test_db_path() -> PathBuf { + let id = TEST_COUNTER.fetch_add(1, Ordering::SeqCst); + let mut dir = std::env::temp_dir(); + dir.push(format!("seolens_cli_test_{}_{}.db", std::process::id(), id)); + dir +} + +fn mock_page_report(crawl_id: &str, url: &str, status_code: u16) -> PageReport { + PageReport { + crawl_id: crawl_id.into(), + url: url.to_string(), + url_hash: seo_lens::core::url::url_hash(url), + final_url: Some(url.to_string()), + status_code, + content_type: "text/html; charset=utf-8".into(), + size_bytes: 1024, + ttfb_ms: 80, + crawl_depth: 1, + title: Some("Test Page".to_string()), + title_length: 9, + meta_description: Some("Test description".to_string()), + meta_desc_length: 16, + canonical_url: Some(url.to_string()), + html_lang: Some("en".into()), + robots_flags: RobotsFlags::NONE, + is_sitemap_url: false, + is_internal: true, + h1_primary: Some("Heading".to_string()), + h1_count: 1, + h2_headings: vec![], + h3_headings: vec![], + word_count: 250, + content_hash: 111, + simhash: 222, + is_https: true, + has_hsts: true, + has_csp: true, + has_x_frame: true, + has_x_content_type: true, + links: vec![], + images: vec![], + schemas: vec![], + hreflangs: vec![], + issues: vec![], + page_intent: Default::default(), + ..Default::default() + } +} + +#[test] +fn test_cli_argument_parsing_new_commands() { + // 1. issues subcommand + let cli = Cli::try_parse_from([ + "seolens", + "issues", + "session_abc", + "--severity", + "critical", + "--category", + "security", + "--limit", + "25", + "--format", + "json", + ]) + .expect("Failed to parse issues command"); + + match cli.command { + Commands::Issues(args) => { + assert_eq!( + args.session.unwrap_or(args.session_pos.unwrap()), + "session_abc" + ); + assert_eq!(args.severity.as_deref(), Some("critical")); + assert_eq!(args.category.as_deref(), Some("security")); + assert_eq!(args.limit, 25); + assert_eq!(args.format, "json"); + } + _ => panic!("Expected Issues command"), + } + + // 2. check-ai subcommand + let cli = Cli::try_parse_from([ + "seolens", + "check-ai", + "https://example.com", + "--format", + "json", + "--timeout", + "10", + ]) + .expect("Failed to parse check-ai command"); + + match cli.command { + Commands::CheckAi(args) => { + assert_eq!(args.url, "https://example.com"); + assert_eq!(args.format, "json"); + assert_eq!(args.timeout, 10); + } + _ => panic!("Expected CheckAi command"), + } + + // 3. delete subcommand + let cli = Cli::try_parse_from(["seolens", "delete", "session_to_delete"]) + .expect("Failed to parse delete command"); + + match cli.command { + Commands::Delete(args) => { + assert_eq!( + args.session.unwrap_or(args.session_pos.unwrap()), + "session_to_delete" + ); + } + _ => panic!("Expected Delete command"), + } + + // 4. clean subcommand + let cli = Cli::try_parse_from(["seolens", "clean", "--older-than", "14"]) + .expect("Failed to parse clean command"); + + match cli.command { + Commands::Clean(args) => { + assert_eq!(args.older_than, Some(14)); + assert!(!args.all); + } + _ => panic!("Expected Clean command"), + } + + // 5. schema subcommand + let cli = Cli::try_parse_from([ + "seolens", + "schema", + "https://example.com/product", + "--type", + "Product", + "--format", + "json", + ]) + .expect("Failed to parse schema command"); + + match cli.command { + Commands::Schema(args) => { + assert_eq!(args.target, "https://example.com/product"); + assert_eq!(args.expected_type.as_deref(), Some("Product")); + assert_eq!(args.format, "json"); + } + _ => panic!("Expected Schema command"), + } +} + +#[test] +fn test_cli_argument_parsing_extended_flags() { + let cli = Cli::try_parse_from([ + "seolens", + "audit", + "https://example.com", + "--include", + "^/docs/.*", + "--exclude", + "^/docs/v1/.*", + "-H", + "Authorization: Bearer secret", + "-H", + "CF-Access-Client-Id: 12345", + "--sitemap", + "https://example.com/custom-sitemap.xml", + "--name", + "Docs Audit Q3", + "-q", + ]) + .expect("Failed to parse enriched audit command"); + + match cli.command { + Commands::Audit(args) => { + assert_eq!(args.include.as_deref(), Some("^/docs/.*")); + assert_eq!(args.exclude.as_deref(), Some("^/docs/v1/.*")); + assert_eq!(args.headers.len(), 2); + assert_eq!(args.headers[0], "Authorization: Bearer secret"); + assert_eq!(args.headers[1], "CF-Access-Client-Id: 12345"); + assert_eq!( + args.sitemap.as_deref(), + Some("https://example.com/custom-sitemap.xml") + ); + assert_eq!(args.name.as_deref(), Some("Docs Audit Q3")); + assert!(args.quiet); + } + _ => panic!("Expected Audit command"), + } +} + +#[test] +fn test_storage_delete_and_clean_cascading() { + let db_path = unique_test_db_path(); + let db = Database::open(&db_path).unwrap(); + + let session_1 = "sess_del_1"; + let session_2 = "sess_del_2"; + + db.init_crawl(&CrawlSessionInit { + session_id: session_1.into(), + target_url: "https://site1.com".into(), + max_pages: 10, + max_depth: 2, + respect_robots: true, + render_js: false, + }) + .unwrap(); + + db.init_crawl(&CrawlSessionInit { + session_id: session_2.into(), + target_url: "https://site2.com".into(), + max_pages: 10, + max_depth: 2, + respect_robots: true, + render_js: false, + }) + .unwrap(); + + // Insert page and issue for session 1 + let mut page = mock_page_report(session_1, "https://site1.com/p1", 200); + let issue = IssueFinding { + code: RuleId::ErrHttp4xxClientError, + category: IssueCategory::Indexability, + severity: Severity::Critical, + title: "Broken Page".into(), + message: "Not found".into(), + target_url: "https://site1.com/p1".into(), + source_page_url: None, + }; + page.issues.push(issue.clone()); + + db.save_page_batch(session_1, &[page]).unwrap(); + db.save_issue_batch(session_1, &[issue]).unwrap(); + + // Verify session 1 exists + assert!(db.get_crawl(session_1).unwrap().is_some()); + assert_eq!(db.get_crawl_pages(session_1, 10, 0).unwrap().len(), 1); + assert_eq!(db.get_crawl_issues(session_1, None, None).unwrap().len(), 2); + + // Delete session 1 + let deleted = db.delete_crawl(session_1).unwrap(); + assert!(deleted, "Expected delete_crawl to return true"); + + // Verify session 1 is gone and cascading deletion cleaned up pages and issues + assert!(db.get_crawl(session_1).unwrap().is_none()); + assert_eq!(db.get_crawl_pages(session_1, 10, 0).unwrap().len(), 0); + assert_eq!(db.get_crawl_issues(session_1, None, None).unwrap().len(), 0); + + // Session 2 should remain intact + assert!(db.get_crawl(session_2).unwrap().is_some()); + + // Clean all + let cleaned = db.clean_all_crawls().unwrap(); + assert_eq!(cleaned, 1); + assert!(db.get_crawl(session_2).unwrap().is_none()); + + let _ = std::fs::remove_file(&db_path); +} + +#[test] +fn test_storage_query_issues_advanced_filtering() { + let db_path = unique_test_db_path(); + let db = Database::open(&db_path).unwrap(); + + let session = "sess_adv_filter"; + db.init_crawl(&CrawlSessionInit { + session_id: session.into(), + target_url: "https://testfilter.com".into(), + max_pages: 10, + max_depth: 2, + respect_robots: true, + render_js: false, + }) + .unwrap(); + + let issues = vec![ + IssueFinding { + code: RuleId::ErrHttp4xxClientError, + category: IssueCategory::Indexability, + severity: Severity::Critical, + title: "404 Error".into(), + message: "Target is 404".into(), + target_url: "https://testfilter.com/broken".into(), + source_page_url: None, + }, + IssueFinding { + code: RuleId::WarnTitleTooLong, + category: IssueCategory::TitleMetadata, + severity: Severity::Warning, + title: "Title Long".into(), + message: "Title length > 60".into(), + target_url: "https://testfilter.com/blog/long-title".into(), + source_page_url: None, + }, + IssueFinding { + code: RuleId::WarnGraphRedirectChain, + category: IssueCategory::Links, + severity: Severity::Alert, + title: "Redirect Chain".into(), + message: "3 redirect hops".into(), + target_url: "https://testfilter.com/blog/redirect".into(), + source_page_url: None, + }, + ]; + + db.save_issue_batch(session, &issues).unwrap(); + + // 1. Filter by severity Critical + let criticals = db + .query_issues_filtered( + session, + &IssueFilterCriteria { + severity: Some(Severity::Critical), + limit: 10, + ..Default::default() + }, + ) + .unwrap(); + assert_eq!(criticals.len(), 1); + assert_eq!(criticals[0].code, RuleId::ErrHttp4xxClientError); + + // 2. Filter by category Titles + let title_issues = db + .query_issues_filtered( + session, + &IssueFilterCriteria { + category: Some(IssueCategory::TitleMetadata), + limit: 10, + ..Default::default() + }, + ) + .unwrap(); + assert_eq!(title_issues.len(), 1); + assert_eq!(title_issues[0].code, RuleId::WarnTitleTooLong); + + // 3. Filter by URL pattern '/blog/' + let blog_issues = db + .query_issues_filtered( + session, + &IssueFilterCriteria { + url_substring: Some("/blog/"), + limit: 10, + ..Default::default() + }, + ) + .unwrap(); + assert_eq!(blog_issues.len(), 2); + + // 4. Filter by rule code + let code_issues = db + .query_issues_filtered( + session, + &IssueFilterCriteria { + code: Some("WARN_GRAPH_REDIRECT_CHAIN"), + limit: 10, + ..Default::default() + }, + ) + .unwrap(); + assert_eq!(code_issues.len(), 1); + assert_eq!( + code_issues[0].target_url, + "https://testfilter.com/blog/redirect" + ); + + let _ = std::fs::remove_file(&db_path); +} + +#[tokio::test] +async fn test_crawl_url_include_and_exclude_filters() { + let server = MockServer::start().await; + let base = server.uri(); + + // Root links to 4 URLs: + // - /blog/article-1 (included) + // - /blog/drafts/test (excluded by drafts rule) + // - /shop/item-1 (excluded by include pattern requiring /blog/) + Mock::given(method("GET")) + .and(path("/")) + .respond_with(ResponseTemplate::new(200).set_body_string(format!( + r#"Home + + Article 1 + Draft + Item 1 + "# + ))) + .mount(&server) + .await; + + Mock::given(method("GET")) + .and(path("/blog/article-1")) + .respond_with(ResponseTemplate::new(200).set_body_string( + "Article 1

Article 1 text

", + )) + .mount(&server) + .await; + + Mock::given(method("GET")) + .and(path("/blog/drafts/test")) + .respond_with(ResponseTemplate::new(200).set_body_string( + "Draft

Draft

", + )) + .mount(&server) + .await; + + Mock::given(method("GET")) + .and(path("/shop/item-1")) + .respond_with(ResponseTemplate::new(200).set_body_string( + "Shop

Shop

", + )) + .mount(&server) + .await; + + let mut config = CrawlConfig::new(&base).unwrap(); + config.include_regex = Some(r".*/blog/.*".to_string()); + config.exclude_regex = Some(r".*/drafts/.*".to_string()); + config.max_pages = 20; + + let result = run_crawl(&config, None).await.unwrap(); + + let crawled_urls: Vec = result.pages.into_iter().map(|p| p.url).collect(); + + // Root is crawled + assert!(crawled_urls.iter().any(|u| u.ends_with('/'))); + // /blog/article-1 matches include and does not match exclude -> crawled + assert!(crawled_urls.iter().any(|u| u.contains("/blog/article-1"))); + // /blog/drafts/test matches exclude -> MUST NOT be crawled + assert!(!crawled_urls.iter().any(|u| u.contains("/blog/drafts/test"))); + // /shop/item-1 does not match include -> MUST NOT be crawled + assert!(!crawled_urls.iter().any(|u| u.contains("/shop/item-1"))); +} + +#[tokio::test] +async fn test_ai_readiness_audit_live_mock() { + let server = MockServer::start().await; + let base = server.uri(); + + // robots.txt disallows GPTBot and PerplexityBot, allows others + Mock::given(method("GET")) + .and(path("/robots.txt")) + .respond_with(ResponseTemplate::new(200).set_body_string( + "User-agent: GPTBot\nDisallow: /\n\nUser-agent: PerplexityBot\nDisallow: /\n\nUser-agent: *\nAllow: /\n", + )) + .mount(&server) + .await; + + // llms.txt is present and structured + Mock::given(method("GET")) + .and(path("/llms.txt")) + .respond_with(ResponseTemplate::new(200).set_body_string( + "# Acme Corp\n> Fast cloud platform for developers\n\n## Docs\n- [API](https://example.com/api): Core REST API\n", + )) + .mount(&server) + .await; + + // llms-full.txt returns 404 + Mock::given(method("GET")) + .and(path("/llms-full.txt")) + .respond_with(ResponseTemplate::new(404)) + .mount(&server) + .await; + + let report = audit_ai_readiness(&base, "SEOLens/1.0", std::time::Duration::from_secs(5)) + .await + .unwrap(); + + assert!(report.llms_txt_found); + assert!(!report.llms_full_txt_found); + assert_eq!(report.citation_search_risk, AiSearchRisk::High); + assert_eq!( + report + .retrieval_bots + .get("PerplexityBot") + .map(|s| s.as_str()), + Some("DISALLOWED") + ); + assert_eq!( + report + .retrieval_bots + .get("OAI-SearchBot") + .map(|s| s.as_str()), + Some("ALLOWED") + ); + assert_eq!( + report.training_bots.get("GPTBot").map(|s| s.as_str()), + Some("DISALLOWED") + ); + assert!(!report.recommendations.is_empty()); +} + +#[test] +fn test_schema_validator_google_rich_results() { + // 1. Valid Product Schema with offers + let valid_product = r#"{ + "@context": "https://schema.org", + "@type": "Product", + "name": "Mechanical Keyboard", + "image": "https://example.com/keyboard.jpg", + "offers": { + "@type": "Offer", + "price": "149.00", + "priceCurrency": "USD", + "availability": "https://schema.org/InStock" + } + }"#; + + let outcome = validate_raw_schema(valid_product, Some("Product")).unwrap(); + assert!(outcome.is_valid_json); + assert_eq!(outcome.detected_type.as_deref(), Some("Product")); + assert!(outcome.is_rich_result_eligible); + assert!(outcome.missing_required_fields.is_empty()); + + // 2. Product Schema missing 'offers' + let incomplete_product = r#"{ + "@context": "https://schema.org", + "@type": "Product", + "name": "Incomplete Keyboard" + }"#; + + let outcome2 = validate_raw_schema(incomplete_product, Some("Product")).unwrap(); + assert!(outcome2.is_valid_json); + assert!(!outcome2.is_rich_result_eligible); + assert!(outcome2 + .missing_required_fields + .contains(&"offers".to_string())); +} + +#[test] +fn test_cli_local_and_db_path_flags() { + // 1. Audit command with --local and -L + let parsed1 = + Cli::try_parse_from(["seolens", "audit", "https://example.com", "--local"]).unwrap(); + if let Commands::Audit(args) = parsed1.command { + assert!(args.local); + assert!(args.db_path.is_none()); + } else { + panic!("Expected Audit command"); + } + + let parsed2 = Cli::try_parse_from(["seolens", "audit", "https://example.com", "-L"]).unwrap(); + if let Commands::Audit(args) = parsed2.command { + assert!(args.local); + } else { + panic!("Expected Audit command"); + } + + // 2. Audit with explicit --db-path + let parsed3 = Cli::try_parse_from([ + "seolens", + "audit", + "https://example.com", + "--db-path", + "/tmp/custom.db", + ]) + .unwrap(); + if let Commands::Audit(args) = parsed3.command { + assert_eq!(args.db_path, Some(PathBuf::from("/tmp/custom.db"))); + assert!(!args.local); + } else { + panic!("Expected Audit command"); + } + + // 3. List command + let parsed_list = Cli::try_parse_from(["seolens", "list", "--local"]).unwrap(); + if let Commands::List(args) = parsed_list.command { + assert!(args.local); + } else { + panic!("Expected List command"); + } + + // 4. Mcp command + let parsed_mcp = Cli::try_parse_from(["seolens", "mcp", "-L"]).unwrap(); + if let Commands::Mcp(args) = parsed_mcp.command { + assert!(args.local); + } else { + panic!("Expected Mcp command"); + } + + // 5. Report command + let parsed_report = Cli::try_parse_from(["seolens", "report", "crawl_123", "--local"]).unwrap(); + if let Commands::Report(args) = parsed_report.command { + assert!(args.local); + } else { + panic!("Expected Report command"); + } + + // 6. Issues command + let parsed_issues = Cli::try_parse_from(["seolens", "issues", "crawl_123", "-L"]).unwrap(); + if let Commands::Issues(args) = parsed_issues.command { + assert!(args.local); + } else { + panic!("Expected Issues command"); + } + + // 7. Delete command + let parsed_del = Cli::try_parse_from(["seolens", "delete", "crawl_123", "--local"]).unwrap(); + if let Commands::Delete(args) = parsed_del.command { + assert!(args.local); + } else { + panic!("Expected Delete command"); + } + + // 8. Clean command + let parsed_clean = Cli::try_parse_from(["seolens", "clean", "--all", "-L"]).unwrap(); + if let Commands::Clean(args) = parsed_clean.command { + assert!(args.local); + } else { + panic!("Expected Clean command"); + } +} diff --git a/tests/csv_export_tests.rs b/tests/csv_export_tests.rs new file mode 100644 index 0000000..963f28c --- /dev/null +++ b/tests/csv_export_tests.rs @@ -0,0 +1,360 @@ +//! # Screaming Frog Compatible CSV Suite Integration Tests (Phase 11) +//! +//! Validates generation of the four industry-standard CSV export files: +//! 1. `internal_all.csv` (Screaming Frog internal crawl table) +//! 2. `issues_all.csv` (Full defect triage log) +//! 3. `response_codes.csv` (URL routing & redirect map) +//! 4. `external_all.csv` (Outbound link audit) + +use hashbrown::HashMap; +use seo_lens::core::models::{ + DiscoveredLink, IssueCategory, IssueFinding, PageReport, RobotsFlags, RuleId, Severity, +}; +use seo_lens::crawler::engine::CrawlResult; +use seo_lens::graph::SiteGraph; +use seo_lens::report::csv::export_csv_suite; +use std::fs; +use std::path::PathBuf; +use std::sync::atomic::{AtomicUsize, Ordering}; +use std::time::Duration; + +static CSV_TEST_COUNTER: AtomicUsize = AtomicUsize::new(1); + +fn unique_test_csv_dir() -> PathBuf { + let id = CSV_TEST_COUNTER.fetch_add(1, Ordering::SeqCst); + let mut dir = std::env::temp_dir(); + dir.push(format!("seolens_csv_test_{}_{}", std::process::id(), id)); + dir +} + +fn create_test_crawl_result() -> CrawlResult { + let mut graph = SiteGraph::new(); + + let url_home = "https://example.com/".to_string(); + let url_about = "https://example.com/about".to_string(); + let url_redirect = "https://example.com/old-page".to_string(); + let url_noindex = "https://example.com/privacy".to_string(); + + graph.add_node(&url_home, 200, 0, true); + graph.add_node(&url_about, 200, 1, true); + graph.add_node(&url_redirect, 301, 1, false); + graph.add_node(&url_noindex, 200, 1, false); + + // Links: home -> about, home -> old-page, home -> privacy, home -> external + graph.add_edge( + &url_home, + &url_about, + seo_lens::graph::LinkEdgeType::InternalHyperlink, + false, + "About Us", + ); + graph.add_edge( + &url_home, + &url_redirect, + seo_lens::graph::LinkEdgeType::Redirect, + false, + "Old Page", + ); + + let page_home = PageReport { + crawl_id: "crawl_test".into(), + url: url_home.clone(), + url_hash: 1, + final_url: Some(url_home.clone()), + status_code: 200, + content_type: "text/html; charset=utf-8".into(), + size_bytes: 4096, + ttfb_ms: 120, + crawl_depth: 0, + title: Some("Example Home, Company & Co.".to_string()), + title_length: 29, + meta_description: Some("Meta description with, commas and \"quotes\".".to_string()), + meta_desc_length: 44, + canonical_url: Some(url_home.clone()), + html_lang: Some("en".into()), + robots_flags: RobotsFlags::NONE, + is_sitemap_url: true, + is_internal: true, + h1_primary: Some("Welcome to Example".to_string()), + h1_count: 1, + word_count: 500, + links: vec![ + DiscoveredLink { + source_url: url_home.clone(), + target_url: url_about.clone(), + target_url_hash: 2, + anchor_text: "About Us".to_string(), + is_internal: true, + is_nofollow: false, + is_image_link: false, + is_target_blank: false, + has_opener_or_referrer: true, + status_code: Some(200), + }, + DiscoveredLink { + source_url: url_home.clone(), + target_url: "https://external-partner.org/partners".to_string(), + target_url_hash: 99, + anchor_text: "External Partner".to_string(), + is_internal: false, + is_nofollow: true, + is_image_link: false, + is_target_blank: true, + has_opener_or_referrer: true, + status_code: Some(200), + }, + ], + ..Default::default() + }; + + let page_about = PageReport { + crawl_id: "crawl_test".into(), + url: url_about.clone(), + url_hash: 2, + final_url: Some(url_about.clone()), + status_code: 200, + content_type: "text/html".into(), + size_bytes: 2048, + ttfb_ms: 85, + crawl_depth: 1, + title: Some("About Us - Example".to_string()), + title_length: 18, + meta_description: None, + meta_desc_length: 0, + canonical_url: Some(url_about.clone()), + html_lang: Some("en".into()), + robots_flags: RobotsFlags::NONE, + is_sitemap_url: true, + is_internal: true, + h1_primary: Some("About Us".to_string()), + h1_count: 1, + word_count: 320, + ..Default::default() + }; + + let page_redirect = PageReport { + crawl_id: "crawl_test".into(), + url: url_redirect.clone(), + url_hash: 3, + final_url: Some(url_home.clone()), + status_code: 301, + content_type: "text/html".into(), + size_bytes: 300, + ttfb_ms: 45, + crawl_depth: 1, + is_internal: true, + ..Default::default() + }; + + let page_noindex = PageReport { + crawl_id: "crawl_test".into(), + url: url_noindex.clone(), + url_hash: 4, + final_url: Some(url_noindex.clone()), + status_code: 200, + content_type: "text/html".into(), + size_bytes: 1500, + ttfb_ms: 90, + crawl_depth: 1, + title: Some("Privacy Policy".to_string()), + title_length: 14, + robots_flags: RobotsFlags::NOINDEX, + is_internal: true, + ..Default::default() + }; + + let issues = vec![ + IssueFinding { + code: RuleId::WarnMetaDescMissing, + category: IssueCategory::TitleMetadata, + severity: Severity::Warning, + title: "Missing Meta Description".into(), + message: "The page does not declare a meta description.".to_string(), + target_url: url_about.clone(), + source_page_url: Some(url_home.clone()), + }, + IssueFinding { + code: RuleId::AlertIndexingBlockedNoindex, + category: IssueCategory::Indexability, + severity: Severity::Alert, + title: "Noindex Directive Detected".into(), + message: "Page has noindex directive in robots meta.".to_string(), + target_url: url_noindex.clone(), + source_page_url: None, + }, + ]; + + CrawlResult { + target_url: "https://example.com/".to_string(), + pages: vec![page_home, page_about, page_redirect, page_noindex], + graph, + pagerank: HashMap::new(), + issues, + duration: Duration::from_secs(5), + sitemap_urls: vec![url_home.clone(), url_about.clone()], + aimd_delay_ms: 50, + health_score: 85, + } +} + +#[test] +fn test_export_csv_suite_generates_all_four_files() { + let result = create_test_crawl_result(); + let temp_dir = unique_test_csv_dir(); + + let exported = export_csv_suite(&result, &temp_dir).expect("Export CSV suite should succeed"); + + assert_eq!( + exported.len(), + 4, + "Should return 4 generated CSV file paths" + ); + + let internal_csv = temp_dir.join("csv").join("internal_all.csv"); + let issues_csv = temp_dir.join("csv").join("issues_all.csv"); + let response_codes_csv = temp_dir.join("csv").join("response_codes.csv"); + let external_csv = temp_dir.join("csv").join("external_all.csv"); + + assert!(internal_csv.exists(), "internal_all.csv must exist"); + assert!(issues_csv.exists(), "issues_all.csv must exist"); + assert!(response_codes_csv.exists(), "response_codes.csv must exist"); + assert!(external_csv.exists(), "external_all.csv must exist"); + + // 1. Validate internal_all.csv schema and contents + let internal_content = fs::read_to_string(&internal_csv).unwrap(); + let mut internal_lines = internal_content.lines(); + let internal_header = internal_lines.next().expect("Header line"); + assert_eq!( + internal_header, + "Address,Status Code,Status,Content Type,Size (Bytes),Word Count,Title 1,Title 1 Length,Meta Description 1,Meta Description 1 Length,H1-1,H1-1 Length,Canonical Link Element 1,Indexability,Indexability Status,Inlinks,Outlinks,Crawl Depth,Response Time (ms)" + ); + + // Verify comma and quote escaping on home page + assert!(internal_content.contains(r#""Example Home, Company & Co.""#)); + assert!(internal_content.contains(r#""Meta description with, commas and ""quotes"".""#)); + assert!(internal_content.contains("Indexable")); + assert!(internal_content.contains("Non-Indexable")); + + // 2. Validate issues_all.csv schema and contents + let issues_content = fs::read_to_string(&issues_csv).unwrap(); + let mut issues_lines = issues_content.lines(); + let issues_header = issues_lines.next().expect("Header line"); + assert_eq!( + issues_header, + "Issue Code,Issue Name,Severity,Category,URL,Source URL,Details,Recommendation" + ); + assert!(issues_content.contains("WARN_META_DESC_MISSING")); + assert!(issues_content.contains("ALERT_INDEXING_BLOCKED_NOINDEX")); + + // 3. Validate response_codes.csv schema and contents + let response_codes_content = fs::read_to_string(&response_codes_csv).unwrap(); + let mut response_codes_lines = response_codes_content.lines(); + let response_codes_header = response_codes_lines.next().expect("Header line"); + assert_eq!( + response_codes_header, + "URL,Status Code,Status,Redirect URL,Redirect Type,Inlinks Count" + ); + assert!(response_codes_content.contains("https://example.com/old-page")); + assert!(response_codes_content.contains("301")); + assert!(response_codes_content.contains("Permanent")); + + // 4. Validate external_all.csv schema and contents + let external_content = fs::read_to_string(&external_csv).unwrap(); + let mut external_lines = external_content.lines(); + let external_header = external_lines.next().expect("Header line"); + assert_eq!( + external_header, + "Source URL,Destination URL,Anchor Text,Status Code,Is Nofollow" + ); + assert!(external_content.contains("https://external-partner.org/partners")); + assert!(external_content.contains("External Partner")); + assert!(external_content.contains("true")); // is_nofollow + + // Cleanup + let _ = fs::remove_dir_all(&temp_dir); +} + +#[tokio::test] +async fn test_cli_report_command_generates_csv_suite() { + use seo_lens::cli::args::{Cli, Commands, ReportArgs}; + use seo_lens::cli::commands::execute; + use seo_lens::storage::{CrawlSessionInit, Database}; + + let temp_dir = unique_test_csv_dir(); + let db_path = temp_dir.join("test.db"); + let reports_dir = temp_dir.join("reports"); + let session_id = "crawl_csv_test_1"; + + let db = Database::open(&db_path).expect("Open test database"); + db.init_crawl_session(&CrawlSessionInit { + session_id: session_id.to_string(), + target_url: "https://example.com/".to_string(), + max_pages: 10, + max_depth: 2, + respect_robots: true, + render_js: false, + }) + .expect("Init session"); + + let (writer_handle, writer_task) = db + .spawn_writer(session_id, 1, Duration::from_millis(50)) + .expect("Spawn writer"); + + let test_page = PageReport { + crawl_id: session_id.into(), + url: "https://example.com/".to_string(), + url_hash: 1, + final_url: Some("https://example.com/".to_string()), + status_code: 200, + content_type: "text/html".into(), + size_bytes: 1024, + ttfb_ms: 50, + crawl_depth: 0, + title: Some("Example Home".to_string()), + title_length: 12, + h1_primary: Some("Welcome".to_string()), + h1_count: 1, + is_internal: true, + links: vec![DiscoveredLink { + source_url: "https://example.com/".to_string(), + target_url: "https://external.org/test".to_string(), + target_url_hash: 10, + anchor_text: "External Link".to_string(), + is_internal: false, + is_nofollow: true, + is_image_link: false, + is_target_blank: false, + has_opener_or_referrer: true, + status_code: Some(200), + }], + ..Default::default() + }; + + writer_handle.save_page(test_page).await.expect("Save page"); + writer_handle.flush().await.expect("Flush"); + db.update_crawl_status(session_id, "completed", None, 1, 0, 0, 1, Some(90)) + .expect("Update status"); + writer_handle.shutdown().await.expect("Shutdown"); + writer_task.await.expect("Join").expect("Result"); + + let report_cli = Cli { + command: Commands::Report(ReportArgs { + session: Some(session_id.to_string()), + session_pos: None, + format: Some("csv".to_string()), + output_dir: Some(reports_dir.clone()), + db_path: Some(db_path.clone()), + ..Default::default() + }), + }; + + execute(report_cli).await.expect("Report command execution"); + + assert!(reports_dir.join("csv").join("internal_all.csv").exists()); + assert!(reports_dir.join("csv").join("issues_all.csv").exists()); + assert!(reports_dir.join("csv").join("response_codes.csv").exists()); + assert!(reports_dir.join("csv").join("external_all.csv").exists()); + + // Cleanup + let _ = fs::remove_dir_all(&temp_dir); +} diff --git a/tests/html_export_tests.rs b/tests/html_export_tests.rs new file mode 100644 index 0000000..7aa544a --- /dev/null +++ b/tests/html_export_tests.rs @@ -0,0 +1,307 @@ +//! # Standalone HTML Report Exporter Integration Tests (Phase 11) +//! +//! Validates generation of self-contained, offline-ready HTML visual audit reports: +//! 1. Single-file self-contained HTML (all CSS and JS inlined). +//! 2. Zero external dependencies (no CDNs, no external fonts or trackers). +//! 3. Interactive components: live search bar, severity filter chips, issue accordions, pages table. +//! 4. CLI report command integration (`--format html`). + +use hashbrown::HashMap; +use seo_lens::core::models::{ + DiscoveredLink, IssueCategory, IssueFinding, PageReport, RobotsFlags, RuleId, Severity, +}; +use seo_lens::crawler::engine::CrawlResult; +use seo_lens::graph::SiteGraph; +use seo_lens::report::html::export_html_report; +use std::fs; +use std::path::PathBuf; +use std::sync::atomic::{AtomicUsize, Ordering}; +use std::time::Duration; + +static HTML_TEST_COUNTER: AtomicUsize = AtomicUsize::new(1); + +fn unique_test_html_dir() -> PathBuf { + let id = HTML_TEST_COUNTER.fetch_add(1, Ordering::SeqCst); + let mut dir = std::env::temp_dir(); + dir.push(format!("seolens_html_test_{}_{}", std::process::id(), id)); + dir +} + +fn create_test_crawl_result() -> CrawlResult { + let mut graph = SiteGraph::new(); + + let url_home = "https://example.com/".to_string(); + let url_about = "https://example.com/about".to_string(); + let url_broken = "https://example.com/broken-page".to_string(); + + graph.add_node(&url_home, 200, 0, true); + graph.add_node(&url_about, 200, 1, true); + graph.add_node(&url_broken, 404, 1, false); + + graph.add_edge( + &url_home, + &url_about, + seo_lens::graph::LinkEdgeType::InternalHyperlink, + false, + "About Us", + ); + graph.add_edge( + &url_home, + &url_broken, + seo_lens::graph::LinkEdgeType::InternalHyperlink, + false, + "Broken Page", + ); + + let page_home = PageReport { + crawl_id: "crawl_test".into(), + url: url_home.clone(), + url_hash: 1, + final_url: Some(url_home.clone()), + status_code: 200, + content_type: "text/html; charset=utf-8".into(), + size_bytes: 4096, + ttfb_ms: 120, + crawl_depth: 0, + title: Some("Example Home | Best Products".to_string()), + title_length: 28, + meta_description: Some("Meta description for example home page.".to_string()), + meta_desc_length: 39, + canonical_url: Some(url_home.clone()), + html_lang: Some("en".into()), + robots_flags: RobotsFlags::NONE, + is_sitemap_url: true, + is_internal: true, + h1_primary: Some("Welcome to Example".to_string()), + h1_count: 1, + word_count: 500, + links: vec![ + DiscoveredLink { + source_url: url_home.clone(), + target_url: url_about.clone(), + target_url_hash: 2, + anchor_text: "About Us".to_string(), + is_internal: true, + is_nofollow: false, + is_image_link: false, + is_target_blank: false, + has_opener_or_referrer: true, + status_code: Some(200), + }, + DiscoveredLink { + source_url: url_home.clone(), + target_url: url_broken.clone(), + target_url_hash: 3, + anchor_text: "Broken Page".to_string(), + is_internal: true, + is_nofollow: false, + is_image_link: false, + is_target_blank: false, + has_opener_or_referrer: true, + status_code: Some(404), + }, + ], + ..Default::default() + }; + + let page_about = PageReport { + crawl_id: "crawl_test".into(), + url: url_about.clone(), + url_hash: 2, + final_url: Some(url_about.clone()), + status_code: 200, + content_type: "text/html".into(), + size_bytes: 2048, + ttfb_ms: 85, + crawl_depth: 1, + title: Some("About Us - Example".to_string()), + title_length: 18, + meta_description: None, + meta_desc_length: 0, + canonical_url: Some(url_about.clone()), + html_lang: Some("en".into()), + robots_flags: RobotsFlags::NONE, + is_sitemap_url: true, + is_internal: true, + h1_primary: Some("About Us".to_string()), + h1_count: 1, + word_count: 320, + ..Default::default() + }; + + let page_broken = PageReport { + crawl_id: "crawl_test".into(), + url: url_broken.clone(), + url_hash: 3, + final_url: None, + status_code: 404, + content_type: "text/html".into(), + size_bytes: 500, + ttfb_ms: 60, + crawl_depth: 1, + is_internal: true, + ..Default::default() + }; + + let issues = vec![ + IssueFinding { + code: RuleId::ErrHttp4xxClientError, + category: IssueCategory::HttpTransport, + severity: Severity::Critical, + title: "HTTP 404 Client Error".into(), + message: "Page responded with 404 Not Found status code.".to_string(), + target_url: url_broken.clone(), + source_page_url: Some(url_home.clone()), + }, + IssueFinding { + code: RuleId::WarnMetaDescMissing, + category: IssueCategory::TitleMetadata, + severity: Severity::Warning, + title: "Missing Meta Description".into(), + message: "The page does not declare a meta description.".to_string(), + target_url: url_about.clone(), + source_page_url: Some(url_home.clone()), + }, + ]; + + CrawlResult { + target_url: "https://example.com/".to_string(), + pages: vec![page_home, page_about, page_broken], + graph, + pagerank: HashMap::new(), + issues, + duration: Duration::from_secs(3), + sitemap_urls: vec![url_home.clone(), url_about.clone()], + aimd_delay_ms: 50, + health_score: 72, + } +} + +#[test] +fn test_export_html_report_creates_standalone_offline_report() { + let result = create_test_crawl_result(); + let temp_dir = unique_test_html_dir(); + + let output_path = + export_html_report(&result, &temp_dir).expect("Export HTML report should succeed"); + + assert!(output_path.exists(), "HTML report file must exist"); + assert!(output_path + .extension() + .map(|s| s == "html") + .unwrap_or(false)); + + let html_content = fs::read_to_string(&output_path).expect("Read generated HTML"); + + // 1. Validate proper HTML5 structure + assert!(html_content.contains("")); + assert!(html_content.contains("")); + assert!(html_content.contains("")); + assert!(html_content.contains("")); + assert!(html_content.contains("")); + + // 2. Validate zero external CDN calls or tracking scripts + assert!(!html_content.contains("fonts.googleapis.com")); + assert!(!html_content.contains("cdnjs.cloudflare.com")); + assert!(!html_content.contains("unpkg.com")); + assert!(!html_content.contains("cdn.jsdelivr.net")); + assert!(!html_content.contains("google-analytics.com")); + + // 3. Validate embedded CSS and JavaScript + assert!(html_content.contains("")); + assert!(html_content.contains("")); + + // 4. Validate Audit Data Content + assert!(html_content.contains("https://example.com/")); + assert!(html_content.contains("72/100") || html_content.contains("72")); + assert!(html_content.contains("ERR_HTTP_4XX_CLIENT_ERROR")); + assert!(html_content.contains("WARN_META_DESC_MISSING")); + assert!(html_content.contains("https://example.com/broken-page")); + assert!(html_content.contains("https://example.com/about")); + + // 5. Validate Interactive UI components + assert!( + html_content.contains("id=\"issueSearch\"") || html_content.contains("id=\"pageSearch\"") + ); + assert!(html_content.contains("filterIssues") || html_content.contains("data-severity")); + assert!(html_content.contains("copyRemediation") || html_content.contains("Copy")); + + // Cleanup + let _ = fs::remove_dir_all(&temp_dir); +} + +#[tokio::test] +async fn test_cli_report_command_generates_html_report() { + use seo_lens::cli::args::{Cli, Commands, ReportArgs}; + use seo_lens::cli::commands::execute; + use seo_lens::storage::{CrawlSessionInit, Database}; + + let temp_dir = unique_test_html_dir(); + let db_path = temp_dir.join("test.db"); + let reports_dir = temp_dir.join("reports"); + let session_id = "crawl_html_test_1"; + + let db = Database::open(&db_path).expect("Open test database"); + db.init_crawl_session(&CrawlSessionInit { + session_id: session_id.to_string(), + target_url: "https://example.com/".to_string(), + max_pages: 5, + max_depth: 2, + respect_robots: true, + render_js: false, + }) + .expect("Init session"); + + let (writer_handle, writer_task) = db + .spawn_writer(session_id, 1, Duration::from_millis(50)) + .expect("Spawn writer"); + + let test_page = PageReport { + crawl_id: session_id.into(), + url: "https://example.com/".to_string(), + url_hash: 1, + final_url: Some("https://example.com/".to_string()), + status_code: 200, + content_type: "text/html".into(), + size_bytes: 1024, + ttfb_ms: 50, + crawl_depth: 0, + title: Some("Example Home".to_string()), + title_length: 12, + is_internal: true, + ..Default::default() + }; + + writer_handle.save_page(test_page).await.expect("Save page"); + writer_handle.flush().await.expect("Flush"); + db.update_crawl_status(session_id, "completed", None, 1, 0, 0, 1, Some(85)) + .expect("Update status"); + writer_handle.shutdown().await.expect("Shutdown"); + writer_task.await.expect("Join").expect("Result"); + + let report_cli = Cli { + command: Commands::Report(ReportArgs { + session: Some(session_id.to_string()), + session_pos: None, + format: Some("html".to_string()), + output_dir: Some(reports_dir.clone()), + db_path: Some(db_path.clone()), + ..Default::default() + }), + }; + + execute(report_cli).await.expect("Report command execution"); + + let expected_file = reports_dir.join("example_com_audit.html"); + assert!( + expected_file.exists(), + "example_com_audit.html must be generated" + ); + + // Cleanup + let _ = fs::remove_dir_all(&temp_dir); +} diff --git a/tests/mcp_tests.rs b/tests/mcp_tests.rs new file mode 100644 index 0000000..8afea39 --- /dev/null +++ b/tests/mcp_tests.rs @@ -0,0 +1,629 @@ +//! # Model Context Protocol (MCP) Integration Tests +//! +//! Validates the native JSON-RPC 2.0 stdio MCP server: +//! - Protocol negotiation (`initialize`, `ping`). +//! - Complete catalog of 8 core tools (`tools/list`). +//! - Synchronous single-page auditing (`seo_quick_page_check`). +//! - Non-blocking asynchronous audit start (`seo_start_audit`) & telemetry polling (`seo_audit_status`). +//! - Token-efficient Markdown report generation (`seo_get_markdown_report`) with zero ANSI codes. +//! - Filtered issue queries (`seo_query_issues`). +//! - Generative Engine Optimization readiness (`seo_check_ai_readiness`). +//! - Schema.org Rich Results validation (`seo_validate_schema`). +//! - Session cleanup (`seo_cleanup_session`). +//! - MCP Resources (`resources/list`, `resources/read`). + +use seo_lens::core::models::{IssueCategory, IssueFinding, RuleId, Severity}; +use seo_lens::mcp::protocol::{handle_jsonrpc_request, McpContext}; +use seo_lens::storage::{CrawlSessionInit, Database}; +use std::path::PathBuf; +use std::sync::atomic::{AtomicUsize, Ordering}; +use std::sync::Arc; +use wiremock::matchers::{method, path}; +use wiremock::{Mock, MockServer, ResponseTemplate}; + +static MCP_TEST_COUNTER: AtomicUsize = AtomicUsize::new(2000); + +fn unique_test_db_path() -> PathBuf { + let id = MCP_TEST_COUNTER.fetch_add(1, Ordering::SeqCst); + let mut p = std::env::temp_dir(); + p.push(format!("seolens_mcp_test_{}_{}.db", std::process::id(), id)); + p +} + +#[tokio::test] +async fn test_mcp_initialize_and_ping() { + let db_path = unique_test_db_path(); + let ctx = Arc::new(McpContext::new(Some(db_path))); + + // 1. Initialize request + let init_req = serde_json::json!({ + "jsonrpc": "2.0", + "id": 1, + "method": "initialize", + "params": { + "protocolVersion": "2024-11-05", + "capabilities": {}, + "clientInfo": { + "name": "test-agent", + "version": "1.0.0" + } + } + }); + + let resp_str = handle_jsonrpc_request(&init_req.to_string(), &ctx).await; + let resp: serde_json::Value = serde_json::from_str(&resp_str).expect("Valid JSON response"); + + assert_eq!(resp["jsonrpc"], "2.0"); + assert_eq!(resp["id"], 1); + assert_eq!(resp["result"]["protocolVersion"], "2024-11-05"); + assert_eq!(resp["result"]["serverInfo"]["name"], "seolens"); + assert!(resp["result"]["capabilities"]["tools"].is_object()); + assert!(resp["result"]["capabilities"]["resources"].is_object()); + + // 2. Ping request + let ping_req = serde_json::json!({ + "jsonrpc": "2.0", + "id": 2, + "method": "ping" + }); + + let ping_resp_str = handle_jsonrpc_request(&ping_req.to_string(), &ctx).await; + let ping_resp: serde_json::Value = + serde_json::from_str(&ping_resp_str).expect("Valid JSON response"); + assert_eq!(ping_resp["jsonrpc"], "2.0"); + assert_eq!(ping_resp["id"], 2); + assert_eq!(ping_resp["result"], serde_json::json!({})); +} + +#[tokio::test] +async fn test_mcp_tools_list_schema() { + let db_path = unique_test_db_path(); + let ctx = Arc::new(McpContext::new(Some(db_path))); + + let req = serde_json::json!({ + "jsonrpc": "2.0", + "id": 10, + "method": "tools/list" + }); + + let resp_str = handle_jsonrpc_request(&req.to_string(), &ctx).await; + let resp: serde_json::Value = serde_json::from_str(&resp_str).expect("Valid JSON response"); + + let tools = resp["result"]["tools"].as_array().expect("Tools array"); + assert_eq!(tools.len(), 8, "Must expose exactly 8 core MCP tools"); + + let tool_names: Vec<&str> = tools + .iter() + .map(|t| t["name"].as_str().expect("Tool name")) + .collect(); + + assert!(tool_names.contains(&"seo_start_audit")); + assert!(tool_names.contains(&"seo_audit_status")); + assert!(tool_names.contains(&"seo_get_markdown_report")); + assert!(tool_names.contains(&"seo_quick_page_check")); + assert!(tool_names.contains(&"seo_query_issues")); + assert!(tool_names.contains(&"seo_check_ai_readiness")); + assert!(tool_names.contains(&"seo_validate_schema")); + assert!(tool_names.contains(&"seo_cleanup_session")); + + // Verify schemas have inputSchema with properties + for tool in tools { + assert!(tool["inputSchema"]["type"] == "object"); + assert!(tool["description"].is_string()); + } +} + +#[tokio::test] +async fn test_mcp_quick_page_check_sync() { + let server = MockServer::start().await; + let html = r#" + + + Pricing Plans - Fast SaaS + + + + +

Pricing

+

Sign up today and scale with us.

+ + +"#; + + Mock::given(method("GET")) + .and(path("/pricing")) + .respond_with( + ResponseTemplate::new(200) + .set_body_string(html) + .insert_header("content-type", "text/html"), + ) + .mount(&server) + .await; + + let db_path = unique_test_db_path(); + let ctx = Arc::new(McpContext::new(Some(db_path))); + + let test_url = format!("{}/pricing", server.uri()); + let req = serde_json::json!({ + "jsonrpc": "2.0", + "id": 20, + "method": "tools/call", + "params": { + "name": "seo_quick_page_check", + "arguments": { + "url": test_url + } + } + }); + + let resp_str = handle_jsonrpc_request(&req.to_string(), &ctx).await; + let resp: serde_json::Value = serde_json::from_str(&resp_str).expect("Valid JSON response"); + assert_eq!(resp["id"], 20); + + let content_text = resp["result"]["content"][0]["text"] + .as_str() + .expect("Content text"); + let result_obj: serde_json::Value = + serde_json::from_str(content_text).expect("Parsed tool JSON"); + + assert_eq!(result_obj["status_code"], 200); + assert_eq!(result_obj["title"], "Pricing Plans - Fast SaaS"); + assert_eq!( + result_obj["meta_description"], + "Transparent pricing for engineering teams." + ); + assert_eq!(result_obj["h1"], "Pricing"); + assert_eq!(result_obj["canonical_url"], "https://example.com/pricing"); + assert_eq!(result_obj["is_indexable"], true); + + // Image missing alt attribute should be flagged + let issues = result_obj["issues_detected"] + .as_array() + .expect("Issues array"); + let has_img_alt_issue = issues + .iter() + .any(|i| i["code"].as_str() == Some("WARN_IMAGE_MISSING_ALT")); + assert!( + has_img_alt_issue, + "Should detect missing alt tag on /logo.png" + ); +} + +#[tokio::test] +async fn test_mcp_start_audit_non_blocking_and_status() { + let server = MockServer::start().await; + let html = + r#"Root

Root

"#; + Mock::given(method("GET")) + .and(path("/")) + .respond_with(ResponseTemplate::new(200).set_body_string(html)) + .mount(&server) + .await; + + let db_path = unique_test_db_path(); + let ctx = Arc::new(McpContext::new(Some(db_path.clone()))); + + // 1. Start audit + let start_req = serde_json::json!({ + "jsonrpc": "2.0", + "id": 30, + "method": "tools/call", + "params": { + "name": "seo_start_audit", + "arguments": { + "url": server.uri(), + "max_pages": 10, + "max_depth": 2 + } + } + }); + + let start_time = std::time::Instant::now(); + let resp_str = handle_jsonrpc_request(&start_req.to_string(), &ctx).await; + let elapsed = start_time.elapsed(); + + // Must return in under 1.0 second (non-blocking async guarantee) + assert!( + elapsed.as_millis() < 1000, + "seo_start_audit took {}ms, exceeding 1000ms SLA", + elapsed.as_millis() + ); + + let resp: serde_json::Value = serde_json::from_str(&resp_str).expect("Valid JSON response"); + let content_text = resp["result"]["content"][0]["text"] + .as_str() + .expect("Content text"); + let start_res: serde_json::Value = + serde_json::from_str(content_text).expect("Parsed tool JSON"); + + let session_id = start_res["session_id"] + .as_str() + .expect("Session ID string") + .to_string(); + assert_eq!(start_res["status"], "queued"); + assert_eq!(start_res["target_url"], server.uri()); + assert_eq!(start_res["poll_interval_seconds"], 15); + + // 2. Poll audit status + let status_req = serde_json::json!({ + "jsonrpc": "2.0", + "id": 31, + "method": "tools/call", + "params": { + "name": "seo_audit_status", + "arguments": { + "session_id": session_id + } + } + }); + + let status_resp_str = handle_jsonrpc_request(&status_req.to_string(), &ctx).await; + let status_resp: serde_json::Value = + serde_json::from_str(&status_resp_str).expect("Valid JSON response"); + let status_text = status_resp["result"]["content"][0]["text"] + .as_str() + .expect("Content text"); + let status_res: serde_json::Value = + serde_json::from_str(status_text).expect("Parsed status JSON"); + + assert_eq!(status_res["session_id"], session_id); + assert!(status_res["status"].is_string()); + assert!(status_res["issues_count"].is_object()); +} + +#[tokio::test] +async fn test_mcp_get_markdown_report_token_efficiency_zero_ansi() { + let db_path = unique_test_db_path(); + let db = Database::open(&db_path).expect("Open test database"); + let session = "crawl_mcp_token_test"; + + // Setup crawl session and issues + db.init_crawl_session(&CrawlSessionInit { + session_id: session.to_string(), + target_url: "https://agent-efficiency.test".to_string(), + max_pages: 50, + max_depth: 3, + respect_robots: true, + render_js: false, + }) + .expect("Init session"); + + let issues = vec![ + IssueFinding { + code: RuleId::ErrHttp4xxClientError, + category: IssueCategory::Indexability, + severity: Severity::Critical, + title: "Dead Internal 404 Page".into(), + message: "HTTP 404 Not Found returned".into(), + target_url: "https://agent-efficiency.test/dead-link".into(), + source_page_url: Some("https://agent-efficiency.test/home".into()), + }, + IssueFinding { + code: RuleId::WarnTitleTooLong, + category: IssueCategory::TitleMetadata, + severity: Severity::Warning, + title: "Title Exceeds 60 Characters".into(), + message: "Length is 82 chars".into(), + target_url: "https://agent-efficiency.test/product".into(), + source_page_url: None, + }, + ]; + db.save_issue_batch(session, &issues).expect("Save issues"); + + let ctx = Arc::new(McpContext::new(Some(db_path))); + + let req = serde_json::json!({ + "jsonrpc": "2.0", + "id": 40, + "method": "tools/call", + "params": { + "name": "seo_get_markdown_report", + "arguments": { + "session_id": session, + "top_issues_limit": 10, + "include_urls": true + } + } + }); + + let resp_str = handle_jsonrpc_request(&req.to_string(), &ctx).await; + let resp: serde_json::Value = serde_json::from_str(&resp_str).expect("Valid JSON response"); + let md_report = resp["result"]["content"][0]["text"] + .as_str() + .expect("Report markdown"); + + // Strict LLM Token-Efficiency Guarantees: + // 1. Zero ANSI escape sequences + assert!( + !md_report.contains("\x1b["), + "LLM report must contain ZERO ANSI escape codes" + ); + // 2. Zero ASCII-art banners + assert!( + !md_report.contains("β–ˆβ–ˆ"), + "LLM report must not contain decorative ASCII art banners" + ); + // 3. Clear Markdown structure + assert!(md_report.contains("# Technical SEO Audit")); + assert!(md_report.contains("ERR_HTTP_4XX_CLIENT_ERROR")); + assert!(md_report.contains("WARN_TITLE_TOO_LONG")); + assert!(md_report.contains("Action for Agent")); +} + +#[tokio::test] +async fn test_mcp_query_issues_and_cleanup() { + let db_path = unique_test_db_path(); + let db = Database::open(&db_path).expect("Open test database"); + let session = "crawl_mcp_cleanup_test"; + + db.init_crawl_session(&CrawlSessionInit { + session_id: session.to_string(), + target_url: "https://cleanup.test".to_string(), + max_pages: 10, + max_depth: 2, + respect_robots: true, + render_js: false, + }) + .expect("Init session"); + + let issues = vec![IssueFinding { + code: RuleId::ErrHttp5xxServerError, + category: IssueCategory::HttpTransport, + severity: Severity::Critical, + title: "Server 500 Error".into(), + message: "Internal Error".into(), + target_url: "https://cleanup.test/api/fail".into(), + source_page_url: None, + }]; + db.save_issue_batch(session, &issues).expect("Save issues"); + + let ctx = Arc::new(McpContext::new(Some(db_path.clone()))); + + // 1. Query issues + let query_req = serde_json::json!({ + "jsonrpc": "2.0", + "id": 50, + "method": "tools/call", + "params": { + "name": "seo_query_issues", + "arguments": { + "session_id": session, + "severity": "critical" + } + } + }); + + let resp_str = handle_jsonrpc_request(&query_req.to_string(), &ctx).await; + let resp: serde_json::Value = serde_json::from_str(&resp_str).expect("Valid JSON response"); + let text = resp["result"]["content"][0]["text"] + .as_str() + .expect("Content text"); + let query_res: serde_json::Value = serde_json::from_str(text).expect("Parsed query JSON"); + + assert_eq!(query_res["total_matching"], 1); + assert_eq!(query_res["issues"][0]["code"], "ERR_HTTP_5XX_SERVER_ERROR"); + + // 2. Cleanup session + let cleanup_req = serde_json::json!({ + "jsonrpc": "2.0", + "id": 51, + "method": "tools/call", + "params": { + "name": "seo_cleanup_session", + "arguments": { + "session_id": session + } + } + }); + + let clean_resp_str = handle_jsonrpc_request(&cleanup_req.to_string(), &ctx).await; + let clean_resp: serde_json::Value = + serde_json::from_str(&clean_resp_str).expect("Valid JSON response"); + let clean_text = clean_resp["result"]["content"][0]["text"] + .as_str() + .expect("Content text"); + let clean_res: serde_json::Value = serde_json::from_str(clean_text).expect("Parsed clean JSON"); + + assert_eq!(clean_res["session_id"], session); + assert_eq!(clean_res["purged"], true); + + // Verify session no longer exists in DB + let verify_db = Database::open(&db_path).expect("Reopen db"); + let crawl = verify_db.get_crawl(session).expect("Query deleted crawl"); + assert!( + crawl.is_none(), + "Crawl session should be purged from SQLite" + ); +} + +#[tokio::test] +async fn test_mcp_check_ai_and_validate_schema_tools() { + let server = MockServer::start().await; + let robots_txt = "User-agent: PerplexityBot\nDisallow: /\n\nUser-agent: GPTBot\nDisallow: /"; + Mock::given(method("GET")) + .and(path("/robots.txt")) + .respond_with(ResponseTemplate::new(200).set_body_string(robots_txt)) + .mount(&server) + .await; + Mock::given(method("GET")) + .and(path("/llms.txt")) + .respond_with(ResponseTemplate::new(200).set_body_string("# LLMS.txt content")) + .mount(&server) + .await; + Mock::given(method("GET")) + .and(path("/llms-full.txt")) + .respond_with(ResponseTemplate::new(404)) + .mount(&server) + .await; + + let db_path = unique_test_db_path(); + let ctx = Arc::new(McpContext::new(Some(db_path))); + + // 1. Test seo_check_ai_readiness + let ai_req = serde_json::json!({ + "jsonrpc": "2.0", + "id": 60, + "method": "tools/call", + "params": { + "name": "seo_check_ai_readiness", + "arguments": { + "url": server.uri() + } + } + }); + + let resp_str = handle_jsonrpc_request(&ai_req.to_string(), &ctx).await; + let resp: serde_json::Value = serde_json::from_str(&resp_str).expect("Valid JSON response"); + let text = resp["result"]["content"][0]["text"] + .as_str() + .expect("Content text"); + let ai_res: serde_json::Value = serde_json::from_str(text).expect("Parsed AI check JSON"); + + assert_eq!(ai_res["llms_txt_found"], true); + assert_eq!(ai_res["llms_full_txt_found"], false); + assert_eq!( + ai_res["ai_crawler_access"]["retrieval_citation_bots"]["PerplexityBot"], + "DISALLOWED" + ); + + // 2. Test seo_validate_schema + let valid_json_ld = serde_json::json!({ + "@context": "https://schema.org", + "@type": "Product", + "name": "Developer Mechanical Keyboard", + "offers": { + "@type": "Offer", + "price": "149.99", + "priceCurrency": "USD" + } + }); + + let schema_req = serde_json::json!({ + "jsonrpc": "2.0", + "id": 61, + "method": "tools/call", + "params": { + "name": "seo_validate_schema", + "arguments": { + "json_ld": valid_json_ld.to_string(), + "target_type": "Product" + } + } + }); + + let schema_resp_str = handle_jsonrpc_request(&schema_req.to_string(), &ctx).await; + let schema_resp: serde_json::Value = + serde_json::from_str(&schema_resp_str).expect("Valid JSON response"); + let schema_text = schema_resp["result"]["content"][0]["text"] + .as_str() + .expect("Content text"); + let schema_res: serde_json::Value = + serde_json::from_str(schema_text).expect("Parsed schema JSON"); + + assert_eq!(schema_res["is_valid_json"], true); + assert_eq!(schema_res["detected_type"], "Product"); + assert_eq!(schema_res["is_rich_result_eligible"], true); +} + +#[tokio::test] +async fn test_mcp_resources_list_and_read() { + let db_path = unique_test_db_path(); + let db = Database::open(&db_path).expect("Open test database"); + let session = "crawl_mcp_resource_test"; + + db.init_crawl_session(&CrawlSessionInit { + session_id: session.to_string(), + target_url: "https://resource-test.com".to_string(), + max_pages: 5, + max_depth: 2, + respect_robots: true, + render_js: false, + }) + .expect("Init session"); + + let ctx = Arc::new(McpContext::new(Some(db_path))); + + // 1. resources/list + let list_req = serde_json::json!({ + "jsonrpc": "2.0", + "id": 70, + "method": "resources/list" + }); + + let resp_str = handle_jsonrpc_request(&list_req.to_string(), &ctx).await; + let resp: serde_json::Value = serde_json::from_str(&resp_str).expect("Valid JSON response"); + let resources = resp["result"]["resources"] + .as_array() + .expect("Resources array"); + + let uris: Vec<&str> = resources + .iter() + .map(|r| r["uri"].as_str().expect("URI")) + .collect(); + assert!(uris.contains(&"seo://crawls")); + + // 2. resources/read seo://crawls + let read_req = serde_json::json!({ + "jsonrpc": "2.0", + "id": 71, + "method": "resources/read", + "params": { + "uri": "seo://crawls" + } + }); + + let read_resp_str = handle_jsonrpc_request(&read_req.to_string(), &ctx).await; + let read_resp: serde_json::Value = + serde_json::from_str(&read_resp_str).expect("Valid JSON response"); + let content = read_resp["result"]["contents"][0]["text"] + .as_str() + .expect("Content text"); + let sessions: Vec = + serde_json::from_str(content).expect("Parsed sessions array"); + assert!(sessions.iter().any(|s| s["session_id"] == session)); + + // 3. resources/read seo://crawls/{id}/summary + let summary_uri = format!("seo://crawls/{session}/summary"); + let summary_req = serde_json::json!({ + "jsonrpc": "2.0", + "id": 72, + "method": "resources/read", + "params": { + "uri": summary_uri + } + }); + + let summary_resp_str = handle_jsonrpc_request(&summary_req.to_string(), &ctx).await; + let summary_resp: serde_json::Value = + serde_json::from_str(&summary_resp_str).expect("Valid JSON response"); + let summary_content = summary_resp["result"]["contents"][0]["text"] + .as_str() + .expect("Summary text"); + let summary_obj: serde_json::Value = + serde_json::from_str(summary_content).expect("Parsed summary"); + assert_eq!(summary_obj["session_id"], session); + assert_eq!(summary_obj["target_url"], "https://resource-test.com"); +} + +#[tokio::test] +async fn test_mcp_server_io_stream() { + let db_path = unique_test_db_path(); + let ping_req = serde_json::json!({ + "jsonrpc": "2.0", + "id": 99, + "method": "ping" + }); + let input_bytes = format!("{}\n", ping_req); + let reader = tokio::io::BufReader::new(input_bytes.as_bytes()); + let mut output_bytes = Vec::new(); + + seo_lens::mcp::run_mcp_server_io(reader, &mut output_bytes, Some(db_path)) + .await + .expect("Run MCP server IO"); + + let output_str = String::from_utf8(output_bytes).expect("Valid UTF-8 output"); + let resp: serde_json::Value = + serde_json::from_str(output_str.trim()).expect("Valid JSON response"); + assert_eq!(resp["id"], 99); + assert_eq!(resp["result"], serde_json::json!({})); +} diff --git a/tests/storage_tests.rs b/tests/storage_tests.rs index 2b2469d..bdec65d 100644 --- a/tests/storage_tests.rs +++ b/tests/storage_tests.rs @@ -415,6 +415,7 @@ async fn test_cli_commands_list_and_report() { let list_cli = Cli { command: Commands::List(ListArgs { db_path: Some(db_path.clone()), + ..Default::default() }), }; execute(list_cli).await.expect("List command execution"); @@ -428,6 +429,7 @@ async fn test_cli_commands_list_and_report() { format: Some("json".to_string()), output_dir: Some(temp_reports.clone()), db_path: Some(db_path.clone()), + ..Default::default() }), }; execute(report_cli).await.expect("Report command execution"); @@ -439,3 +441,43 @@ async fn test_cli_commands_list_and_report() { let _ = std::fs::remove_dir_all(&temp_reports); let _ = std::fs::remove_file(&db_path); } + +#[test] +fn test_database_path_resolution() { + use seo_lens::storage::{default_db_path, local_db_path, resolve_db_path}; + + // 1. local_db_path returns .seolens/seolens.db + let local = local_db_path(); + assert_eq!(local, PathBuf::from(".seolens").join("seolens.db")); + + // 2. Explicit path takes highest precedence + let explicit = PathBuf::from("/custom/db/path.sqlite"); + assert_eq!(resolve_db_path(Some(explicit.clone()), false), explicit); + assert_eq!(resolve_db_path(Some(explicit.clone()), true), explicit); + + // 3. Local flag forces local_db_path when no explicit path given + assert_eq!(resolve_db_path(None, true), local); + + // 4. SEOLENS_DB_PATH environment variable override + let orig_env = std::env::var("SEOLENS_DB_PATH").ok(); + let temp_env_path = std::env::temp_dir().join("test_env_seolens.db"); + std::env::set_var("SEOLENS_DB_PATH", temp_env_path.to_str().unwrap()); + + assert_eq!(default_db_path(), temp_env_path); + assert_eq!(resolve_db_path(None, false), temp_env_path); + + // Restore or unset env var + match orig_env { + Some(val) => std::env::set_var("SEOLENS_DB_PATH", val), + None => std::env::remove_var("SEOLENS_DB_PATH"), + } + + // 5. Without env var, standard resolution returns OS data dir (if it exists) + if std::env::var("SEOLENS_DB_PATH").is_err() && !local.exists() { + if let Some(expected_os_dir) = dirs::data_dir() { + let expected = expected_os_dir.join("seolens").join("seolens.db"); + assert_eq!(default_db_path(), expected); + assert_eq!(resolve_db_path(None, false), expected); + } + } +}