From 370173cd05d7c4c4b52c746bacb7704b746150b4 Mon Sep 17 00:00:00 2001 From: Alok Kumar Date: Sun, 30 Aug 2026 10:49:22 +0530 Subject: [PATCH] Complete Backlink Intelligence v1.0.0 --- .github/ISSUE_TEMPLATE/bug_report.yml | 2 +- .github/workflows/ci.yml | 10 +- CHANGELOG.md | 37 ++- CONTRIBUTING.md | 2 +- Dockerfile | 6 + README.md | 397 +++++++++++++++---------- SECURITY.md | 24 +- backlink_intelligence/__init__.py | 2 +- backlink_intelligence/audit.py | 51 ++++ backlink_intelligence/cli.py | 88 ++++-- backlink_intelligence/fetcher.py | 71 +++++ backlink_intelligence/html_utils.py | 146 +++++++++ backlink_intelligence/link_analysis.py | 56 ++++ backlink_intelligence/models.py | 138 +++++++++ backlink_intelligence/monitor.py | 59 ++++ backlink_intelligence/placement.py | 84 ++++++ backlink_intelligence/portfolio.py | 28 ++ backlink_intelligence/qualify.py | 64 ++++ backlink_intelligence/relevance.py | 57 ++++ backlink_intelligence/reporting.py | 26 ++ backlink_intelligence/safety.py | 57 ++++ docs/architecture.md | 138 ++------- docs/ethical-crawling.md | 31 +- docs/limitations.md | 34 +-- docs/methodology.md | 98 ++---- docs/roadmap.md | 213 +++---------- docs/signal-definitions.md | 97 ++---- examples/README.md | 41 ++- pyproject.toml | 11 +- sample-data/backlinks.csv | 3 + sample-data/links.csv | 2 + sample-data/prospects.csv | 2 +- tests/test_audit.py | 22 ++ tests/test_cli.py | 9 +- tests/test_html_utils.py | 26 ++ tests/test_link_analysis.py | 16 + tests/test_monitor.py | 18 ++ tests/test_placement.py | 23 ++ tests/test_portfolio.py | 20 ++ tests/test_qualify.py | 16 + tests/test_relevance.py | 18 ++ tests/test_safety.py | 23 ++ 42 files changed, 1579 insertions(+), 687 deletions(-) create mode 100644 Dockerfile create mode 100644 backlink_intelligence/audit.py create mode 100644 backlink_intelligence/fetcher.py create mode 100644 backlink_intelligence/html_utils.py create mode 100644 backlink_intelligence/link_analysis.py create mode 100644 backlink_intelligence/models.py create mode 100644 backlink_intelligence/monitor.py create mode 100644 backlink_intelligence/placement.py create mode 100644 backlink_intelligence/portfolio.py create mode 100644 backlink_intelligence/qualify.py create mode 100644 backlink_intelligence/relevance.py create mode 100644 backlink_intelligence/reporting.py create mode 100644 backlink_intelligence/safety.py create mode 100644 sample-data/backlinks.csv create mode 100644 sample-data/links.csv create mode 100644 tests/test_audit.py create mode 100644 tests/test_html_utils.py create mode 100644 tests/test_link_analysis.py create mode 100644 tests/test_monitor.py create mode 100644 tests/test_placement.py create mode 100644 tests/test_portfolio.py create mode 100644 tests/test_qualify.py create mode 100644 tests/test_relevance.py create mode 100644 tests/test_safety.py diff --git a/.github/ISSUE_TEMPLATE/bug_report.yml b/.github/ISSUE_TEMPLATE/bug_report.yml index 1645daf..ffba7dc 100644 --- a/.github/ISSUE_TEMPLATE/bug_report.yml +++ b/.github/ISSUE_TEMPLATE/bug_report.yml @@ -24,7 +24,7 @@ body: id: version attributes: label: Version - placeholder: e.g. 0.1.0 + placeholder: e.g. 1.0.0 - type: input id: python attributes: diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 0109dab..51212b6 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -14,7 +14,7 @@ jobs: runs-on: ubuntu-latest strategy: matrix: - python-version: ["3.11", "3.12"] + python-version: ["3.11", "3.12", "3.13"] steps: - name: Checkout uses: actions/checkout@v4 @@ -23,6 +23,12 @@ jobs: with: python-version: ${{ matrix.python-version }} - name: Install package - run: python -m pip install -e . + run: python -m pip install . + - name: Compile package + run: python -m compileall -q backlink_intelligence - name: Run tests run: python -m unittest discover -s tests -v + - name: CLI smoke test + run: | + backlink-intelligence --version + backlink-intelligence status diff --git a/CHANGELOG.md b/CHANGELOG.md index 4bfb408..e9172cd 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,16 +1,33 @@ # Changelog -All notable changes to Backlink Intelligence will be documented here. +All notable changes to Backlink Intelligence are documented here. -The project follows semantic versioning once functional releases begin. - -## [Unreleased] +## 1.0.0 - 2026-08-30 ### Added -- Initial repository foundation -- Project methodology and architecture documentation -- Public release roadmap from `v0.1.0` to `v1.0.0` -- Minimal installable package and CLI status command -- Contribution, security, crawl-safety, and limitation guidance -- CI workflow for the foundation package +- Existing backlink evidence auditing. +- Safe bounded HTTP/HTTPS fetcher with private-network protections. +- HTML metadata, paragraph, heading, and link extraction. +- Contextual placement classification for main/editorial content, navigation, sidebar, footer, and unknown locations. +- Deterministic page and context relevance analysis. +- Outbound-link density and external-domain evidence. +- Bulk prospect qualification from CSV. +- Contextual link placement ranking. +- Before/After placement recommendations with editorial-preservation indicators. +- Backlink monitoring with persisted JSON baselines and change detection. +- Backlink portfolio analysis for anchors, destinations, and placements. +- Human-readable and JSON audit reporting. +- CLI commands: `audit`, `qualify`, `place`, `monitor`, `portfolio`, and `status`. +- Offline unit-test suite and GitHub Actions CI. + +### Design principles + +- No paid SEO API required. +- No paid AI/model API required. +- Evidence-first output instead of an unexplained universal backlink score. +- Human review required for placement drafts and workflow recommendations. + +## 0.0.1 - 2026-08-30 + +- Initial repository foundation, methodology, roadmap, contribution guidance, and CLI scaffold. diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 201256c..a4f8c4c 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -19,7 +19,7 @@ Contributions should preserve the project's core principles: git clone https://github.com/alok-vibe-code/backlink-intelligence.git cd backlink-intelligence python -m venv .venv -source .venv/bin/activate # Windows: .venv\\Scripts\\activate +source .venv/bin/activate # Windows: .venv\Scripts\activate python -m pip install -e . python -m unittest discover -s tests -v ``` diff --git a/Dockerfile b/Dockerfile new file mode 100644 index 0000000..7c1d795 --- /dev/null +++ b/Dockerfile @@ -0,0 +1,6 @@ +FROM python:3.12-slim +WORKDIR /app +COPY . . +RUN python -m pip install --no-cache-dir . +ENTRYPOINT ["backlink-intelligence"] +CMD ["--help"] diff --git a/README.md b/README.md index ff5c106..09d4a0a 100644 --- a/README.md +++ b/README.md @@ -2,243 +2,332 @@ **Open-source backlink intelligence based on evidence, context, and editorial fit.** -Backlink Intelligence is a local-first, open-source SEO toolkit for evaluating backlink opportunities without reducing link quality to a single authority metric. +Backlink Intelligence is a local-first Python toolkit for SEO professionals who want to evaluate backlink opportunities beyond a single domain-level authority metric. -The project is designed around a practical workflow: +It covers the complete working cycle: -**Discover → Qualify → Place → Monitor** +**Qualify → Audit → Place → Monitor → Analyze** -It will help SEO professionals inspect page-level evidence, evaluate topical and contextual relevance, identify natural link-placement opportunities, generate transparent Before/After placement recommendations, and monitor acquired links for later changes. +The core tool works without paid SEO APIs or paid AI APIs. -> **Project status:** Foundation / pre-alpha. The repository architecture and methodology are live. Functional backlink auditing begins with `v0.1.0`. +## What it does + +| Command | Purpose | Status | +| --- | --- | --- | +| `audit` | Inspect an existing backlink and its page-level evidence | ✅ Available | +| `qualify` | Evaluate backlink prospects from CSV | ✅ Available | +| `place` | Find contextual placement opportunities and produce Before/After suggestions | ✅ Available | +| `monitor` | Detect changes to acquired backlinks over time | ✅ Available | +| `portfolio` | Review anchor, destination, and placement distributions | ✅ Available | ## Why this project exists -Domain-level metrics can be useful inputs, but they do not explain the full context of an individual backlink opportunity. A link also has a source page, a destination, an anchor, a surrounding paragraph, a placement type, link attributes, crawl/indexability signals, and an outbound-link neighborhood. +DA, DR, and similar third-party metrics can be useful inputs, but they do not describe the full quality of an individual link placement. -Backlink Intelligence is being built to surface that evidence directly. +A backlink also has: -The project does **not** attempt to reproduce Google's ranking systems, label a link as objectively "good" or "bad," or claim that any individual signal determines search performance. +- a source page and destination page, +- topical and contextual alignment, +- an anchor, +- a surrounding paragraph, +- a placement type, +- link attributes, +- crawl/indexability evidence, +- and an outbound-link neighborhood. -Instead, it aims to answer questions such as: +Backlink Intelligence surfaces those observable signals so an SEO can make a more informed decision. -- Is the source page topically aligned with the destination? -- Does the target page genuinely expand the surrounding discussion? -- Where does the backlink appear on the page? -- Is the anchor natural in context? -- Does the source page show unusual outbound-link patterns that deserve review? -- Where could a target link fit naturally in an existing article? -- How much editorial rewriting would that placement require? -- Did an acquired backlink later disappear or change? +It does **not** claim to reproduce Google's ranking systems, predict penalties, or provide a universal "Google backlink score." -## Core workflow +## Installation -| Stage | Question | Planned capabilities | -| --- | --- | --- | -| **Discover** | Which pages deserve attention? | Bulk prospect input and future discovery helpers | -| **Qualify** | Is this opportunity worth pursuing? | Relevance, destination fit, source-page evidence, outbound-link analysis, review flags | -| **Place** | Where can the link naturally fit? | Paragraph ranking, anchor analysis, contextual placement, Before/After recommendations | -| **Monitor** | Did the placement remain intact? | Link existence, anchor/attribute changes, redirects, indexability changes, historical snapshots | +Python 3.11+ is required. + +```bash +python -m pip install . +``` + +For development: + +```bash +git clone https://github.com/alok-vibe-code/backlink-intelligence.git +cd backlink-intelligence +python -m pip install -e . +``` + +Verify the installation: + +```bash +backlink-intelligence --version +backlink-intelligence status +``` -## Signature feature: contextual placement recommendations +## 1. Audit an existing backlink -The planned placement engine accepts: +```bash +backlink-intelligence audit \ + "https://publisher.example/article" \ + "https://brand.example/target-page" +``` + +Typical evidence includes: + +- source/target HTTP status, +- backlink found/not found, +- anchor text, +- `nofollow`, `sponsored`, and `ugc` attributes, +- placement classification, +- source indexability, +- page/context relevance, +- external-link counts, +- review flags, +- recommendation, +- and analysis confidence. + +JSON output: + +```bash +backlink-intelligence audit SOURCE_URL TARGET_URL --json +``` + +Save JSON: + +```bash +backlink-intelligence audit SOURCE_URL TARGET_URL --output audit.json +``` + +## 2. Qualify prospects in bulk + +Create a CSV: + +```csv +source_url,target_url,preferred_anchor +https://publisher.example/post,https://brand.example/page,agentic ai course +``` + +Run: + +```bash +backlink-intelligence qualify prospects.csv --output qualification-report.csv +``` -- a source article URL, -- a target URL, and -- a preferred keyword or anchor. +The report includes evidence such as relevance, placement potential, outbound-link density, review flags, confidence, and a workflow recommendation: -It will identify the strongest candidate paragraphs and return explainable placement recommendations. +- `prioritize` +- `manual_review` +- `low_priority` -### Before +Bulk commands cache repeated URLs during a run and use a small delay between rows by default. Use `--delay` to adjust that delay responsibly. -> The original publisher paragraph is shown without modification. +## 3. Find contextual link placements -### After +This is the signature workflow. -> The same paragraph is shown with the target resource integrated naturally, while preserving as much of the original copy as possible. +```bash +backlink-intelligence place \ + "https://publisher.example/article" \ + "https://brand.example/target" \ + --anchor "agentic ai course" \ + --top 3 +``` -Each recommendation is intended to include: +For each recommended paragraph the tool returns: -- paragraph/context match, +- paragraph number, +- context-fit level, - requested anchor, - suggested anchor when the requested wording is awkward, - placement strategy, - editorial intervention level, - original-text preservation, -- reasoning for the recommendation, -- and warnings when a paragraph should **not** be used. +- **Before paragraph**, +- **After paragraph**, +- reasons, +- and review flags. -The goal is not keyword insertion. The goal is **editorially defensible contextual placement**. +The deterministic rewrite engine deliberately favors minimal editorial change. Its output is a placement draft for human review, not an instruction to publish automatically. -## Evidence-first design +## 4. Monitor acquired backlinks -Backlink Intelligence prefers transparent dimensions over an unexplained universal score. +Create a CSV: -Example output direction: +```csv +source_url,target_url,expected_anchor +https://publisher.example/article,https://brand.example/page,AI guide +``` -- **Page relevance:** High -- **Context relevance:** Very High -- **Destination fit:** High -- **Placement:** Editorial Context -- **Anchor fit:** Strong -- **Risk signals:** Review -- **Analysis confidence:** High -- **Recommendation:** Prioritize +First check creates the baseline: -A recommendation should always be accompanied by the evidence and limitations that produced it. +```bash +backlink-intelligence monitor links.csv \ + --state backlink-state.json \ + --output monitor-report.csv +``` -## Planned analysis areas +Run the same command later to detect: -### Existing backlink audit +- removed links, +- anchor changes, +- `rel` attribute changes, +- placement changes, +- source/target status changes, +- canonical changes, +- and robots changes. -- source and target HTTP status -- redirects and final URLs -- title and H1 extraction -- canonical and robots directives -- backlink presence -- anchor text -- `rel` attributes (`nofollow`, `sponsored`, `ugc`) -- surrounding sentence and paragraph -- approximate placement classification -- main-content evidence +## 5. Analyze a backlink portfolio -### Topical and contextual relevance +Input CSV should contain `target_url` and can optionally include `anchor` and `placement`. -Relevance will be evaluated at multiple levels: +```bash +backlink-intelligence portfolio backlinks.csv --output portfolio.json +``` -1. broader domain/topic signals, -2. source page ↔ destination page, -3. source paragraph/context ↔ destination page. +The report summarizes: -The core implementation is intended to remain local-first and explainable, initially using techniques such as term overlap, TF-IDF, cosine similarity, heading/title alignment, and phrase overlap. +- anchor-category distribution, +- destination distribution, +- placement distribution. -### Source-page and outbound-link evidence +## Evidence model -- total external links -- unique external domains -- external-link density -- follow/nofollow distribution -- repeated commercial patterns -- outbound-link neighborhood -- thin-content indicators -- indexability evidence +Backlink Intelligence intentionally avoids an unexplained universal 0–100 backlink score. -These are review signals, not a proprietary "toxicity score." +Instead it exposes dimensions such as: -### Prospect qualification +- **Relevance:** low / medium / high / very high +- **Placement:** editorial context / navigation / sidebar / footer / unknown +- **Risk signals:** explicit review flags +- **Confidence:** low / medium / high +- **Recommendation:** workflow guidance, not a ranking claim -Bulk CSV analysis is planned to help prioritize outreach targets using evidence such as: +See [Methodology](docs/methodology.md) and [Signal Definitions](docs/signal-definitions.md). -- relevance, -- destination fit, -- editorial fit, -- outbound-link behavior, -- review flags, -- and confidence. +## How relevance works -### Link monitoring +The v1 engine is deterministic and local. It combines: -Planned monitoring will detect changes including: +- normalized term overlap, +- cosine similarity over term-frequency vectors, +- title/H1 alignment, +- heading alignment, +- page-to-page similarity, +- paragraph-to-target similarity, +- and shared topical terms. -- backlink removal, -- anchor changes, -- follow → nofollow changes, -- `sponsored`/`ugc` attribute changes, -- target changes, -- source/target redirects, -- 404/410 responses, -- noindex changes, -- canonical changes, -- and material placement changes. +This makes the result reproducible and inspectable without requiring an LLM API. -## Roadmap +## Link placement philosophy -| Release | Milestone | Status | -| --- | --- | --- | -| `v0.1.0` | Backlink evidence auditor | Planned | -| `v0.2.0` | Context and placement classification | Planned | -| `v0.3.0` | Relevance engine | Planned | -| `v0.4.0` | Source quality and outbound-link evidence | Planned | -| `v0.5.0` | Bulk prospect qualification | Planned | -| `v0.6.0` | Contextual placement recommender | Planned | -| `v0.7.0` | Before/After placement recommendations | Planned | -| `v0.8.0` | Backlink monitoring | Planned | -| `v0.9.0` | Portfolio analysis and historical snapshots | Planned | -| `v1.0.0` | Stable integrated toolkit | Planned | +The placement engine follows three principles: -See the detailed [project roadmap](docs/roadmap.md). +1. **Editorial fit before keyword insertion.** +2. **Preserve publisher copy whenever possible.** +3. **Show the evidence and let a human approve the final wording.** -## Planned interfaces +When the exact requested anchor already exists naturally, the tool uses minimal insertion. Otherwise it adds a conservative contextual sentence rather than rewriting the entire paragraph. -The analysis engine is intended to support several interfaces without duplicating the underlying logic: +## Crawl and security safeguards -- **CLI** for technical SEOs and developers -- **CSV / JSON / HTML reports** for SEO workflows -- **Python package** for integrations -- **Local browser UI** for non-technical users -- **Public web interface** as a later phase after the core engine is stable +The fetcher is intentionally bounded: -A third-party user should never need to edit source code just to perform an analysis. +- only HTTP/HTTPS URLs, +- embedded URL credentials blocked, +- localhost/private/link-local/reserved IPs blocked, +- DNS-resolved private addresses blocked, +- redirects revalidated, +- redirect count limited, +- request timeout, +- maximum HTML response size, +- non-HTML content rejected, +- identifiable user agent. -## Installation +See [SECURITY.md](SECURITY.md) and [Ethical Crawling](docs/ethical-crawling.md). -The foundation package is already installable for development, but backlink-analysis commands are not yet implemented. +## Current limitations -```bash -python -m pip install -e . -backlink-intelligence --version -backlink-intelligence status -``` +The v1 parser focuses on server-delivered HTML. Pages whose meaningful content or links are rendered only by client-side JavaScript may require browser rendering, which is intentionally not bundled into the zero-dependency core. -## Current CLI +The Before/After engine is deterministic and conservative. It does not attempt unrestricted AI copywriting. Optional model-assisted rewriting can be added later without making it mandatory for core functionality. -```bash -backlink-intelligence --help -backlink-intelligence --version -backlink-intelligence status -``` +See [Limitations](docs/limitations.md). -The `audit`, `qualify`, `place`, and `monitor` commands will be introduced incrementally according to the roadmap. +## Testing -## Documentation +The repository includes deterministic offline tests for: -- [Methodology](docs/methodology.md) -- [Architecture](docs/architecture.md) -- [Roadmap](docs/roadmap.md) -- [Signal definitions](docs/signal-definitions.md) -- [Limitations](docs/limitations.md) -- [Ethical crawling](docs/ethical-crawling.md) +- HTML/meta/link extraction, +- placement classification, +- URL normalization, +- outbound-link evidence, +- relevance, +- contextual placement ranking, +- Before/After generation, +- monitoring change detection, +- portfolio analysis, +- URL safety, +- CLI behavior. + +Run: + +```bash +python -m unittest discover -s tests -v +``` ## Methodology background -The project builds on the evidence-based backlink evaluation ideas discussed in: +This project implements ideas developed in: **[Backlink Quality Beyond DA & DR](https://alokblog.com/backlink-quality-beyond-da-dr/)** -The article explains the conceptual motivation. This repository is intended to turn that methodology into transparent, testable software. +The article explains the conceptual framework. This repository turns that framework into transparent, testable software. -## Local-first and API-optional +## Phase 2: public website integration -The core project is intended to work without requiring paid SEO or AI APIs. +The GitHub repository is the source of truth for the analysis engine. A later phase can expose selected capabilities through a public interface on `alokblog.com`, using the same package rather than duplicating SEO logic. -Optional integrations may be added later for users who already have access to services such as SEO data providers or language-model APIs, but they should enrich rather than gate the core workflow. +The initial public web version is expected to focus on: -## Security and crawl safety +- single backlink audit, +- contextual placement analysis, +- Before/After placement recommendations. -Because the project accepts arbitrary URLs, URL safety is a first-class engineering requirement. Future network-enabled releases will include protections for private-network targets, redirect validation, response-size limits, timeouts, rate limiting, and other controls. +Bulk crawling and continuous monitoring are better suited to the local/open-source version unless hosted infrastructure is deliberately provisioned for them. -See [SECURITY.md](SECURITY.md) and [Ethical Crawling](docs/ethical-crawling.md). +## Project structure -## Contributing +```text +backlink_intelligence/ +├── audit.py +├── cli.py +├── fetcher.py +├── html_utils.py +├── link_analysis.py +├── models.py +├── monitor.py +├── placement.py +├── portfolio.py +├── qualify.py +├── relevance.py +├── reporting.py +└── safety.py +``` + +## Docker + +Build and run the CLI without installing Python packages into your host environment: -Contributions, test cases, documentation improvements, and evidence-based methodology discussions are welcome. +```bash +docker build -t backlink-intelligence . +docker run --rm backlink-intelligence --help +``` + +## Contributing -Read [CONTRIBUTING.md](CONTRIBUTING.md) before opening a pull request. +Contributions, test cases, parser improvements, and evidence-based methodology discussions are welcome. Read [CONTRIBUTING.md](CONTRIBUTING.md) first. ## License -Released under the [MIT License](LICENSE). +MIT. See [LICENSE](LICENSE). ## Disclaimer diff --git a/SECURITY.md b/SECURITY.md index 9aad419..db92208 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -8,26 +8,24 @@ Use GitHub's private vulnerability reporting feature when enabled for this repos ## Security model -Backlink Intelligence is expected to process URLs supplied by users. Network-enabled releases therefore treat URL fetching as security-sensitive functionality. +Backlink Intelligence processes URLs supplied by users. Network-enabled functionality therefore treats URL fetching as security-sensitive. -Future implementations should protect against, at minimum: +The v1 fetcher blocks or bounds, at minimum: -- localhost and loopback targets, -- private and link-local IP ranges, -- cloud metadata endpoints, -- DNS rebinding and redirect-to-private-network behavior, +- unsupported schemes, +- credentials embedded in URLs, +- localhost and common local hostnames, +- private, loopback, link-local, reserved, multicast, and unspecified IP ranges, +- DNS resolution to non-public addresses, +- redirects to non-public addresses, - excessive redirect chains, - oversized responses, -- decompression bombs, -- unexpectedly slow endpoints, -- unsupported schemes, -- unsafe local file paths, -- use as a generic open proxy, -- and uncontrolled concurrency during bulk analysis. +- unexpectedly slow endpoints through request timeouts, +- and non-HTML content. ## Hosted deployment warning -A command-line tool that fetches user-supplied URLs and an internet-facing web service have different risk profiles. Do not expose the crawler as a public web endpoint without additional validation, rate limiting, abuse prevention, request isolation, and SSRF defenses. +A command-line tool that fetches user-supplied URLs and an internet-facing web service have different risk profiles. Do not expose the crawler as a public web endpoint without additional validation, centralized rate limiting, caching, abuse prevention, request isolation, logging, and SSRF defenses. ## Secrets diff --git a/backlink_intelligence/__init__.py b/backlink_intelligence/__init__.py index de9941d..df30da3 100644 --- a/backlink_intelligence/__init__.py +++ b/backlink_intelligence/__init__.py @@ -1,3 +1,3 @@ """Backlink Intelligence package.""" -__version__ = "0.0.1" +__version__ = "1.0.0" diff --git a/backlink_intelligence/audit.py b/backlink_intelligence/audit.py new file mode 100644 index 0000000..5ef6452 --- /dev/null +++ b/backlink_intelligence/audit.py @@ -0,0 +1,51 @@ +from __future__ import annotations + +from .fetcher import FetchConfig, fetch_page +from .link_analysis import find_backlink, outbound_evidence +from .models import AuditResult +from .relevance import analyze_relevance + + +def _confidence(source_ok: bool, target_ok: bool, found: bool) -> str: + score = int(source_ok) + int(target_ok) + int(found) + return "high" if score == 3 else "medium" if score >= 2 else "low" + + +def _recommendation(found: bool, relevance: str, placement: str, concerns: list[str]) -> str: + if not found: + return "not_found" + if "source_not_indexable" in concerns or "target_unavailable" in concerns: + return "manual_review" + if relevance in {"high", "very_high"} and placement == "editorial_context" and len(concerns) <= 1: + return "strong_candidate" + if relevance == "low" or placement in {"footer", "navigation"}: + return "low_priority" + return "manual_review" + + +def audit_backlink(source_url: str, target_url: str, config: FetchConfig | None = None) -> AuditResult: + source = fetch_page(source_url, config) + target = fetch_page(target_url, config) + backlink = find_backlink(source, target_url) + relevance = analyze_relevance(source, target, backlink.context) + outbound = outbound_evidence(source) + reasons: list[str] = [] + concerns: list[str] = list(outbound.review_flags) + if backlink.found: + reasons.append("target_link_found") + if backlink.placement == "editorial_context": + reasons.append("editorial_context_placement") + if "nofollow" not in backlink.rel: + reasons.append("link_is_followed") + if relevance.level in {"high", "very_high"}: + reasons.append("strong_topical_alignment") + if source.is_indexable: + reasons.append("source_indexable") + else: + concerns.append("source_not_indexable") + if target.status_code != 200: + concerns.append("target_unavailable") + if "sponsored" in backlink.rel: + concerns.append("sponsored_attribute_present") + recommendation = _recommendation(backlink.found, relevance.level, backlink.placement, concerns) + return AuditResult(source=source, target=target, backlink=backlink, relevance=relevance, outbound=outbound, recommendation=recommendation, confidence=_confidence(source.status_code == 200, target.status_code == 200, backlink.found), reasons=reasons, concerns=sorted(set(concerns))) diff --git a/backlink_intelligence/cli.py b/backlink_intelligence/cli.py index 4f87aa0..900176f 100644 --- a/backlink_intelligence/cli.py +++ b/backlink_intelligence/cli.py @@ -1,47 +1,73 @@ -"""Command-line entry point for the Backlink Intelligence foundation.""" +"""Command-line interface for Backlink Intelligence.""" from __future__ import annotations import argparse +import sys from backlink_intelligence import __version__ +from .audit import audit_backlink +from .monitor import monitor_csv +from .placement import suggest_placements +from .portfolio import analyze_portfolio +from .qualify import qualify_csv +from .reporting import audit_text, write_json def build_parser() -> argparse.ArgumentParser: - parser = argparse.ArgumentParser( - prog="backlink-intelligence", - description=( - "Evidence-first backlink intelligence for auditing, qualification, " - "contextual placement, and monitoring." - ), - ) - parser.add_argument( - "--version", - action="version", - version=f"%(prog)s {__version__}", - ) - - subparsers = parser.add_subparsers(dest="command") - subparsers.add_parser( - "status", - help="Show the current implementation status and next planned release.", - ) + parser = argparse.ArgumentParser(prog="backlink-intelligence", description="Evidence-first backlink intelligence for auditing, qualification, contextual placement, monitoring, and portfolio review.") + parser.add_argument("--version", action="version", version=f"%(prog)s {__version__}") + sub = parser.add_subparsers(dest="command") + sub.add_parser("status", help="Show implementation status.") + audit = sub.add_parser("audit", help="Audit an existing backlink.") + audit.add_argument("source_url"); audit.add_argument("target_url"); audit.add_argument("--json", action="store_true", dest="as_json"); audit.add_argument("--output", help="Optional JSON output file.") + qualify = sub.add_parser("qualify", help="Qualify backlink prospects from CSV.") + qualify.add_argument("input_csv"); qualify.add_argument("--output", default="qualification-report.csv"); qualify.add_argument("--delay", type=float, default=0.5, help="Delay between bulk rows in seconds.") + place = sub.add_parser("place", help="Find contextual link placement opportunities.") + place.add_argument("source_url"); place.add_argument("target_url"); place.add_argument("--anchor", required=True, help="Preferred anchor / keyword."); place.add_argument("--top", type=int, default=3); place.add_argument("--json", action="store_true", dest="as_json"); place.add_argument("--output", help="Optional JSON output file.") + monitor = sub.add_parser("monitor", help="Monitor backlinks listed in a CSV file.") + monitor.add_argument("input_csv"); monitor.add_argument("--state", default="backlink-state.json"); monitor.add_argument("--output", default="monitor-report.csv"); monitor.add_argument("--delay", type=float, default=0.5, help="Delay between monitoring rows in seconds.") + portfolio = sub.add_parser("portfolio", help="Analyze anchor/destination/placement distributions.") + portfolio.add_argument("input_csv"); portfolio.add_argument("--output", help="Optional JSON output file.") return parser +def _placement_text(items) -> str: + if not items: + return "No suitable placement opportunities could be generated." + chunks: list[str] = [] + for item in items: + chunks.extend([f"PLACEMENT OPPORTUNITY #{item.rank}", f"Paragraph: {item.paragraph_index}", f"Context fit: {item.context_level}", f"Similarity score: {item.score:.3f}", f"Strategy: {item.strategy}", f"Editorial intervention:{item.intervention}", f"Text preservation: {item.preservation_percent:.1f}%", "", "BEFORE", item.before, "", "AFTER", item.after]) + if item.reasons: chunks.extend(["", "Why this placement:", *[f" + {v}" for v in item.reasons]]) + if item.warnings: chunks.extend(["", "Review flags:", *[f" ! {v}" for v in item.warnings]]) + chunks.extend(["", "-" * 72, ""]) + return "\n".join(chunks).rstrip() + + def main(argv: list[str] | None = None) -> int: - parser = build_parser() - args = parser.parse_args(argv) - - if args.command == "status": - print("Backlink Intelligence foundation is installed.") - print("Current version: 0.0.1 (pre-alpha foundation)") - print("Next milestone: v0.1.0 Backlink Evidence Auditor") - print("Planned workflow: Discover -> Qualify -> Place -> Monitor") - return 0 - - parser.print_help() - return 0 + parser = build_parser(); args = parser.parse_args(argv) + try: + if args.command == "status": + print("Backlink Intelligence is installed and functional."); print(f"Version: {__version__}"); print("Available: audit, qualify, place, monitor, portfolio"); print("Workflow: Discover -> Qualify -> Place -> Monitor -> Analyze"); return 0 + if args.command == "audit": + result = audit_backlink(args.source_url, args.target_url) + if args.output: write_json(result.to_dict(), args.output) + print(write_json(result.to_dict()) if args.as_json else audit_text(result)); return 0 if result.source.status_code else 2 + if args.command == "qualify": + rows = qualify_csv(args.input_csv, args.output, delay_seconds=args.delay); print(f"Qualified {len(rows)} prospect(s). Report: {args.output}"); return 0 + if args.command == "place": + items = suggest_placements(args.source_url, args.target_url, args.anchor, top_n=args.top); payload = [item.to_dict() for item in items] + if args.output: write_json(payload, args.output) + print(write_json(payload) if args.as_json else _placement_text(items)); return 0 if items else 3 + if args.command == "monitor": + rows = monitor_csv(args.input_csv, args.state, args.output, delay_seconds=args.delay); changed = sum("unchanged" not in row["changes"] and "baseline_created" not in row["changes"] for row in rows); print(f"Checked {len(rows)} link(s). Changed: {changed}. Report: {args.output}"); return 0 + if args.command == "portfolio": + result = analyze_portfolio(args.input_csv); print(write_json(result, args.output)); return 0 + parser.print_help(); return 0 + except KeyboardInterrupt: + print("Cancelled.", file=sys.stderr); return 130 + except Exception as exc: + print(f"Error: {exc}", file=sys.stderr); return 1 if __name__ == "__main__": diff --git a/backlink_intelligence/fetcher.py b/backlink_intelligence/fetcher.py new file mode 100644 index 0000000..de338f9 --- /dev/null +++ b/backlink_intelligence/fetcher.py @@ -0,0 +1,71 @@ +from __future__ import annotations + +import gzip +import io +from dataclasses import dataclass +from urllib.error import HTTPError, URLError +from urllib.request import HTTPRedirectHandler, Request, build_opener + +from .html_utils import parse_page +from .models import PageEvidence +from .safety import UnsafeURLError, validate_public_url + + +@dataclass(slots=True) +class FetchConfig: + timeout: float = 12.0 + max_bytes: int = 2_000_000 + max_redirects: int = 5 + user_agent: str = "BacklinkIntelligence/1.0 (+https://github.com/alok-vibe-code/backlink-intelligence)" + + +class SafeRedirectHandler(HTTPRedirectHandler): + def __init__(self, max_redirects: int) -> None: + super().__init__() + self.max_redirects = max_redirects + self.count = 0 + + def redirect_request(self, req, fp, code, msg, headers, newurl): + self.count += 1 + if self.count > self.max_redirects: + raise UnsafeURLError("Redirect limit exceeded.") + validate_public_url(newurl) + return super().redirect_request(req, fp, code, msg, headers, newurl) + + +def _decode_body(raw: bytes, encoding_header: str, content_type: str) -> str: + if "gzip" in encoding_header.lower(): + raw = gzip.GzipFile(fileobj=io.BytesIO(raw)).read() + charset = "utf-8" + marker = "charset=" + if marker in content_type.lower(): + charset = content_type.lower().split(marker, 1)[1].split(";", 1)[0].strip() or "utf-8" + try: + return raw.decode(charset, errors="replace") + except LookupError: + return raw.decode("utf-8", errors="replace") + + +def fetch_page(url: str, config: FetchConfig | None = None) -> PageEvidence: + config = config or FetchConfig() + validate_public_url(url) + redirect_handler = SafeRedirectHandler(config.max_redirects) + opener = build_opener(redirect_handler) + request = Request(url, headers={"User-Agent": config.user_agent, "Accept": "text/html,application/xhtml+xml;q=0.9,*/*;q=0.1", "Accept-Encoding": "gzip"}) + try: + with opener.open(request, timeout=config.timeout) as response: + final_url = response.geturl() + validate_public_url(final_url) + status = getattr(response, "status", 200) + content_type = response.headers.get("Content-Type", "") + if "text/html" not in content_type.lower() and "application/xhtml" not in content_type.lower(): + return PageEvidence(requested_url=url, final_url=final_url, status_code=status, error=f"Unsupported content type: {content_type or 'unknown'}") + raw = response.read(config.max_bytes + 1) + if len(raw) > config.max_bytes: + return PageEvidence(requested_url=url, final_url=final_url, status_code=status, error=f"Response exceeded max_bytes={config.max_bytes}") + html = _decode_body(raw, response.headers.get("Content-Encoding", ""), content_type) + return parse_page(html, requested_url=url, final_url=final_url, status_code=status) + except HTTPError as exc: + return PageEvidence(requested_url=url, final_url=exc.geturl(), status_code=exc.code, error=str(exc)) + except (URLError, TimeoutError, UnsafeURLError, OSError) as exc: + return PageEvidence(requested_url=url, final_url=url, status_code=0, error=str(exc)) diff --git a/backlink_intelligence/html_utils.py b/backlink_intelligence/html_utils.py new file mode 100644 index 0000000..07dc09f --- /dev/null +++ b/backlink_intelligence/html_utils.py @@ -0,0 +1,146 @@ +from __future__ import annotations + +import re +from html.parser import HTMLParser +from urllib.parse import urljoin + +from .models import PageEvidence, PageLink + +_WS = re.compile(r"\s+") + + +def clean_text(value: str) -> str: + return _WS.sub(" ", value or "").strip() + + +class EvidenceHTMLParser(HTMLParser): + def __init__(self, base_url: str) -> None: + super().__init__(convert_charrefs=True) + self.base_url = base_url + self.title_parts: list[str] = [] + self.h1_parts: list[str] = [] + self.headings: list[str] = [] + self.paragraphs: list[str] = [] + self.links: list[PageLink] = [] + self.canonical = "" + self.robots: list[str] = [] + self._stack: list[str] = [] + self._title_depth = 0 + self._h1_depth = 0 + self._heading_tag: str | None = None + self._heading_parts: list[str] = [] + self._paragraph_depth = 0 + self._paragraph_parts: list[str] = [] + self._current_link: dict | None = None + self._skip_depth = 0 + self._landmark_stack: list[str] = [] + + def _placement(self) -> str: + landmarks = set(self._landmark_stack) + if "footer" in landmarks: + return "footer" + if "nav" in landmarks: + return "navigation" + if "aside" in landmarks: + return "sidebar" + if "header" in landmarks: + return "navigation" + if "article" in landmarks or "main" in landmarks: + return "editorial_context" + return "unknown" + + def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None: + tag = tag.lower() + attrs_dict = {k.lower(): (v or "") for k, v in attrs} + self._stack.append(tag) + if tag in {"script", "style", "noscript", "template", "svg"}: + self._skip_depth += 1 + if tag in {"article", "main", "nav", "aside", "footer", "header"}: + self._landmark_stack.append(tag) + if tag == "title": + self._title_depth += 1 + if tag == "h1": + self._h1_depth += 1 + if tag in {"h1", "h2", "h3", "h4", "h5", "h6"}: + self._heading_tag = tag + self._heading_parts = [] + if tag == "p": + self._paragraph_depth += 1 + if self._paragraph_depth == 1: + self._paragraph_parts = [] + if tag == "link" and attrs_dict.get("rel", "").lower() == "canonical": + href = clean_text(attrs_dict.get("href", "")) + if href: + self.canonical = urljoin(self.base_url, href) + if tag == "meta" and attrs_dict.get("name", "").lower() in {"robots", "googlebot"}: + for item in attrs_dict.get("content", "").lower().split(","): + item = clean_text(item) + if item and item not in self.robots: + self.robots.append(item) + if tag == "a": + href = clean_text(attrs_dict.get("href", "")) + rel = tuple(sorted({v.lower() for v in attrs_dict.get("rel", "").split() if v})) + self._current_link = {"href": urljoin(self.base_url, href) if href else "", "text": [], "rel": rel, "placement": self._placement()} + + def handle_endtag(self, tag: str) -> None: + tag = tag.lower() + if tag == "title" and self._title_depth: + self._title_depth -= 1 + if tag == "h1" and self._h1_depth: + self._h1_depth -= 1 + if self._heading_tag == tag: + text = clean_text(" ".join(self._heading_parts)) + if text: + self.headings.append(text) + self._heading_tag = None + self._heading_parts = [] + if tag == "p" and self._paragraph_depth: + self._paragraph_depth -= 1 + if self._paragraph_depth == 0: + text = clean_text(" ".join(self._paragraph_parts)) + if text: + self.paragraphs.append(text) + self._paragraph_parts = [] + if tag == "a" and self._current_link is not None: + text = clean_text(" ".join(self._current_link["text"])) + paragraph = clean_text(" ".join(self._paragraph_parts)) if self._paragraph_parts else "" + href = self._current_link["href"] + if href: + self.links.append(PageLink(href=href, text=text, rel=self._current_link["rel"], context=paragraph, paragraph=paragraph, placement=self._current_link["placement"])) + self._current_link = None + if tag in {"script", "style", "noscript", "template", "svg"} and self._skip_depth: + self._skip_depth -= 1 + if tag in {"article", "main", "nav", "aside", "footer", "header"}: + for i in range(len(self._landmark_stack) - 1, -1, -1): + if self._landmark_stack[i] == tag: + del self._landmark_stack[i] + break + if self._stack: + self._stack.pop() + + def handle_data(self, data: str) -> None: + if self._skip_depth: + return + text = clean_text(data) + if not text: + return + if self._title_depth: + self.title_parts.append(text) + if self._h1_depth: + self.h1_parts.append(text) + if self._heading_tag: + self._heading_parts.append(text) + if self._paragraph_depth: + self._paragraph_parts.append(text) + if self._current_link is not None: + self._current_link["text"].append(text) + + +def parse_page(html: str, *, requested_url: str, final_url: str, status_code: int) -> PageEvidence: + parser = EvidenceHTMLParser(final_url) + parser.feed(html) + title = clean_text(" ".join(parser.title_parts)) + h1 = clean_text(" ".join(parser.h1_parts)) + paragraphs = [p for p in parser.paragraphs if len(p.split()) >= 3] + text = clean_text(" ".join([title, h1, *parser.headings, *paragraphs])) + return PageEvidence(requested_url=requested_url, final_url=final_url, status_code=status_code, title=title, h1=h1, canonical=parser.canonical, robots=tuple(parser.robots), text=text, paragraphs=paragraphs, headings=parser.headings, links=parser.links, word_count=len(text.split())) diff --git a/backlink_intelligence/link_analysis.py b/backlink_intelligence/link_analysis.py new file mode 100644 index 0000000..35e0112 --- /dev/null +++ b/backlink_intelligence/link_analysis.py @@ -0,0 +1,56 @@ +from __future__ import annotations + +from urllib.parse import urlparse, urlunparse + +from .models import BacklinkEvidence, OutboundEvidence, PageEvidence + + +def normalize_url(url: str) -> str: + p = urlparse(url) + scheme = (p.scheme or "https").lower() + host = (p.hostname or "").lower() + port = p.port + netloc = host + if port and not ((scheme == "https" and port == 443) or (scheme == "http" and port == 80)): + netloc = f"{host}:{port}" + path = p.path or "/" + if path != "/": + path = path.rstrip("/") + return urlunparse((scheme, netloc, path, "", p.query, "")) + + +def same_destination(a: str, b: str) -> bool: + try: + return normalize_url(a) == normalize_url(b) + except ValueError: + return False + + +def find_backlink(source: PageEvidence, target_url: str) -> BacklinkEvidence: + for link in source.links: + if same_destination(link.href, target_url): + return BacklinkEvidence(found=True, source_url=source.final_url, target_url=target_url, anchor=link.text, rel=link.rel, context=link.context, paragraph=link.paragraph, placement=link.placement) + return BacklinkEvidence(found=False, source_url=source.final_url, target_url=target_url) + + +def outbound_evidence(page: PageEvidence) -> OutboundEvidence: + source_host = (urlparse(page.final_url).hostname or "").lower() + external = [] + for link in page.links: + host = (urlparse(link.href).hostname or "").lower() + if host and host != source_host and not host.endswith("." + source_host): + external.append(link) + domains = {(urlparse(l.href).hostname or "").lower() for l in external} + nofollow = sum("nofollow" in l.rel for l in external) + sponsored = sum("sponsored" in l.rel for l in external) + ugc = sum("ugc" in l.rel for l in external) + follow = len(external) - nofollow + density = (len(external) / max(page.word_count, 1)) * 1000 + flags: list[str] = [] + if page.word_count and density > 45: + flags.append("high_external_link_density") + if len(domains) >= 35: + flags.append("many_unique_external_domains") + if len(external) >= 40 and sponsored == 0 and nofollow / max(len(external), 1) < 0.1: + flags.append("many_followed_external_links") + return OutboundEvidence(total_links=len(page.links), external_links=len(external), unique_external_domains=len(domains), external_links_per_1000_words=round(density, 2), follow_links=follow, nofollow_links=nofollow, sponsored_links=sponsored, ugc_links=ugc, review_flags=flags) diff --git a/backlink_intelligence/models.py b/backlink_intelligence/models.py new file mode 100644 index 0000000..f03beeb --- /dev/null +++ b/backlink_intelligence/models.py @@ -0,0 +1,138 @@ +from __future__ import annotations + +from dataclasses import asdict, dataclass, field +from typing import Any + + +@dataclass(slots=True) +class PageLink: + href: str + text: str + rel: tuple[str, ...] = () + context: str = "" + paragraph: str = "" + placement: str = "unknown" + + @property + def is_nofollow(self) -> bool: + return "nofollow" in self.rel + + @property + def is_sponsored(self) -> bool: + return "sponsored" in self.rel + + @property + def is_ugc(self) -> bool: + return "ugc" in self.rel + + +@dataclass(slots=True) +class PageEvidence: + requested_url: str + final_url: str + status_code: int + title: str = "" + h1: str = "" + canonical: str = "" + robots: tuple[str, ...] = () + text: str = "" + paragraphs: list[str] = field(default_factory=list) + headings: list[str] = field(default_factory=list) + links: list[PageLink] = field(default_factory=list) + word_count: int = 0 + error: str = "" + + @property + def is_indexable(self) -> bool: + blocked = {"noindex", "none"} + return self.status_code == 200 and not any(v in blocked for v in self.robots) + + +@dataclass(slots=True) +class BacklinkEvidence: + found: bool + source_url: str + target_url: str + anchor: str = "" + rel: tuple[str, ...] = () + context: str = "" + paragraph: str = "" + placement: str = "unknown" + + +@dataclass(slots=True) +class RelevanceEvidence: + page_similarity: float + context_similarity: float + title_similarity: float + heading_similarity: float + level: str + shared_terms: list[str] = field(default_factory=list) + + +@dataclass(slots=True) +class OutboundEvidence: + total_links: int + external_links: int + unique_external_domains: int + external_links_per_1000_words: float + follow_links: int + nofollow_links: int + sponsored_links: int + ugc_links: int + review_flags: list[str] = field(default_factory=list) + + +@dataclass(slots=True) +class AuditResult: + source: PageEvidence + target: PageEvidence + backlink: BacklinkEvidence + relevance: RelevanceEvidence + outbound: OutboundEvidence + recommendation: str + confidence: str + reasons: list[str] = field(default_factory=list) + concerns: list[str] = field(default_factory=list) + + def to_dict(self) -> dict[str, Any]: + return asdict(self) + + +@dataclass(slots=True) +class PlacementSuggestion: + rank: int + paragraph_index: int + score: float + context_level: str + requested_anchor: str + suggested_anchor: str + strategy: str + before: str + after: str + added_words: int + preservation_percent: float + intervention: str + reasons: list[str] = field(default_factory=list) + warnings: list[str] = field(default_factory=list) + + def to_dict(self) -> dict[str, Any]: + return asdict(self) + + +@dataclass(slots=True) +class MonitorSnapshot: + source_url: str + target_url: str + source_status: int + target_status: int + link_found: bool + anchor: str + rel: tuple[str, ...] + placement: str + source_canonical: str + source_robots: tuple[str, ...] + checked_at: str + + def to_dict(self) -> dict[str, Any]: + return asdict(self) diff --git a/backlink_intelligence/monitor.py b/backlink_intelligence/monitor.py new file mode 100644 index 0000000..d73938a --- /dev/null +++ b/backlink_intelligence/monitor.py @@ -0,0 +1,59 @@ +from __future__ import annotations + +import csv +import json +import time +from datetime import datetime, timezone +from pathlib import Path + +from .fetcher import FetchConfig, fetch_page +from .link_analysis import find_backlink +from .models import MonitorSnapshot, PageEvidence + + +def _snapshot_from_pages(source_url: str, target_url: str, source: PageEvidence, target: PageEvidence) -> MonitorSnapshot: + backlink = find_backlink(source, target_url) + return MonitorSnapshot(source_url=source_url, target_url=target_url, source_status=source.status_code, target_status=target.status_code, link_found=backlink.found, anchor=backlink.anchor, rel=backlink.rel, placement=backlink.placement, source_canonical=source.canonical, source_robots=source.robots, checked_at=datetime.now(timezone.utc).isoformat()) + + +def snapshot_link(source_url: str, target_url: str, config: FetchConfig | None = None) -> MonitorSnapshot: + return _snapshot_from_pages(source_url, target_url, fetch_page(source_url, config), fetch_page(target_url, config)) + + +def compare_snapshots(old: dict | None, new: MonitorSnapshot, expected_anchor: str = "") -> list[str]: + if old is None: + return ["baseline_created"] + changes: list[str] = [] + if bool(old.get("link_found")) and not new.link_found: changes.append("link_removed") + if old.get("anchor", "") != new.anchor: changes.append("anchor_changed") + if set(old.get("rel", [])) != set(new.rel): changes.append("rel_attributes_changed") + if old.get("placement", "") != new.placement: changes.append("placement_changed") + if old.get("source_status") != new.source_status: changes.append("source_status_changed") + if old.get("target_status") != new.target_status: changes.append("target_status_changed") + if old.get("source_canonical", "") != new.source_canonical: changes.append("canonical_changed") + if set(old.get("source_robots", [])) != set(new.source_robots): changes.append("robots_changed") + if expected_anchor and new.link_found and new.anchor != expected_anchor: changes.append("anchor_differs_from_expected") + return changes or ["unchanged"] + + +def monitor_csv(input_path: str | Path, state_path: str | Path, output_path: str | Path, config: FetchConfig | None = None, *, delay_seconds: float = 0.5) -> list[dict]: + state_file = Path(state_path) + old_state = json.loads(state_file.read_text(encoding="utf-8")) if state_file.exists() else {} + results: list[dict] = []; new_state: dict[str, dict] = {}; cache: dict[str, PageEvidence] = {} + def cached(url: str) -> PageEvidence: + if url not in cache: cache[url] = fetch_page(url, config) + return cache[url] + with Path(input_path).open(newline="", encoding="utf-8-sig") as handle: + reader = csv.DictReader(handle) + required = {"source_url", "target_url"}; missing = required - set(reader.fieldnames or []) + if missing: raise ValueError(f"CSV missing required columns: {', '.join(sorted(missing))}") + for row in reader: + source_url = row.get("source_url", "").strip(); target_url = row.get("target_url", "").strip(); expected_anchor = row.get("expected_anchor", "").strip(); key = f"{source_url}|{target_url}" + snap = _snapshot_from_pages(source_url, target_url, cached(source_url), cached(target_url)); changes = compare_snapshots(old_state.get(key), snap, expected_anchor); new_state[key] = snap.to_dict() + results.append({"source_url": source_url, "target_url": target_url, "link_found": snap.link_found, "anchor": snap.anchor, "rel": " ".join(snap.rel), "placement": snap.placement, "source_status": snap.source_status, "target_status": snap.target_status, "changes": ";".join(changes), "checked_at": snap.checked_at}) + if delay_seconds > 0: time.sleep(delay_seconds) + state_file.write_text(json.dumps(new_state, indent=2), encoding="utf-8") + if results: + with Path(output_path).open("w", newline="", encoding="utf-8") as handle: + writer = csv.DictWriter(handle, fieldnames=list(results[0].keys())); writer.writeheader(); writer.writerows(results) + return results diff --git a/backlink_intelligence/placement.py b/backlink_intelligence/placement.py new file mode 100644 index 0000000..af15079 --- /dev/null +++ b/backlink_intelligence/placement.py @@ -0,0 +1,84 @@ +from __future__ import annotations + +from .fetcher import FetchConfig, fetch_page +from .models import PageEvidence, PlacementSuggestion +from .relevance import similarity, tokens + + +def _intervention(preservation: float, added_words: int) -> str: + if preservation >= 95 and added_words <= 24: + return "low" + if preservation >= 85 and added_words <= 40: + return "medium" + return "high" + + +def _anchor_naturalness(anchor: str) -> tuple[str, list[str]]: + warnings: list[str] = [] + words = anchor.split() + if not anchor.strip(): + return "weak", ["empty_anchor"] + if len(words) > 7: + warnings.append("long_anchor") + if anchor.isupper() and len(anchor) > 5: + warnings.append("all_caps_anchor") + if any(ch in anchor for ch in ["|", "[", "]", "{"]): + warnings.append("awkward_anchor_characters") + return ("strong" if not warnings else "medium"), warnings + + +def _select_anchor(preferred: str, target_title: str) -> tuple[str, list[str]]: + preferred = preferred.strip() + if not preferred: + return (target_title.strip() or "this related resource"), [] + _, warnings = _anchor_naturalness(preferred) + if warnings and target_title.strip(): + return target_title.strip(), warnings + ["suggested_anchor_differs_from_requested"] + return preferred, warnings + + +def _compose_after(paragraph: str, anchor: str, target_url: str, target_title: str) -> tuple[str, str]: + linked = f"[{anchor}]({target_url})" + idx = paragraph.lower().find(anchor.lower()) + if idx >= 0: + return "minimal_insertion", paragraph[:idx] + linked + paragraph[idx + len(anchor):] + topic = target_title.strip() + sentence = f"For a more detailed resource on {topic}, see {linked}." if topic and topic.lower() != anchor.lower() else f"For a more detailed resource on this topic, see {linked}." + return "contextual_sentence", paragraph.rstrip() + " " + sentence + + +def rank_placements(source: PageEvidence, target: PageEvidence, preferred_anchor: str, target_url: str, *, top_n: int = 3) -> list[PlacementSuggestion]: + if source.status_code != 200 or target.status_code != 200: + return [] + anchor, anchor_warnings = _select_anchor(preferred_anchor, target.title) + target_profile = " ".join([target.title, target.h1, *target.headings, target.text[:12000]]) + candidates: list[tuple[float, int, str]] = [] + for i, paragraph in enumerate(source.paragraphs, start=1): + wc = len(paragraph.split()) + if wc < 18 or wc > 260: + continue + score = similarity(paragraph, target_profile) + anchor_terms = set(tokens(anchor)) + anchor_overlap = len(anchor_terms & set(tokens(paragraph))) / max(len(anchor_terms), 1) + candidates.append((round((0.84 * score) + (0.16 * anchor_overlap), 4), i, paragraph)) + candidates.sort(key=lambda item: (-item[0], item[1])) + suggestions: list[PlacementSuggestion] = [] + for rank, (score, index, paragraph) in enumerate(candidates[:max(top_n, 1)], start=1): + strategy, after = _compose_after(paragraph, anchor, target_url, target.title) + original_words = max(len(paragraph.split()), 1) + added = max(len(after.split()) - original_words, 0) + preservation = 100.0 + warnings = list(anchor_warnings) + reasons = ["paragraph_has_strong_target_similarity"] if score >= 0.25 else ["best_available_context_match"] + reasons.append("anchor_already_present_in_original_copy" if strategy == "minimal_insertion" else "publisher_copy_preserved") + if score < 0.12: + warnings.append("weak_context_match_manual_review_required") + context_level = "very_high" if score >= 0.48 else "high" if score >= 0.30 else "medium" if score >= 0.15 else "low" + suggestions.append(PlacementSuggestion(rank=rank, paragraph_index=index, score=score, context_level=context_level, requested_anchor=preferred_anchor, suggested_anchor=anchor, strategy=strategy, before=paragraph, after=after, added_words=added, preservation_percent=preservation, intervention=_intervention(preservation, added), reasons=reasons, warnings=warnings)) + return suggestions + + +def suggest_placements(source_url: str, target_url: str, preferred_anchor: str, *, top_n: int = 3, config: FetchConfig | None = None) -> list[PlacementSuggestion]: + source = fetch_page(source_url, config) + target = fetch_page(target_url, config) + return rank_placements(source, target, preferred_anchor, target_url, top_n=top_n) diff --git a/backlink_intelligence/portfolio.py b/backlink_intelligence/portfolio.py new file mode 100644 index 0000000..6029cb6 --- /dev/null +++ b/backlink_intelligence/portfolio.py @@ -0,0 +1,28 @@ +from __future__ import annotations + +import csv +from collections import Counter +from pathlib import Path +from urllib.parse import urlparse + + +def classify_anchor(anchor: str, target_url: str = "") -> str: + anchor = (anchor or "").strip() + if not anchor: return "empty" + lower = anchor.lower() + if lower.startswith("http://") or lower.startswith("https://") or lower.startswith("www."): return "naked_url" + if lower in {"click here", "here", "website", "this page", "learn more", "read more", "source"}: return "generic" + host = (urlparse(target_url).hostname or "").lower().removeprefix("www."); brand = host.split(".")[0].replace("-", " ") if host else "" + if brand and brand in lower: return "branded" + return "exact_or_commercial" if len(lower.split()) <= 5 else "descriptive" + + +def analyze_portfolio(input_path: str | Path) -> dict: + anchors: Counter[str] = Counter(); destinations: Counter[str] = Counter(); placements: Counter[str] = Counter(); total = 0 + with Path(input_path).open(newline="", encoding="utf-8-sig") as handle: + reader = csv.DictReader(handle) + if "target_url" not in (reader.fieldnames or []): raise ValueError("CSV must contain target_url.") + for row in reader: + total += 1; target = row.get("target_url", "").strip(); anchor = row.get("anchor", row.get("expected_anchor", "")).strip(); placement = row.get("placement", "unknown").strip() or "unknown" + anchors[classify_anchor(anchor, target)] += 1; destinations[target or "unknown"] += 1; placements[placement] += 1 + return {"total_links": total, "anchor_distribution": dict(anchors.most_common()), "destination_distribution": dict(destinations.most_common()), "placement_distribution": dict(placements.most_common())} diff --git a/backlink_intelligence/qualify.py b/backlink_intelligence/qualify.py new file mode 100644 index 0000000..6caae57 --- /dev/null +++ b/backlink_intelligence/qualify.py @@ -0,0 +1,64 @@ +from __future__ import annotations + +import csv +import time +from pathlib import Path + +from .fetcher import FetchConfig, fetch_page +from .link_analysis import outbound_evidence +from .models import PageEvidence +from .placement import rank_placements +from .relevance import analyze_relevance + + +def _qualify_pages(source: PageEvidence, target: PageEvidence, source_url: str, target_url: str, preferred_anchor: str) -> dict: + relevance = analyze_relevance(source, target) + outbound = outbound_evidence(source) + placements = rank_placements(source, target, preferred_anchor, target_url, top_n=1) + placement_score = placements[0].score if placements else 0.0 + flags = list(outbound.review_flags) + if not source.is_indexable: + flags.append("source_not_indexable") + if source.status_code != 200: + flags.append("source_unavailable") + if target.status_code != 200: + flags.append("target_unavailable") + if source.status_code == 200 and target.status_code == 200 and relevance.level in {"high", "very_high"} and placement_score >= 0.18 and len(flags) <= 1: + recommendation = "prioritize" + elif source.status_code != 200 or relevance.level == "low" or "source_not_indexable" in flags: + recommendation = "low_priority" + else: + recommendation = "manual_review" + confidence = "high" if source.status_code == 200 and target.status_code == 200 else "low" + return {"source_url": source_url, "target_url": target_url, "preferred_anchor": preferred_anchor, "source_status": source.status_code, "target_status": target.status_code, "source_indexable": source.is_indexable, "page_relevance": relevance.level, "page_similarity": relevance.page_similarity, "title_similarity": relevance.title_similarity, "placement_potential": placements[0].context_level if placements else "unknown", "placement_score": placement_score, "external_links": outbound.external_links, "unique_external_domains": outbound.unique_external_domains, "external_links_per_1000_words": outbound.external_links_per_1000_words, "review_flags": ";".join(sorted(set(flags))), "recommendation": recommendation, "confidence": confidence} + + +def qualify_prospect(source_url: str, target_url: str, preferred_anchor: str = "", config: FetchConfig | None = None) -> dict: + return _qualify_pages(fetch_page(source_url, config), fetch_page(target_url, config), source_url, target_url, preferred_anchor) + + +def qualify_csv(input_path: str | Path, output_path: str | Path, config: FetchConfig | None = None, *, delay_seconds: float = 0.5) -> list[dict]: + rows: list[dict] = [] + cache: dict[str, PageEvidence] = {} + def cached(url: str) -> PageEvidence: + if url not in cache: + cache[url] = fetch_page(url, config) + return cache[url] + with Path(input_path).open(newline="", encoding="utf-8-sig") as handle: + reader = csv.DictReader(handle) + required = {"source_url", "target_url"} + missing = required - set(reader.fieldnames or []) + if missing: + raise ValueError(f"CSV missing required columns: {', '.join(sorted(missing))}") + for row in reader: + source_url = row.get("source_url", "").strip() + target_url = row.get("target_url", "").strip() + preferred_anchor = row.get("preferred_anchor", "").strip() + rows.append(_qualify_pages(cached(source_url), cached(target_url), source_url, target_url, preferred_anchor)) + if delay_seconds > 0: + time.sleep(delay_seconds) + if rows: + with Path(output_path).open("w", newline="", encoding="utf-8") as handle: + writer = csv.DictWriter(handle, fieldnames=list(rows[0].keys())) + writer.writeheader(); writer.writerows(rows) + return rows diff --git a/backlink_intelligence/relevance.py b/backlink_intelligence/relevance.py new file mode 100644 index 0000000..70d0bfd --- /dev/null +++ b/backlink_intelligence/relevance.py @@ -0,0 +1,57 @@ +from __future__ import annotations + +import math +import re +from collections import Counter + +from .models import PageEvidence, RelevanceEvidence + +_TOKEN = re.compile(r"[a-z0-9][a-z0-9+#.-]{1,}", re.I) +_STOP = {"a", "an", "and", "are", "as", "at", "be", "by", "for", "from", "has", "have", "how", "in", "into", "is", "it", "its", "of", "on", "or", "that", "the", "their", "this", "to", "was", "we", "what", "when", "where", "which", "with", "you", "your", "can", "will", "more"} + + +def tokens(text: str) -> list[str]: + return [t.lower().strip(".-") for t in _TOKEN.findall(text or "") if t.lower().strip(".-") not in _STOP] + + +def _cosine(a: str, b: str) -> float: + ca, cb = Counter(tokens(a)), Counter(tokens(b)) + if not ca or not cb: + return 0.0 + common = set(ca) & set(cb) + numerator = sum(ca[t] * cb[t] for t in common) + da = math.sqrt(sum(v * v for v in ca.values())) + db = math.sqrt(sum(v * v for v in cb.values())) + return numerator / (da * db) if da and db else 0.0 + + +def _overlap(a: str, b: str) -> float: + sa, sb = set(tokens(a)), set(tokens(b)) + if not sa or not sb: + return 0.0 + return len(sa & sb) / len(sa | sb) + + +def _level(score: float) -> str: + if score >= 0.55: + return "very_high" + if score >= 0.35: + return "high" + if score >= 0.18: + return "medium" + return "low" + + +def similarity(a: str, b: str) -> float: + return round((0.72 * _cosine(a, b)) + (0.28 * _overlap(a, b)), 4) + + +def analyze_relevance(source: PageEvidence, target: PageEvidence, context: str = "") -> RelevanceEvidence: + title = similarity(" ".join([source.title, source.h1]), " ".join([target.title, target.h1])) + headings = similarity(" ".join(source.headings), " ".join(target.headings)) + page = similarity(source.text, target.text) + ctx = similarity(context, target.text) if context else 0.0 + combined = 0.45 * page + 0.25 * title + 0.15 * headings + 0.15 * ctx + source_terms, target_terms = set(tokens(source.text)), set(tokens(target.text)) + shared = sorted(source_terms & target_terms, key=lambda t: (-len(t), t))[:15] + return RelevanceEvidence(page_similarity=round(page, 4), context_similarity=round(ctx, 4), title_similarity=round(title, 4), heading_similarity=round(headings, 4), level=_level(combined), shared_terms=shared) diff --git a/backlink_intelligence/reporting.py b/backlink_intelligence/reporting.py new file mode 100644 index 0000000..ea6df90 --- /dev/null +++ b/backlink_intelligence/reporting.py @@ -0,0 +1,26 @@ +from __future__ import annotations + +import json +from dataclasses import asdict, is_dataclass +from pathlib import Path +from typing import Any + + +def jsonable(value: Any) -> Any: + if is_dataclass(value): return asdict(value) + if isinstance(value, tuple): return list(value) + return value + + +def write_json(data: Any, path: str | Path | None = None) -> str: + text = json.dumps(data, indent=2, ensure_ascii=False, default=jsonable) + if path: Path(path).write_text(text + "\n", encoding="utf-8") + return text + + +def audit_text(result) -> str: + backlink = result.backlink + lines = ["BACKLINK INTELLIGENCE AUDIT", "", f"Source status: {result.source.status_code}", f"Target status: {result.target.status_code}", f"Link found: {'Yes' if backlink.found else 'No'}", f"Anchor: {backlink.anchor or '-'}", f"Rel attributes: {' '.join(backlink.rel) or 'follow/default'}", f"Placement: {backlink.placement}", f"Source indexable: {'Yes' if result.source.is_indexable else 'No'}", f"Page relevance: {result.relevance.level}", f"Page similarity: {result.relevance.page_similarity:.3f}", f"Context similarity: {result.relevance.context_similarity:.3f}", f"External links: {result.outbound.external_links}", f"External domains: {result.outbound.unique_external_domains}", f"Recommendation: {result.recommendation}", f"Confidence: {result.confidence}"] + if result.reasons: lines.extend(["", "Evidence:", *[f" + {v}" for v in result.reasons]]) + if result.concerns: lines.extend(["", "Review flags:", *[f" ! {v}" for v in result.concerns]]) + return "\n".join(lines) diff --git a/backlink_intelligence/safety.py b/backlink_intelligence/safety.py new file mode 100644 index 0000000..1cbe388 --- /dev/null +++ b/backlink_intelligence/safety.py @@ -0,0 +1,57 @@ +from __future__ import annotations + +import ipaddress +import socket +from urllib.parse import urlparse + + +class UnsafeURLError(RuntimeError): + pass + + +def _is_public_ip(value: str) -> bool: + ip = ipaddress.ip_address(value) + return not ( + ip.is_private + or ip.is_loopback + or ip.is_link_local + or ip.is_multicast + or ip.is_reserved + or ip.is_unspecified + ) + + +def validate_public_url(url: str, *, resolve_dns: bool = True) -> str: + parsed = urlparse(url) + if parsed.scheme not in {"http", "https"}: + raise UnsafeURLError("Only http:// and https:// URLs are supported.") + if not parsed.hostname: + raise UnsafeURLError("URL must include a hostname.") + if parsed.username or parsed.password: + raise UnsafeURLError("Credentials embedded in URLs are not allowed.") + + host = parsed.hostname.strip("[]").lower() + if host in {"localhost", "localhost.localdomain"} or host.endswith(".local"): + raise UnsafeURLError("Local/private hostnames are not allowed.") + + try: + if not _is_public_ip(host): + raise UnsafeURLError("Private, loopback, reserved, or link-local IPs are not allowed.") + return url + except ValueError: + pass + + if resolve_dns: + try: + infos = socket.getaddrinfo(host, parsed.port or (443 if parsed.scheme == "https" else 80)) + except socket.gaierror as exc: + raise UnsafeURLError(f"Hostname could not be resolved: {host}") from exc + addresses = {info[4][0] for info in infos} + if not addresses: + raise UnsafeURLError(f"Hostname could not be resolved: {host}") + for address in addresses: + if not _is_public_ip(address): + raise UnsafeURLError( + f"Hostname resolves to a non-public address ({address}); request blocked." + ) + return url diff --git a/docs/architecture.md b/docs/architecture.md index 6270ace..39a6b1b 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -1,114 +1,36 @@ # Architecture -Backlink Intelligence is intended to have one reusable analysis engine with multiple interfaces. +Backlink Intelligence separates analysis logic from interfaces so the same engine can power the CLI, automation, and a later website integration. ```text - Backlink Intelligence Engine - | - +-------------------+-------------------+ - | | | - CLI Local Web UI Python Package - | | | - CSV/JSON Browser Integrations +CLI / CSV / future web API + | + v + Backlink Intelligence engine + | + +-------+-----------------------------+ + | | | | | +Fetcher Parser Relevance Placement Monitoring + | | | | | +Safety Evidence Scoring Before/After History ``` -A later public web deployment should reuse the same engine rather than reimplement SEO logic in a separate codebase. - -## Planned engine layers - -### Fetching and safety - -Responsible for: - -- URL normalization and validation -- network safety controls -- timeouts and redirect limits -- response-size limits -- caching -- polite rate limiting - -### Extraction - -Responsible for: - -- HTML parsing -- main-content extraction -- metadata extraction -- headings -- link extraction -- context extraction - -### Evidence models - -Typed structures representing: - -- source page -- target page -- detected backlink -- surrounding context -- link attributes -- crawl/indexability observations - -### Relevance - -Responsible for explainable relevance signals between: - -- source and target, -- context and target, -- and later sampled domain/topic evidence. - -### Placement - -Responsible for: - -- candidate paragraph ranking -- anchor fit -- insertion strategy selection -- Before/After recommendations -- editorial intervention measurement -- rejection reasons - -### Qualification - -Combines evidence into transparent review dimensions such as: - -- relevance -- destination fit -- editorial fit -- review signals -- confidence -- recommendation - -### Monitoring - -Responsible for: - -- baseline snapshots -- change detection -- retention history -- status transitions - -### Reporting - -Intended output formats: - -- terminal -- JSON -- CSV -- HTML - -## Public website phase - -The future public architecture should resemble: - -```text -alokblog.com/tools/backlink-intelligence/ - | - Web front end - | - Secure analysis API - | - backlink_intelligence package -``` - -The public service requires stronger controls than the local CLI, including SSRF defenses, abuse prevention, rate limiting, request isolation, and bounded resource usage. +## Modules + +- `safety.py`: validates public HTTP/HTTPS destinations and blocks private-network targets. +- `fetcher.py`: bounded fetches, redirect checks, response-size limits, HTML-only processing. +- `html_utils.py`: standard-library HTML extraction. +- `models.py`: typed dataclasses used across commands. +- `link_analysis.py`: URL normalization, backlink detection, outbound-link evidence. +- `relevance.py`: deterministic similarity and shared-term analysis. +- `audit.py`: combines page, backlink, relevance, and outbound evidence. +- `qualify.py`: batch prospect decision workflow. +- `placement.py`: paragraph ranking and Before/After drafts. +- `monitor.py`: persisted baselines and change detection. +- `portfolio.py`: anchor, destination, and placement distributions. +- `reporting.py`: terminal/JSON presentation helpers. +- `cli.py`: command-line interface. + +## Interface rule + +Business logic must not be embedded in the CLI. The future alokblog.com backend should import these modules directly, preserving one source of truth. diff --git a/docs/ethical-crawling.md b/docs/ethical-crawling.md index d04dc15..4c0c4d0 100644 --- a/docs/ethical-crawling.md +++ b/docs/ethical-crawling.md @@ -1,22 +1,21 @@ -# Ethical Crawling +# Ethical crawling -Backlink Intelligence should collect only the information required for the requested analysis and should avoid unnecessary load on third-party websites. +Backlink Intelligence is designed for controlled SEO analysis, not high-volume indiscriminate crawling. -## Principles +## Built-in boundaries -- Use a transparent, configurable user agent. -- Apply conservative per-host rate limits. -- Use timeouts and bounded retries. -- Avoid repeatedly fetching identical resources when cached evidence is sufficient. -- Bound response sizes and redirect chains. -- Do not attempt to bypass authentication, paywalls, CAPTCHAs, or access controls. -- Do not use the project as a generic proxy or content-copying service. -- Treat bulk analysis as a queue, not as unbounded concurrency. +- HTTP/HTTPS only +- identifiable user agent +- timeouts +- response-size limits +- redirect limits +- private-network blocking +- HTML-only processing +- repeated-URL caching inside bulk commands +- configurable delay between bulk rows -## Robots and site policies +## User responsibility -Crawling behavior should be documented and configurable. Users remain responsible for complying with applicable laws, contractual terms, and website policies in their jurisdiction and use case. +Users should respect website terms, applicable laws, access controls, and reasonable request rates. Do not use the tool to bypass authentication, anti-bot controls, or technical restrictions. -## Data minimization - -Reports should store only the evidence required for analysis. A future public web service should avoid permanently storing user-submitted URLs and extracted content unless storage is clearly disclosed and necessary. +For bulk prospect work, keep lists targeted and avoid repeatedly requesting the same website at aggressive rates. A hosted version should add centralized rate limiting, caching, concurrency control, and abuse prevention. diff --git a/docs/limitations.md b/docs/limitations.md index d0e87ff..4e1f682 100644 --- a/docs/limitations.md +++ b/docs/limitations.md @@ -1,35 +1,27 @@ # Limitations -Backlink Intelligence is intended to assist SEO review, not replace professional judgment. +Backlink Intelligence intentionally avoids claims the available evidence cannot support. -## Search-engine behavior - -The tool cannot determine exactly how Google or another search engine values an individual backlink. It does not reproduce proprietary ranking systems. - -## Third-party authority metrics +## JavaScript-rendered content -The core project does not require DA, DR, Authority Score, Trust Flow, or similar proprietary metrics. Optional integrations may expose third-party data later when users provide their own credentials. +The v1 core fetches server-delivered HTML and does not bundle a browser engine. Links or text created only after JavaScript execution may not be visible to the parser. -## Rendering - -Some websites depend heavily on client-side JavaScript. A lightweight HTTP crawler may not observe exactly what a browser renders. The project should report reduced confidence rather than silently treating extraction as complete. - -## Crawl restrictions +## Search-engine behavior -Websites may block automated requests, require authentication, rate-limit clients, or expose content differently by region or user agent. +The project cannot determine how Google or another search engine values a particular backlink. Recommendations are workflow decisions derived from observable evidence. -## Semantic analysis +## Relevance -Relevance estimates are approximations. Local statistical models, embeddings, or language models can all misunderstand context. Results should expose supporting evidence and confidence. +The v1 relevance engine is deterministic lexical analysis. It does not fully capture semantics, multilingual nuance, irony, or deep entity relationships. -## Placement recommendations +## Placement classification -A suggested Before/After paragraph is not permission to modify another publisher's content. It is a drafting aid for editorial review. The final placement should make sense to the publisher and reader. +Semantic HTML landmarks improve classification. Poorly structured pages may be returned as `unknown` even when a human can identify the placement visually. -## Historical monitoring +## Before/After drafts -A missing or changed link may be temporary. Monitoring should distinguish observation from cause and preserve timestamps for later review. +The deterministic rewrite logic is deliberately conservative. It should be reviewed by an editor before use. It is not intended to imitate unrestricted generative copywriting. -## Public web deployment +## Authority metrics and traffic -The local project and a hosted public service have different privacy, abuse, security, and operating-cost considerations. Hosted functionality should be introduced only with appropriate controls. +The zero-cost core does not fetch proprietary DA/DR/traffic datasets. Optional third-party enrichment can be implemented separately when users have lawful access to those services. diff --git a/docs/methodology.md b/docs/methodology.md index f668d60..fb5e745 100644 --- a/docs/methodology.md +++ b/docs/methodology.md @@ -1,93 +1,51 @@ # Methodology -Backlink Intelligence is designed around one principle: +Backlink Intelligence follows an evidence-first methodology. It does not infer a proprietary search-engine score. -> **Evaluate observable link evidence and editorial context instead of treating a single third-party authority metric as the final decision.** +## 1. Source accessibility and indexability -## What the project evaluates +The tool records response status, final URL, canonical URL, and robots directives. These are observable technical signals. -The methodology is organized around five evidence groups. +## 2. Link evidence -### 1. Source-page evidence +For an existing backlink it records: -Examples include: +- whether the target URL is present, +- anchor text, +- `rel` attributes, +- surrounding paragraph/context, +- broad placement classification. -- HTTP availability and redirects -- title, H1, and main content -- canonical and robots directives -- approximate content depth -- external-link counts and unique external domains -- outbound-link density and neighborhood +## 3. Relevance -### 2. Link evidence +The deterministic v1 engine compares source and destination using term-frequency cosine similarity plus normalized term overlap. It reports separate page, title/H1, heading, and contextual similarities. -Examples include: +These similarity values are workflow evidence, not ranking probabilities. -- whether the target link exists -- anchor text -- link type and attributes -- surrounding sentence and paragraph -- approximate location within the document -- whether the link appears to be inside editorial main content +## 4. Outbound-link behavior -### 3. Relevance evidence +The tool measures external links, unique external domains, link density, and rel-attribute distributions. High-density patterns are review flags, not a "toxic link" verdict. -Relevance is intended to be measured at several levels rather than as one binary label: +## 5. Prospect qualification -- broader site/domain topic signals -- source page ↔ destination page -- local link context ↔ destination page +The qualifier combines availability, indexability, relevance, placement potential, and review flags into one of three operational recommendations: -Initial implementations should prefer explainable local methods. More advanced semantic models may be added as optional components later. +- prioritize +- manual review +- low priority -### 4. Editorial-fit evidence +The recommendation is intentionally explainable and reversible by a human reviewer. -For proposed link placements, the tool should examine: +## 6. Contextual placement -- whether the target genuinely expands the source discussion -- whether the preferred anchor is grammatically and semantically natural -- how much original publisher text would need to change -- whether a new sentence is more appropriate than forced insertion -- whether the candidate paragraph should be rejected entirely +Paragraphs are ranked by similarity to the target page with a smaller anchor-term overlap component. Very short and extremely long paragraphs are excluded from candidate generation. -### 5. Lifecycle evidence +Before/After output prioritizes preservation of publisher text. If the anchor already appears naturally, only that phrase is linked. Otherwise a conservative contextual sentence is appended. -After a backlink is acquired, monitoring may examine: +## 7. Monitoring -- continued link presence -- anchor changes -- `rel` changes -- target changes -- source/target redirects -- HTTP availability -- canonical/noindex changes -- material placement changes +A monitoring baseline captures the observable state of a link. Future checks compare link existence, anchor, rel attributes, placement, source/target status, canonical, and robots directives. -## Recommendation model +## Human review -The project should expose dimensions such as: - -- relevance, -- editorial fit, -- risk/review signals, -- confidence, -- and recommendation. - -A future recommendation such as **Prioritize** should always include the reasons that support it and the concerns that might require human review. - -## What the methodology does not claim - -Backlink Intelligence does not claim to: - -- know how Google values a specific backlink, -- reproduce PageRank or search-engine ranking systems, -- determine whether a site is "toxic" from a proprietary formula, -- guarantee rankings, -- guarantee penalties, -- or replace professional review. - -## Background article - -The conceptual motivation is discussed in: - -[Backlink Quality Beyond DA & DR](https://alokblog.com/backlink-quality-beyond-da-dr/) +All outputs are aids for SEO and editorial review. The tool does not automatically alter external pages, send outreach, purchase links, or publish placements. diff --git a/docs/roadmap.md b/docs/roadmap.md index 60d7850..7b59faa 100644 --- a/docs/roadmap.md +++ b/docs/roadmap.md @@ -1,168 +1,45 @@ -# Development Roadmap - -The project is intentionally incremental. Each milestone should be useful, tested, and documented before broader functionality is added. - -## Foundation: `v0.0.1` - -**Status: Current foundation** - -- repository structure -- methodology -- architecture -- security guidance -- crawl-safety principles -- contribution guidance -- minimal installable CLI -- CI for deterministic foundation tests - -## `v0.1.0` — Backlink Evidence Auditor - -Input: - -- source URL -- target URL - -Planned output: - -- source HTTP status and final URL -- title and H1 -- canonical and robots directives -- backlink found/not found -- anchor text -- link attributes -- surrounding sentence -- surrounding paragraph -- basic placement classification -- terminal and JSON output - -Engineering priorities: - -- URL validation -- SSRF-aware network boundaries -- bounded redirects/timeouts/response size -- deterministic HTML fixtures -- clean evidence models - -## `v0.2.0` — Context and Placement Classification - -- main-content detection -- editorial context -- resource list -- author bio -- navigation -- sidebar -- footer -- comments/UGC -- sponsored areas -- unknown with confidence - -## `v0.3.0` — Relevance Engine - -- title/H1/heading alignment -- important-term overlap -- TF-IDF -- cosine similarity -- page ↔ target relevance -- context ↔ target relevance -- explainable relevance evidence - -## `v0.4.0` — Source Quality and Outbound-Link Evidence - -- external-link counts -- unique external domains -- link density -- follow/nofollow distribution -- outbound neighborhood -- thin-content indicators -- indexability evidence -- review flags - -## `v0.5.0` — Bulk Prospect Qualification - -- CSV input -- controlled crawl queue -- prospect prioritization -- destination fit -- evidence summaries -- manual-review reasons -- CSV and JSON reports - -## `v0.6.0` — Contextual Placement Recommender - -Input: - -- source article -- target page -- preferred anchor - -Capabilities: - -- paragraph ranking -- top placement opportunities -- anchor fit -- reject unsuitable paragraphs -- explain why each candidate was selected - -## `v0.7.0` — Before/After Placement Recommendations - -- minimal insertion strategy -- contextual sentence strategy -- paragraph refinement strategy -- requested vs suggested anchor -- Before paragraph -- After paragraph -- editorial intervention level -- original-text preservation -- placement brief export - -This milestone is the minimum target before beginning the public website integration in parallel. - -## `v0.8.0` — Backlink Monitoring - -- baseline snapshots -- backlink removal -- anchor changes -- follow/nofollow changes -- sponsored/UGC changes -- destination changes -- redirects -- 404/410 -- noindex/canonical changes -- historical change records - -## `v0.9.0` — Portfolio and History - -- anchor-category distribution -- destination distribution -- placement distribution -- retention metrics -- historical comparisons -- manual-review queues - -## `v1.0.0` — Stable Toolkit - -Target qualities: - -- stable CLI -- documented Python API -- CSV/JSON/HTML reporting -- clean install path -- comprehensive deterministic tests -- production-quality URL safety -- Docker packaging if useful -- complete methodology and limitations - -## Phase 2 — Public Website - -Once audit, relevance, placement, and Before/After recommendations are stable, build a public-facing version at a path such as: - -`https://alokblog.com/tools/backlink-intelligence/` - -Initial public scope should favor: - -- single backlink audit -- single placement analysis -- top placement recommendations -- Before/After output - -Bulk analysis and continuous monitoring should remain local-first initially to control cost and abuse. +# Roadmap + +## v1.0.0 core toolkit + +Status: **Implemented** + +- Backlink audit +- Page/link evidence extraction +- Placement classification +- Deterministic relevance analysis +- Outbound-link evidence +- Bulk prospect qualification +- Contextual placement ranking +- Before/After placement recommendations +- Monitoring and historical baseline comparison +- Portfolio distribution analysis +- CLI, CSV, and JSON workflows +- Crawl-safety controls +- Offline tests and CI + +## Phase 2: alokblog.com integration + +Planned after the GitHub engine has been validated on a broader set of real-world pages. + +Public web beta scope: + +1. Single backlink audit +2. Find link placement +3. Before/After recommendation +4. Rate limiting and abuse controls +5. Secure API boundary around the same Python engine + +## Possible future enhancements + +These are intentionally outside the v1 core and should remain optional: + +- JavaScript/browser rendering adapter +- Local browser UI +- HTML/PDF presentation reports +- Optional local embeddings +- Optional LLM rewrite provider +- Optional Ahrefs/Semrush/Moz/Majestic enrichment adapters +- Site-wide outbound-link sampling +- Scheduled monitoring service +- Hosted accounts/workspaces diff --git a/docs/signal-definitions.md b/docs/signal-definitions.md index f18affb..c22d22e 100644 --- a/docs/signal-definitions.md +++ b/docs/signal-definitions.md @@ -1,82 +1,39 @@ -# Signal Definitions +# Signal definitions -This document defines terminology used by Backlink Intelligence. Exact algorithms may evolve, but public output should retain clear definitions. +## Relevance levels -## Page relevance +Relevance combines deterministic page/context similarity signals and is represented as: -How strongly the source page's subject matter aligns with the target page. +- `low` +- `medium` +- `high` +- `very_high` -Planned labels: +Thresholds are intentionally visible in the source code and may evolve with benchmark data. -- Very Low -- Low -- Medium -- High -- Very High +## Placement values -## Context relevance +- `editorial_context`: link detected inside `
` or `
`. +- `navigation`: link detected inside `