From 9da7a2a5ba58b8d3e2ab53b254d9e9a940a9d75b Mon Sep 17 00:00:00 2001 From: Codex Date: Sun, 30 Aug 2026 13:45:04 +0530 Subject: [PATCH] Release Backlink Intelligence public API v1.1.0 --- .env.example | 15 + .github/workflows/ci.yml | 2 +- CHANGELOG.md | 9 + README.md | 15 + backlink_intelligence/__init__.py | 2 +- backlink_intelligence/analysis.py | 126 +++++++ backlink_intelligence/api.py | 531 +++++++++++++++++++++++++++++ backlink_intelligence/cli.py | 13 +- backlink_intelligence/fetcher.py | 316 ++++++++++++++--- backlink_intelligence/models.py | 17 + backlink_intelligence/placement.py | 116 +++++-- backlink_intelligence/safety.py | 100 +++++- pyproject.toml | 12 +- render.yaml | 40 +++ tests/test_analysis.py | 58 ++++ tests/test_api.py | 204 +++++++++++ tests/test_cli.py | 2 +- tests/test_fetcher_security.py | 26 ++ 18 files changed, 1509 insertions(+), 95 deletions(-) create mode 100644 .env.example create mode 100644 backlink_intelligence/analysis.py create mode 100644 backlink_intelligence/api.py create mode 100644 render.yaml create mode 100644 tests/test_analysis.py create mode 100644 tests/test_api.py create mode 100644 tests/test_fetcher_security.py diff --git a/.env.example b/.env.example new file mode 100644 index 0000000..500e051 --- /dev/null +++ b/.env.example @@ -0,0 +1,15 @@ +BI_ENVIRONMENT=development +BI_ALLOWED_ORIGINS=http://localhost:8787,https://alokblog.com +BI_ALLOWED_HOSTS=localhost,127.0.0.1,api.alokblog.com +BI_TURNSTILE_HOSTNAME=alokblog.com +BI_TURNSTILE_SECRET= +BI_MIN_CONTEXT_SCORE=0.15 +BI_MIN_DESTINATION_SCORE=0.08 +BI_MAX_OPPORTUNITIES=3 +BI_RATE_LIMIT_PER_HOUR=5 +BI_MAX_CONCURRENCY=2 +BI_ANALYSIS_TIMEOUT=35 +BI_CONNECT_TIMEOUT=5 +BI_READ_TIMEOUT=12 +BI_MAX_COMPRESSED_BYTES=1000000 +BI_MAX_DECOMPRESSED_BYTES=2000000 diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 51212b6..233615b 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -23,7 +23,7 @@ jobs: with: python-version: ${{ matrix.python-version }} - name: Install package - run: python -m pip install . + run: python -m pip install ".[api,test]" - name: Compile package run: python -m compileall -q backlink_intelligence - name: Run tests diff --git a/CHANGELOG.md b/CHANGELOG.md index 7f21fc9..20262b3 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,14 @@ # Changelog +## 1.1.0 - 2026-08-30 + +- Add the shared placement analysis service and FastAPI v1 interface. +- Add genuine no-match and editorial-review outcomes. +- Return Unicode-safe structured text and link segments instead of browser offsets. +- Pin outbound connections to validated public IP addresses across redirects and robots requests. +- Add compressed and decompressed response limits, strict API validation, Turnstile verification, rate limits, concurrency controls, and production Host/origin enforcement. +- Add Render Free deployment configuration and v1 contract tests. + All notable changes to Backlink Intelligence are documented here. ## 1.0.1 - 2026-08-30 diff --git a/README.md b/README.md index 5317b89..8f6c2ad 100644 --- a/README.md +++ b/README.md @@ -322,3 +322,18 @@ MIT. See [LICENSE](LICENSE). ## Disclaimer Backlink Intelligence is an independent open-source SEO research and workflow tool. It is not affiliated with Google, Ahrefs, Semrush, Moz, Majestic, or any other search engine or SEO platform. Outputs should be treated as evidence for professional review, not as guarantees of ranking impact, penalties, or search-engine behavior. + +## Public beta API + +Version 1.1 adds an optional FastAPI service for the public Backlink Placement Analyzer. Install it with: + +```bash +pip install ".[api]" +uvicorn backlink_intelligence.api:app --host 127.0.0.1 --port 8000 +``` + +The public endpoint is `POST /v1/place`. It accepts a source URL, target URL, preferred anchor, and Cloudflare Turnstile token. A valid analysis returns either `completed` or `no_suitable_placement`; the latter is a successful HTTP 200 outcome, not an API error. + +Generated copy is returned as plain `after_text` plus `after_segments` containing only text and link records. The API never returns executable markup. Numeric scores are internal ranking evidence and must not be presented as probabilities, authority scores, ranking potential, or percentage quality. Beta thresholds are private server configuration and are not included in responses. + +Production deployment settings are documented in `render.yaml` and `.env.example`. Free hosting has capacity, cold-start, and bandwidth limits; the service does not require a database, paid SEO data, or an LLM API. diff --git a/backlink_intelligence/__init__.py b/backlink_intelligence/__init__.py index 35fbf6a..a05836f 100644 --- a/backlink_intelligence/__init__.py +++ b/backlink_intelligence/__init__.py @@ -1,3 +1,3 @@ """Backlink Intelligence package.""" -__version__ = "1.0.1" +__version__ = "1.1.0" diff --git a/backlink_intelligence/analysis.py b/backlink_intelligence/analysis.py new file mode 100644 index 0000000..43e35c6 --- /dev/null +++ b/backlink_intelligence/analysis.py @@ -0,0 +1,126 @@ +from __future__ import annotations + +import os +from dataclasses import dataclass, field + +from .fetcher import FetchConfig, fetch_page +from .models import PageEvidence, PlacementSuggestion +from .placement import rank_placements +from .safety import validate_public_url + + +def _float_setting( + name: str, default: float, minimum: float = 0.0, maximum: float = 1.0 +) -> float: + try: + value = float(os.getenv(name, str(default))) + except ValueError: + return default + return value if minimum <= value <= maximum else default + + +def _int_setting(name: str, default: int, minimum: int, maximum: int) -> int: + try: + value = int(os.getenv(name, str(default))) + except ValueError: + return default + return min(max(value, minimum), maximum) + + +@dataclass(slots=True) +class AnalysisConfig: + min_context_score: float = 0.15 + min_destination_score: float = 0.08 + max_opportunities: int = 3 + fetch: FetchConfig = field(default_factory=FetchConfig) + + @classmethod + def from_environment(cls) -> "AnalysisConfig": + fetch = FetchConfig( + connect_timeout=_float_setting("BI_CONNECT_TIMEOUT", 5.0, 1.0, 15.0), + read_timeout=_float_setting("BI_READ_TIMEOUT", 12.0, 2.0, 30.0), + max_compressed_bytes=_int_setting( + "BI_MAX_COMPRESSED_BYTES", 1_000_000, 64_000, 2_000_000 + ), + max_decompressed_bytes=_int_setting( + "BI_MAX_DECOMPRESSED_BYTES", 2_000_000, 128_000, 4_000_000 + ), + max_redirects=_int_setting("BI_MAX_REDIRECTS", 3, 0, 3), + ) + return cls( + min_context_score=_float_setting("BI_MIN_CONTEXT_SCORE", 0.15), + min_destination_score=_float_setting("BI_MIN_DESTINATION_SCORE", 0.08), + max_opportunities=_int_setting("BI_MAX_OPPORTUNITIES", 3, 1, 3), + fetch=fetch, + ) + + +@dataclass(slots=True) +class PlacementAnalysis: + status: str + source: PageEvidence + target: PageEvidence + opportunities: list[PlacementSuggestion] + analysis_warnings: list[str] = field(default_factory=list) + + +class PlacementAnalyzer: + """One analysis service shared by the API and command-line interface.""" + + def __init__(self, config: AnalysisConfig | None = None) -> None: + self.config = config or AnalysisConfig.from_environment() + + def analyze( + self, + source_url: str, + target_url: str, + preferred_anchor: str, + *, + max_opportunities: int | None = None, + ) -> PlacementAnalysis: + source_url = validate_public_url(source_url, resolve_dns=False) + target_url = validate_public_url(target_url, resolve_dns=False) + source = fetch_page(source_url, self.config.fetch) + target = fetch_page(target_url, self.config.fetch) + if source.status_code != 200 or target.status_code != 200: + return PlacementAnalysis( + status="failed", + source=source, + target=target, + opportunities=[], + ) + + top_n = self.config.max_opportunities + if max_opportunities is not None: + top_n = min(max(max_opportunities, 1), self.config.max_opportunities) + opportunities = rank_placements( + source, + target, + preferred_anchor, + target_url, + top_n=top_n, + min_context_score=self.config.min_context_score, + min_destination_score=self.config.min_destination_score, + ) + warnings: list[str] = [] + if not source.is_indexable: + warnings.append("source_page_is_not_indexable") + if not target.is_indexable: + warnings.append("target_page_is_not_indexable") + return PlacementAnalysis( + status="completed" if opportunities else "no_suitable_placement", + source=source, + target=target, + opportunities=opportunities, + analysis_warnings=warnings, + ) + + +def analyze_placement( + source_url: str, + target_url: str, + preferred_anchor: str, + *, + config: AnalysisConfig | None = None, +) -> PlacementAnalysis: + return PlacementAnalyzer(config).analyze(source_url, target_url, preferred_anchor) diff --git a/backlink_intelligence/api.py b/backlink_intelligence/api.py new file mode 100644 index 0000000..73271d4 --- /dev/null +++ b/backlink_intelligence/api.py @@ -0,0 +1,531 @@ +from __future__ import annotations + +import asyncio +import hashlib +import json +import logging +import os +import time +import uuid +from collections import defaultdict, deque +from concurrent.futures import ThreadPoolExecutor +from datetime import datetime, timezone +from typing import Annotated, Callable, Literal +from urllib.parse import urlencode +from urllib.request import Request as URLRequest +from urllib.request import urlopen + +from fastapi import Depends, FastAPI, Request +from fastapi.exceptions import RequestValidationError +from fastapi.responses import JSONResponse, Response +from pydantic import BaseModel, ConfigDict, Field, model_validator + +from . import __version__ +from .analysis import AnalysisConfig, PlacementAnalyzer +from .models import PageEvidence, PlacementSuggestion, TextSegment +from .safety import UnsafeURLError, validate_public_url + +API_VERSION = "1" +LOGGER = logging.getLogger("backlink_intelligence.api") +LOGGER.setLevel(logging.INFO) + + +def _csv_environment(name: str, default: str) -> tuple[str, ...]: + return tuple( + value.strip().lower() + for value in os.getenv(name, default).split(",") + if value.strip() + ) + + +class APISettings: + def __init__(self) -> None: + self.environment = os.getenv("BI_ENVIRONMENT", "development").strip().lower() + self.allowed_origins = _csv_environment( + "BI_ALLOWED_ORIGINS", "https://alokblog.com" + ) + self.allowed_hosts = _csv_environment("BI_ALLOWED_HOSTS", "api.alokblog.com") + self.turnstile_secret = os.getenv("BI_TURNSTILE_SECRET", "").strip() + self.turnstile_hostname = os.getenv( + "BI_TURNSTILE_HOSTNAME", "alokblog.com" + ).strip().lower() + self.rate_limit = self._integer("BI_RATE_LIMIT_PER_HOUR", 5, 1, 100) + self.max_concurrency = self._integer("BI_MAX_CONCURRENCY", 2, 1, 8) + self.analysis_timeout = self._integer("BI_ANALYSIS_TIMEOUT", 35, 10, 90) + + @staticmethod + def _integer(name: str, default: int, minimum: int, maximum: int) -> int: + try: + value = int(os.getenv(name, str(default))) + except ValueError: + return default + return min(max(value, minimum), maximum) + + @property + def production(self) -> bool: + return self.environment == "production" + + +SETTINGS = APISettings() + + +class APIError(Exception): + def __init__(self, status_code: int, code: str, message: str) -> None: + super().__init__(message) + self.status_code = status_code + self.code = code + self.message = message + + +URLField = Annotated[str, Field(min_length=10, max_length=2048)] + + +class PlaceRequest(BaseModel): + model_config = ConfigDict(extra="forbid", str_strip_whitespace=True) + + source_url: URLField + target_url: URLField + anchor: Annotated[str, Field(min_length=2, max_length=120)] + challenge_token: Annotated[str, Field(min_length=1, max_length=4096)] + + @model_validator(mode="after") + def validate_urls(self) -> "PlaceRequest": + try: + source = validate_public_url(self.source_url, resolve_dns=False) + target = validate_public_url(self.target_url, resolve_dns=False) + except UnsafeURLError as exc: + raise ValueError(str(exc)) from exc + if source == target: + raise ValueError("Source URL and target URL must be different.") + self.source_url = source + self.target_url = target + return self + + +class SegmentResponse(BaseModel): + type: Literal["text", "link"] + text: str + url: str | None = None + + @model_validator(mode="after") + def validate_segment(self) -> "SegmentResponse": + if not self.text: + raise ValueError("Segment text must not be empty.") + if self.type == "link": + if not self.url: + raise ValueError("Link segments require a URL.") + self.url = validate_public_url(self.url, resolve_dns=False) + elif self.url is not None: + raise ValueError("Text segments must not include a URL.") + return self + + +class OpportunityResponse(BaseModel): + rank: int + paragraph_index: int + score: float + context_level: Literal["low", "medium", "high", "very_high"] + destination_score: float + destination_fit: Literal["low", "medium", "high", "very_high"] + requested_anchor: str + suggested_anchor: str + strategy: Literal["minimal_insertion", "contextual_sentence"] + recommendation_status: Literal["recommended", "manual_review"] + review_required: bool + intervention: Literal["low", "medium", "high"] + preservation_percent: float + before_text: str + after_text: str + after_segments: list[SegmentResponse] + reasons: list[str] + warnings: list[str] + + @model_validator(mode="after") + def validate_segments(self) -> "OpportunityResponse": + if not self.after_segments: + raise ValueError("Structured after_segments are required.") + if "".join(segment.text for segment in self.after_segments) != self.after_text: + raise ValueError("Structured segments must reconstruct after_text exactly.") + if sum(segment.type == "link" for segment in self.after_segments) != 1: + raise ValueError("Exactly one link segment is required.") + if self.review_required != (self.recommendation_status == "manual_review"): + raise ValueError("Recommendation status and review requirement disagree.") + return self + + +class PageSummary(BaseModel): + url: str + final_url: str + title: str + status_code: int + + +class PlaceResponse(BaseModel): + success: Literal[True] = True + status: Literal["completed", "no_suitable_placement"] + request_id: str + api_version: Literal["1"] = "1" + engine_version: str + source: PageSummary + target: PageSummary + opportunities: list[OpportunityResponse] + analysis_warnings: list[str] + + @model_validator(mode="after") + def validate_outcome(self) -> "PlaceResponse": + if len(self.opportunities) > 3: + raise ValueError("The public beta returns at most three opportunities.") + if self.status == "no_suitable_placement" and self.opportunities: + raise ValueError("A no-match response must not contain opportunities.") + if self.status == "completed" and not self.opportunities: + raise ValueError("A completed response must contain an opportunity.") + for opportunity in self.opportunities: + link_url = next( + segment.url + for segment in opportunity.after_segments + if segment.type == "link" + ) + if link_url != self.target.url: + raise ValueError("The structured link URL must equal the target URL.") + return self + + +def _error_payload(request_id: str, code: str, message: str) -> dict: + return { + "success": False, + "request_id": request_id, + "api_version": API_VERSION, + "error": {"code": code, "message": message}, + } + + +class InMemoryProtection: + def __init__(self) -> None: + self.rate_windows: dict[str, deque[float]] = defaultdict(deque) + self.active_ips: set[str] = set() + self.lock = asyncio.Lock() + self.semaphore = asyncio.Semaphore(SETTINGS.max_concurrency) + self.salt = os.urandom(32) + + def key(self, request: Request) -> str: + address = request.client.host if request.client else "unknown" + return hashlib.sha256(self.salt + address.encode("utf-8", "replace")).hexdigest() + + async def reserve(self, key: str) -> None: + now = time.monotonic() + async with self.lock: + window = self.rate_windows[key] + while window and now - window[0] >= 3600: + window.popleft() + if len(window) >= SETTINGS.rate_limit: + raise APIError(429, "rate_limited", "The hourly analysis limit has been reached.") + if key in self.active_ips: + raise APIError(503, "service_busy", "An analysis is already running for this connection.") + self.active_ips.add(key) + + try: + await asyncio.wait_for(self.semaphore.acquire(), timeout=0.05) + except TimeoutError as exc: + async with self.lock: + self.active_ips.discard(key) + raise APIError(503, "service_busy", "The analysis service is currently busy.") from exc + + async with self.lock: + self.rate_windows[key].append(now) + + async def release(self, key: str) -> None: + self.semaphore.release() + async with self.lock: + self.active_ips.discard(key) + + +PROTECTION = InMemoryProtection() +ANALYSIS_EXECUTOR = ThreadPoolExecutor( + max_workers=SETTINGS.max_concurrency, + thread_name_prefix="backlink-analysis", +) +USED_CHALLENGES: dict[str, float] = {} + + +def _verify_turnstile_remote(token: str) -> dict: + body = urlencode( + {"secret": SETTINGS.turnstile_secret, "response": token} + ).encode("ascii") + request = URLRequest( + "https://challenges.cloudflare.com/turnstile/v0/siteverify", + data=body, + headers={"Content-Type": "application/x-www-form-urlencoded"}, + method="POST", + ) + with urlopen(request, timeout=8) as response: + raw = response.read(32_769) + if len(raw) > 32_768: + raise APIError(503, "challenge_failed", "Challenge verification was unavailable.") + return json.loads(raw.decode("utf-8")) + + +def verify_challenge(token: str) -> bool: + if not SETTINGS.turnstile_secret: + return not SETTINGS.production + + token_hash = hashlib.sha256(token.encode("utf-8", "replace")).hexdigest() + now = time.time() + for old_hash, expires_at in list(USED_CHALLENGES.items()): + if expires_at <= now: + del USED_CHALLENGES[old_hash] + if token_hash in USED_CHALLENGES: + return False + + try: + result = _verify_turnstile_remote(token) + challenge_time = datetime.fromisoformat( + str(result.get("challenge_ts", "")).replace("Z", "+00:00") + ) + except (APIError, OSError, ValueError, TypeError, json.JSONDecodeError): + return False + age = datetime.now(timezone.utc) - challenge_time.astimezone(timezone.utc) + valid = bool( + result.get("success") + and str(result.get("hostname", "")).lower() == SETTINGS.turnstile_hostname + and str(result.get("action", "")) == "backlink_intelligence" + and -60 <= age.total_seconds() <= 300 + ) + if valid: + USED_CHALLENGES[token_hash] = now + 300 + return valid + + +def challenge_verifier() -> Callable[[str], bool]: + return verify_challenge + + +async def _release_when_finished(future: asyncio.Future, key: str) -> None: + try: + await future + except BaseException: + pass + finally: + await PROTECTION.release(key) + + +def _page_summary(page: PageEvidence) -> PageSummary: + return PageSummary( + url=page.requested_url, + final_url=page.final_url, + title=page.title, + status_code=page.status_code, + ) + + +def _segment_response(segment: TextSegment) -> SegmentResponse: + return SegmentResponse( + type=segment.type, + text=segment.text, + url=segment.url if segment.type == "link" else None, + ) + + +def _opportunity_response( + item: PlacementSuggestion, normalized_target_url: str +) -> OpportunityResponse: + response = OpportunityResponse( + rank=item.rank, + paragraph_index=item.paragraph_index, + score=item.score, + context_level=item.context_level, + destination_score=item.destination_score, + destination_fit=item.destination_fit, + requested_anchor=item.requested_anchor, + suggested_anchor=item.suggested_anchor, + strategy=item.strategy, + recommendation_status=item.recommendation_status, + review_required=item.review_required, + intervention=item.intervention, + preservation_percent=item.preservation_percent, + before_text=item.before, + after_text=item.after_text, + after_segments=[_segment_response(segment) for segment in item.after_segments], + reasons=item.reasons, + warnings=item.warnings, + ) + link_url = next( + segment.url for segment in response.after_segments if segment.type == "link" + ) + if link_url != normalized_target_url: + raise APIError( + 500, + "internal_error", + "The analysis returned an invalid target link.", + ) + return response + + +def _analysis_failure(source: PageEvidence, target: PageEvidence) -> APIError: + page = source if source.status_code != 200 else target + role = "source" if page is source else "target" + error = page.error + if "RobotsDeniedError" in error: + return APIError(403, "crawl_blocked", "The publisher's robots policy blocks analysis.") + if "UnsupportedContentError" in error: + return APIError(415, "unsupported_content", "The page is not supported HTML content.") + if "UnsafeURLError" in error: + return APIError(400, "unsafe_url", "The submitted URL did not pass the public-network safety check.") + code = "source_unavailable" if role == "source" else "target_unavailable" + return APIError(422, code, f"The {role} page could not be analyzed.") + + +app = FastAPI( + title="Backlink Intelligence API", + version=API_VERSION, + docs_url=None if SETTINGS.production else "/docs", + redoc_url=None, + openapi_url=None if SETTINGS.production else "/openapi.json", +) + + +@app.middleware("http") +async def security_and_observability(request: Request, call_next): + request_id = uuid.uuid4().hex + request.state.request_id = request_id + started = time.monotonic() + origin = request.headers.get("origin", "").lower() + host = request.headers.get("host", "").split(":", 1)[0].lower() + + if request.url.path != "/health": + if SETTINGS.production and host not in SETTINGS.allowed_hosts: + return JSONResponse( + _error_payload(request_id, "invalid_request", "Request host is not allowed."), + status_code=400, + ) + if origin and origin not in SETTINGS.allowed_origins: + return JSONResponse( + _error_payload(request_id, "invalid_request", "Request origin is not allowed."), + status_code=403, + ) + if SETTINGS.production and not origin: + return JSONResponse( + _error_payload(request_id, "invalid_request", "Request origin is required."), + status_code=403, + ) + + if request.method == "OPTIONS": + response: Response = Response(status_code=204) + else: + response = await call_next(request) + response.headers["Cache-Control"] = "no-store" + response.headers["X-Content-Type-Options"] = "nosniff" + response.headers["Referrer-Policy"] = "no-referrer" + response.headers["X-Request-ID"] = request_id + if origin in SETTINGS.allowed_origins: + response.headers["Access-Control-Allow-Origin"] = origin + response.headers["Access-Control-Allow-Methods"] = "GET, POST, OPTIONS" + response.headers["Access-Control-Allow-Headers"] = "Content-Type" + response.headers["Vary"] = "Origin" + LOGGER.info( + "request_id=%s status=%s duration_ms=%d api_version=%s engine_version=%s", + request_id, + response.status_code, + int((time.monotonic() - started) * 1000), + API_VERSION, + __version__, + ) + return response + + +@app.exception_handler(APIError) +async def api_error_handler(request: Request, exc: APIError): + return JSONResponse( + _error_payload(request.state.request_id, exc.code, exc.message), + status_code=exc.status_code, + ) + + +@app.exception_handler(RequestValidationError) +async def validation_error_handler(request: Request, exc: RequestValidationError): + del exc + return JSONResponse( + _error_payload( + request.state.request_id, + "invalid_request", + "The request body did not match the v1 API contract.", + ), + status_code=422, + ) + + +@app.exception_handler(Exception) +async def unexpected_error_handler(request: Request, exc: Exception): + LOGGER.error( + "request_id=%s status=500 code=internal_error exception=%s", + request.state.request_id, + exc.__class__.__name__, + ) + return JSONResponse( + _error_payload( + request.state.request_id, + "internal_error", + "The analysis could not be completed.", + ), + status_code=500, + ) + + +@app.get("/health") +async def health() -> dict: + return { + "status": "ok", + "api_version": API_VERSION, + "engine_version": __version__, + } + + +@app.post("/v1/place", response_model=PlaceResponse) +async def place( + payload: PlaceRequest, + request: Request, + verifier=Depends(challenge_verifier), +) -> PlaceResponse: + request_id = request.state.request_id + if not verifier(payload.challenge_token): + raise APIError(403, "challenge_failed", "The anti-abuse challenge could not be verified.") + + key = PROTECTION.key(request) + await PROTECTION.reserve(key) + release_now = True + try: + analyzer = PlacementAnalyzer(AnalysisConfig.from_environment()) + loop = asyncio.get_running_loop() + analysis_future = loop.run_in_executor( + ANALYSIS_EXECUTOR, + analyzer.analyze, + payload.source_url, + payload.target_url, + payload.anchor, + ) + try: + analysis = await asyncio.wait_for( + asyncio.shield(analysis_future), + timeout=SETTINGS.analysis_timeout, + ) + except TimeoutError as exc: + release_now = False + asyncio.create_task(_release_when_finished(analysis_future, key)) + raise APIError(504, "analysis_timeout", "The analysis exceeded the time limit.") from exc + if analysis.status == "failed": + raise _analysis_failure(analysis.source, analysis.target) + + return PlaceResponse( + status=analysis.status, + request_id=request_id, + engine_version=__version__, + source=_page_summary(analysis.source), + target=_page_summary(analysis.target), + opportunities=[ + _opportunity_response(item, payload.target_url) + for item in analysis.opportunities + ], + analysis_warnings=analysis.analysis_warnings, + ) + finally: + if release_now: + await PROTECTION.release(key) diff --git a/backlink_intelligence/cli.py b/backlink_intelligence/cli.py index 4b45bcd..79d0bba 100644 --- a/backlink_intelligence/cli.py +++ b/backlink_intelligence/cli.py @@ -6,9 +6,9 @@ import sys from backlink_intelligence import __version__ +from .analysis import PlacementAnalyzer from .audit import audit_backlink from .monitor import monitor_csv -from .placement import suggest_placements from .portfolio import analyze_portfolio from .qualify import qualify_csv from .reporting import audit_text, write_json @@ -74,6 +74,7 @@ def _placement_text(items) -> str: f"Requested anchor: {item.requested_anchor}", f"Placed anchor: {item.suggested_anchor}", f"Strategy: {item.strategy}", + f"Recommendation status:{item.recommendation_status}", f"Editorial intervention:{item.intervention}", f"Text preservation: {item.preservation_percent:.1f}%", "", @@ -117,12 +118,18 @@ def main(argv: list[str] | None = None) -> int: return 0 if args.command == "place": - items = suggest_placements(args.source_url, args.target_url, args.anchor, top_n=args.top) + analysis = PlacementAnalyzer().analyze( + args.source_url, + args.target_url, + args.anchor, + max_opportunities=args.top, + ) + items = analysis.opportunities payload = [item.to_dict() for item in items] if args.output: write_json(payload, args.output) print(write_json(payload) if args.as_json else _placement_text(items)) - return 0 if items else 3 + return 0 if analysis.status in {"completed", "no_suitable_placement"} else 3 if args.command == "monitor": rows = monitor_csv(args.input_csv, args.state, args.output, delay_seconds=args.delay) diff --git a/backlink_intelligence/fetcher.py b/backlink_intelligence/fetcher.py index de338f9..a7c68c7 100644 --- a/backlink_intelligence/fetcher.py +++ b/backlink_intelligence/fetcher.py @@ -1,41 +1,238 @@ from __future__ import annotations import gzip +import http.client import io +import socket +import ssl +import zlib from dataclasses import dataclass -from urllib.error import HTTPError, URLError -from urllib.request import HTTPRedirectHandler, Request, build_opener +from urllib import robotparser +from urllib.parse import urljoin, urlsplit, urlunsplit from .html_utils import parse_page from .models import PageEvidence -from .safety import UnsafeURLError, validate_public_url +from .safety import ResolvedURL, UnsafeURLError, resolve_public_url + + +class FetchError(RuntimeError): + """A bounded public fetch failed.""" + + +class RobotsDeniedError(FetchError): + """The publisher's robots policy does not permit this fetch.""" + + +class UnsupportedContentError(FetchError): + """The response is not supported by the evidence parser.""" @dataclass(slots=True) class FetchConfig: - timeout: float = 12.0 - max_bytes: int = 2_000_000 - max_redirects: int = 5 - user_agent: str = "BacklinkIntelligence/1.0 (+https://github.com/alok-vibe-code/backlink-intelligence)" + connect_timeout: float = 5.0 + read_timeout: float = 12.0 + max_compressed_bytes: int = 1_000_000 + max_decompressed_bytes: int = 2_000_000 + max_redirects: int = 3 + max_paragraphs: int = 500 + max_headings: int = 200 + max_text_chars: int = 500_000 + respect_robots: bool = True + user_agent: str = ( + "BacklinkIntelligence/1.1 (+https://github.com/alok-vibe-code/backlink-intelligence)" + ) + + @property + def timeout(self) -> float: + """Compatibility alias for callers that previously used one timeout.""" + + return self.read_timeout + + @property + def max_bytes(self) -> int: + """Compatibility alias for the previous decompressed response limit.""" + + return self.max_decompressed_bytes + + +@dataclass(frozen=True, slots=True) +class RawResponse: + requested_url: str + final_url: str + status_code: int + headers: dict[str, str] + body: bytes + + +class _PinnedHTTPConnection(http.client.HTTPConnection): + def __init__(self, target: ResolvedURL, connect_ip: str, config: FetchConfig) -> None: + super().__init__(target.hostname, target.port, timeout=config.connect_timeout) + self._connect_ip = connect_ip + self._connect_timeout = config.connect_timeout + self._read_timeout = config.read_timeout + + def connect(self) -> None: + self.sock = socket.create_connection( + (self._connect_ip, self.port), timeout=self._connect_timeout + ) + self.sock.settimeout(self._read_timeout) + + +class _PinnedHTTPSConnection(http.client.HTTPSConnection): + def __init__(self, target: ResolvedURL, connect_ip: str, config: FetchConfig) -> None: + super().__init__( + target.hostname, + target.port, + timeout=config.connect_timeout, + context=ssl.create_default_context(), + ) + self._connect_ip = connect_ip + self._connect_timeout = config.connect_timeout + self._read_timeout = config.read_timeout + self._server_hostname = target.hostname + def connect(self) -> None: + raw_socket = socket.create_connection( + (self._connect_ip, self.port), timeout=self._connect_timeout + ) + try: + self.sock = self._context.wrap_socket( + raw_socket, + server_hostname=self._server_hostname, + ) + except Exception: + raw_socket.close() + raise + self.sock.settimeout(self._read_timeout) -class SafeRedirectHandler(HTTPRedirectHandler): - def __init__(self, max_redirects: int) -> None: - super().__init__() - self.max_redirects = max_redirects - self.count = 0 - def redirect_request(self, req, fp, code, msg, headers, newurl): - self.count += 1 - if self.count > self.max_redirects: +def _request_target(url: str) -> str: + parsed = urlsplit(url) + path = parsed.path or "/" + return urlunsplit(("", "", path, parsed.query, "")) + + +def _read_limited(response: http.client.HTTPResponse, limit: int) -> bytes: + chunks: list[bytes] = [] + total = 0 + while True: + chunk = response.read(min(65_536, limit + 1 - total)) + if not chunk: + break + chunks.append(chunk) + total += len(chunk) + if total > limit: + raise FetchError(f"Compressed response exceeded {limit} bytes.") + return b"".join(chunks) + + +def _decompress_limited(raw: bytes, encoding: str, limit: int) -> bytes: + encoding = encoding.lower().strip() + if not encoding or encoding == "identity": + if len(raw) > limit: + raise FetchError(f"Response exceeded {limit} decompressed bytes.") + return raw + + if encoding == "gzip": + stream = gzip.GzipFile(fileobj=io.BytesIO(raw)) + try: + output = stream.read(limit + 1) + except (EOFError, OSError) as exc: + raise FetchError("Invalid gzip response.") from exc + elif encoding == "deflate": + try: + inflater = zlib.decompressobj() + output = inflater.decompress(raw, limit + 1) + if inflater.unconsumed_tail: + raise FetchError(f"Response exceeded {limit} decompressed bytes.") + remaining = limit + 1 - len(output) + if remaining > 0: + output += inflater.flush(remaining) + except zlib.error as exc: + raise FetchError("Invalid deflate response.") from exc + else: + raise UnsupportedContentError(f"Unsupported content encoding: {encoding}") + + if len(output) > limit: + raise FetchError(f"Response exceeded {limit} decompressed bytes.") + return output + + +def _host_header(target: ResolvedURL) -> str: + default_port = 443 if target.scheme == "https" else 80 + host = f"[{target.hostname}]" if ":" in target.hostname else target.hostname + return host if target.port == default_port else f"{host}:{target.port}" + + +def _request_once(url: str, config: FetchConfig, *, accept: str) -> RawResponse: + target = resolve_public_url(url) + if not target.addresses: + raise UnsafeURLError("URL did not resolve to a public address.") + + last_error: Exception | None = None + for connect_ip in target.addresses: + connection: http.client.HTTPConnection + if target.scheme == "https": + connection = _PinnedHTTPSConnection(target, connect_ip, config) + else: + connection = _PinnedHTTPConnection(target, connect_ip, config) + try: + connection.request( + "GET", + _request_target(target.url), + headers={ + "Host": _host_header(target), + "User-Agent": config.user_agent, + "Accept": accept, + "Accept-Encoding": "gzip, deflate", + "Connection": "close", + }, + ) + response = connection.getresponse() + headers = {key.lower(): value for key, value in response.getheaders()} + raw = _read_limited(response, config.max_compressed_bytes) + body = _decompress_limited( + raw, + headers.get("content-encoding", ""), + config.max_decompressed_bytes, + ) + return RawResponse( + requested_url=url, + final_url=target.url, + status_code=response.status, + headers=headers, + body=body, + ) + except (OSError, TimeoutError, ssl.SSLError, http.client.HTTPException) as exc: + last_error = exc + finally: + connection.close() + raise FetchError("Remote host could not be reached securely.") from last_error + + +def _fetch_raw(url: str, config: FetchConfig, *, accept: str) -> RawResponse: + requested_url = resolve_public_url(url).url + current_url = requested_url + for redirect_count in range(config.max_redirects + 1): + response = _request_once(current_url, config, accept=accept) + if response.status_code not in {301, 302, 303, 307, 308}: + return RawResponse( + requested_url=requested_url, + final_url=response.final_url, + status_code=response.status_code, + headers=response.headers, + body=response.body, + ) + if redirect_count >= config.max_redirects: raise UnsafeURLError("Redirect limit exceeded.") - validate_public_url(newurl) - return super().redirect_request(req, fp, code, msg, headers, newurl) + location = response.headers.get("location", "").strip() + if not location: + raise FetchError("Redirect response did not include a location.") + current_url = resolve_public_url(urljoin(current_url, location)).url + raise UnsafeURLError("Redirect limit exceeded.") -def _decode_body(raw: bytes, encoding_header: str, content_type: str) -> str: - if "gzip" in encoding_header.lower(): - raw = gzip.GzipFile(fileobj=io.BytesIO(raw)).read() +def _decode_text(raw: bytes, content_type: str) -> str: charset = "utf-8" marker = "charset=" if marker in content_type.lower(): @@ -46,26 +243,65 @@ def _decode_body(raw: bytes, encoding_header: str, content_type: str) -> str: return raw.decode("utf-8", errors="replace") +def _robots_allowed(url: str, config: FetchConfig) -> bool: + parsed = urlsplit(url) + robots_url = urlunsplit((parsed.scheme, parsed.netloc, "/robots.txt", "", "")) + response = _fetch_raw(robots_url, config, accept="text/plain,*/*;q=0.1") + if response.status_code in {401, 403}: + raise RobotsDeniedError("Publisher robots policy denies automated access.") + if response.status_code == 404: + return True + if response.status_code >= 500: + raise RobotsDeniedError("Publisher robots policy is temporarily unavailable.") + if response.status_code != 200: + return True + + parser = robotparser.RobotFileParser() + parser.set_url(response.final_url) + parser.parse(_decode_text(response.body, response.headers.get("content-type", "")).splitlines()) + return parser.can_fetch(config.user_agent, url) + + def fetch_page(url: str, config: FetchConfig | None = None) -> PageEvidence: config = config or FetchConfig() - validate_public_url(url) - redirect_handler = SafeRedirectHandler(config.max_redirects) - opener = build_opener(redirect_handler) - request = Request(url, headers={"User-Agent": config.user_agent, "Accept": "text/html,application/xhtml+xml;q=0.9,*/*;q=0.1", "Accept-Encoding": "gzip"}) + normalized = url try: - with opener.open(request, timeout=config.timeout) as response: - final_url = response.geturl() - validate_public_url(final_url) - status = getattr(response, "status", 200) - content_type = response.headers.get("Content-Type", "") - if "text/html" not in content_type.lower() and "application/xhtml" not in content_type.lower(): - return PageEvidence(requested_url=url, final_url=final_url, status_code=status, error=f"Unsupported content type: {content_type or 'unknown'}") - raw = response.read(config.max_bytes + 1) - if len(raw) > config.max_bytes: - return PageEvidence(requested_url=url, final_url=final_url, status_code=status, error=f"Response exceeded max_bytes={config.max_bytes}") - html = _decode_body(raw, response.headers.get("Content-Encoding", ""), content_type) - return parse_page(html, requested_url=url, final_url=final_url, status_code=status) - except HTTPError as exc: - return PageEvidence(requested_url=url, final_url=exc.geturl(), status_code=exc.code, error=str(exc)) - except (URLError, TimeoutError, UnsafeURLError, OSError) as exc: - return PageEvidence(requested_url=url, final_url=url, status_code=0, error=str(exc)) + normalized = resolve_public_url(url).url + if config.respect_robots and not _robots_allowed(normalized, config): + raise RobotsDeniedError("Publisher robots policy denies automated access.") + response = _fetch_raw( + normalized, + config, + accept="text/html,application/xhtml+xml;q=0.9,*/*;q=0.1", + ) + content_type = response.headers.get("content-type", "") + if response.status_code != 200: + return PageEvidence( + requested_url=normalized, + final_url=response.final_url, + status_code=response.status_code, + error=f"Remote server returned HTTP {response.status_code}.", + ) + if "text/html" not in content_type.lower() and "application/xhtml" not in content_type.lower(): + raise UnsupportedContentError( + f"Unsupported content type: {content_type or 'unknown'}" + ) + html = _decode_text(response.body, content_type) + evidence = parse_page( + html, + requested_url=normalized, + final_url=response.final_url, + status_code=response.status_code, + ) + evidence.paragraphs = evidence.paragraphs[: config.max_paragraphs] + evidence.headings = evidence.headings[: config.max_headings] + evidence.text = evidence.text[: config.max_text_chars] + evidence.word_count = len(evidence.text.split()) + return evidence + except (FetchError, RobotsDeniedError, UnsupportedContentError, UnsafeURLError) as exc: + return PageEvidence( + requested_url=normalized, + final_url=normalized, + status_code=0, + error=f"{exc.__class__.__name__}: {exc}", + ) diff --git a/backlink_intelligence/models.py b/backlink_intelligence/models.py index 63e1c0c..e123550 100644 --- a/backlink_intelligence/models.py +++ b/backlink_intelligence/models.py @@ -99,6 +99,19 @@ def to_dict(self) -> dict[str, Any]: return asdict(self) +@dataclass(slots=True) +class TextSegment: + type: str + text: str + url: str = "" + + def to_dict(self) -> dict[str, Any]: + payload = {"type": self.type, "text": self.text} + if self.type == "link": + payload["url"] = self.url + return payload + + @dataclass(slots=True) class PlacementSuggestion: rank: int @@ -112,9 +125,13 @@ class PlacementSuggestion: strategy: str before: str after: str + after_text: str + after_segments: list[TextSegment] added_words: int preservation_percent: float intervention: str + recommendation_status: str + review_required: bool reasons: list[str] = field(default_factory=list) warnings: list[str] = field(default_factory=list) diff --git a/backlink_intelligence/placement.py b/backlink_intelligence/placement.py index 27533e1..52161e2 100644 --- a/backlink_intelligence/placement.py +++ b/backlink_intelligence/placement.py @@ -3,7 +3,7 @@ import re from .fetcher import FetchConfig, fetch_page -from .models import PageEvidence, PlacementSuggestion +from .models import PageEvidence, PlacementSuggestion, TextSegment from .relevance import similarity, tokens @@ -97,12 +97,10 @@ def _fallback_anchor_case(anchor: str) -> str: def _contextual_fallback_sentence( paragraph: str, anchor: str, - target_url: str, target_title: str, -) -> tuple[str, str, list[str]]: +) -> tuple[str, str, str, list[str]]: """Create concise deterministic fallback copy without dumping the target title.""" placed_anchor = _fallback_anchor_case(anchor) - linked = f"[{placed_anchor}]({target_url})" target_lower = target_title.lower() paragraph_lower = paragraph.lower() @@ -114,27 +112,47 @@ def _contextual_fallback_sentence( notes.append("anchor_casing_adapted_for_generated_sentence") if cost_intent and cost_context: - sentence = ( - f"These factors are useful when estimating {linked} implementation costs, " - "ongoing operating expenses, and expected ROI." - ) notes.append("destination_intent_used_for_contextual_sentence") - return sentence, placed_anchor, notes + return ( + "These factors are useful when estimating ", + " implementation costs, ongoing operating expenses, and expected ROI.", + placed_anchor, + notes, + ) if cost_intent: - sentence = ( - f"Businesses evaluating this type of automation should also account for {linked} costs, " - "including implementation, integrations, ongoing operation, and expected ROI." - ) notes.append("destination_intent_used_for_contextual_sentence") - return sentence, placed_anchor, notes + return ( + "Businesses evaluating this type of automation should also account for ", + " costs, including implementation, integrations, ongoing operation, and expected ROI.", + placed_anchor, + notes, + ) if any(term in target_lower for term in ("roadmap", "learning", "course", "guide")): - sentence = f"Readers who want a structured next step can explore this {linked} for more detail." - return sentence, placed_anchor, notes + return ( + "Readers who want a structured next step can explore this ", + " for more detail.", + placed_anchor, + notes, + ) + + return ( + "Readers who want additional context can review this ", + " resource.", + placed_anchor, + notes, + ) - sentence = f"Readers who want additional context can review this {linked} resource." - return sentence, placed_anchor, notes + +def _segments(prefix: str, anchor: str, target_url: str, suffix: str) -> list[TextSegment]: + segments: list[TextSegment] = [] + if prefix: + segments.append(TextSegment(type="text", text=prefix)) + segments.append(TextSegment(type="link", text=anchor, url=target_url)) + if suffix: + segments.append(TextSegment(type="text", text=suffix)) + return segments def _compose_after( @@ -142,17 +160,21 @@ def _compose_after( anchor: str, target_url: str, target_title: str, -) -> tuple[str, str, str, list[str]]: +) -> tuple[str, str, str, list[TextSegment], str, list[str]]: """Compose the draft while preserving source grammar/capitalization when possible.""" exact = _find_complete_phrase(paragraph, anchor) if exact is not None: placed_anchor = exact.group(0) linked = f"[{placed_anchor}]({target_url})" after = paragraph[: exact.start()] + linked + paragraph[exact.end() :] + after_text = paragraph + segments = _segments( + paragraph[: exact.start()], placed_anchor, target_url, paragraph[exact.end() :] + ) notes: list[str] = [] if placed_anchor != anchor: notes.append("source_anchor_capitalization_preserved") - return "minimal_insertion", after, placed_anchor, notes + return "minimal_insertion", after, after_text, segments, placed_anchor, notes # If the exact requested form is not present, prefer a complete natural word-form # already in the publisher copy instead of creating artifacts such as [AI Agent]s. @@ -162,17 +184,33 @@ def _compose_after( placed_anchor = match.group(0) linked = f"[{placed_anchor}]({target_url})" after = paragraph[: match.start()] + linked + paragraph[match.end() :] + after_text = paragraph + segments = _segments( + paragraph[: match.start()], placed_anchor, target_url, paragraph[match.end() :] + ) return ( "minimal_insertion", after, + after_text, + segments, placed_anchor, ["anchor_adapted_to_source_grammar", "requested_anchor_not_used_verbatim"], ) - sentence, placed_anchor, notes = _contextual_fallback_sentence( - paragraph, anchor, target_url, target_title + sentence_prefix, sentence_suffix, placed_anchor, notes = _contextual_fallback_sentence( + paragraph, anchor, target_title + ) + prefix = paragraph.rstrip() + " " + sentence_prefix + after_text = prefix + placed_anchor + sentence_suffix + after = prefix + f"[{placed_anchor}]({target_url})" + sentence_suffix + return ( + "contextual_sentence", + after, + after_text, + _segments(prefix, placed_anchor, target_url, sentence_suffix), + placed_anchor, + notes, ) - return "contextual_sentence", paragraph.rstrip() + " " + sentence, placed_anchor, notes def _stem(term: str) -> str: @@ -231,6 +269,8 @@ def rank_placements( target_url: str, *, top_n: int = 3, + min_context_score: float = 0.0, + min_destination_score: float = 0.0, ) -> list[PlacementSuggestion]: if source.status_code != 200 or target.status_code != 200: return [] @@ -250,14 +290,17 @@ def rank_placements( # Destination intent gets meaningful weight so a pricing/cost paragraph beats a # generic paragraph that merely repeats the requested anchor. score = round((0.62 * semantic_score) + (0.30 * destination_score) + (0.08 * anchor_overlap), 4) - candidates.append((score, destination_score, i, paragraph)) + if score >= min_context_score and destination_score >= min_destination_score: + candidates.append((score, destination_score, i, paragraph)) candidates.sort(key=lambda item: (-item[0], -item[1], item[2])) suggestions: list[PlacementSuggestion] = [] for rank, (score, destination_score, index, paragraph) in enumerate(candidates[: max(top_n, 1)], start=1): - strategy, after, placed_anchor, compose_notes = _compose_after(paragraph, anchor, target_url, target.title) + strategy, after, after_text, after_segments, placed_anchor, compose_notes = _compose_after( + paragraph, anchor, target_url, target.title + ) original_words = max(len(paragraph.split()), 1) - after_words = len(after.split()) + after_words = len(after_text.split()) added = max(after_words - original_words, 0) preservation = 100.0 if strategy in {"minimal_insertion", "contextual_sentence"} else 90.0 warnings = list(anchor_warnings) @@ -273,11 +316,18 @@ def rank_placements( reasons.append(note) if destination_score >= 0.14: reasons.append("strong_destination_intent_alignment") - elif destination_score < 0.08: - warnings.append("weak_destination_intent_alignment") - if score < 0.12: - warnings.append("weak_context_match_manual_review_required") context_level = "very_high" if score >= 0.48 else "high" if score >= 0.30 else "medium" if score >= 0.15 else "low" + intervention = _intervention(preservation, added) + near_threshold = ( + score < min_context_score + 0.05 + or destination_score < min_destination_score + 0.03 + ) + review_required = bool( + strategy != "minimal_insertion" + or warnings + or intervention != "low" + or near_threshold + ) suggestions.append( PlacementSuggestion( rank=rank, @@ -291,9 +341,13 @@ def rank_placements( strategy=strategy, before=paragraph, after=after, + after_text=after_text, + after_segments=after_segments, added_words=added, preservation_percent=preservation, - intervention=_intervention(preservation, added), + intervention=intervention, + recommendation_status="manual_review" if review_required else "recommended", + review_required=review_required, reasons=reasons, warnings=warnings, ) diff --git a/backlink_intelligence/safety.py b/backlink_intelligence/safety.py index 1cbe388..32497b6 100644 --- a/backlink_intelligence/safety.py +++ b/backlink_intelligence/safety.py @@ -2,11 +2,23 @@ import ipaddress import socket -from urllib.parse import urlparse +from dataclasses import dataclass +from urllib.parse import SplitResult, urlsplit, urlunsplit class UnsafeURLError(RuntimeError): - pass + """Raised when a URL cannot be fetched without crossing a trust boundary.""" + + +@dataclass(frozen=True, slots=True) +class ResolvedURL: + """A normalized public URL and the public IPs approved for one connection.""" + + url: str + scheme: str + hostname: str + port: int + addresses: tuple[str, ...] def _is_public_ip(value: str) -> bool: @@ -21,37 +33,91 @@ def _is_public_ip(value: str) -> bool: ) -def validate_public_url(url: str, *, resolve_dns: bool = True) -> str: - parsed = urlparse(url) - if parsed.scheme not in {"http", "https"}: +def _normalized_parts(url: str) -> tuple[SplitResult, str, int, str]: + if not isinstance(url, str) or not url.strip(): + raise UnsafeURLError("URL is required.") + if len(url) > 2048: + raise UnsafeURLError("URL exceeds the 2048 character limit.") + + try: + parsed = urlsplit(url.strip()) + port = parsed.port + except ValueError as exc: + raise UnsafeURLError("URL contains an invalid port or hostname.") from exc + + scheme = parsed.scheme.lower() + if scheme not in {"http", "https"}: raise UnsafeURLError("Only http:// and https:// URLs are supported.") if not parsed.hostname: raise UnsafeURLError("URL must include a hostname.") if parsed.username or parsed.password: raise UnsafeURLError("Credentials embedded in URLs are not allowed.") - host = parsed.hostname.strip("[]").lower() - if host in {"localhost", "localhost.localdomain"} or host.endswith(".local"): + try: + hostname = parsed.hostname.rstrip(".").encode("idna").decode("ascii").lower() + except UnicodeError as exc: + raise UnsafeURLError("URL hostname is not valid IDNA.") from exc + if hostname in {"localhost", "localhost.localdomain"} or hostname.endswith(".local"): raise UnsafeURLError("Local/private hostnames are not allowed.") + default_port = 443 if scheme == "https" else 80 + port = port or default_port + if port not in {80, 443}: + raise UnsafeURLError("Only ports 80 and 443 are supported.") + + host_for_url = f"[{hostname}]" if ":" in hostname else hostname + netloc = host_for_url + if port != default_port: + netloc = f"{host_for_url}:{port}" + path = parsed.path or "/" + normalized = urlunsplit((scheme, netloc, path, parsed.query, "")) + return parsed, hostname, port, normalized + + +def resolve_public_url(url: str, *, resolve_dns: bool = True) -> ResolvedURL: + """Normalize a URL and resolve every address before a connection is attempted.""" + + _, hostname, port, normalized = _normalized_parts(url) + addresses: set[str] = set() + try: - if not _is_public_ip(host): - raise UnsafeURLError("Private, loopback, reserved, or link-local IPs are not allowed.") - return url + literal = ipaddress.ip_address(hostname) except ValueError: - pass + literal = None - if resolve_dns: + if literal is not None: + if not _is_public_ip(hostname): + raise UnsafeURLError("Private, loopback, reserved, or link-local IPs are not allowed.") + addresses.add(str(literal)) + elif resolve_dns: try: - infos = socket.getaddrinfo(host, parsed.port or (443 if parsed.scheme == "https" else 80)) + infos = socket.getaddrinfo(hostname, port, type=socket.SOCK_STREAM) except socket.gaierror as exc: - raise UnsafeURLError(f"Hostname could not be resolved: {host}") from exc + raise UnsafeURLError(f"Hostname could not be resolved: {hostname}") from exc addresses = {info[4][0] for info in infos} if not addresses: - raise UnsafeURLError(f"Hostname could not be resolved: {host}") + raise UnsafeURLError(f"Hostname could not be resolved: {hostname}") for address in addresses: - if not _is_public_ip(address): + try: + public = _is_public_ip(address) + except ValueError as exc: + raise UnsafeURLError("Hostname resolved to an invalid address.") from exc + if not public: raise UnsafeURLError( f"Hostname resolves to a non-public address ({address}); request blocked." ) - return url + + scheme = urlsplit(normalized).scheme + return ResolvedURL( + url=normalized, + scheme=scheme, + hostname=hostname, + port=port, + addresses=tuple(sorted(addresses)), + ) + + +def validate_public_url(url: str, *, resolve_dns: bool = True) -> str: + """Compatibility wrapper returning the normalized, validated URL.""" + + return resolve_public_url(url, resolve_dns=resolve_dns).url diff --git a/pyproject.toml b/pyproject.toml index b338136..efaadc6 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "backlink-intelligence" -version = "1.0.1" +version = "1.1.0" description = "Open-source backlink intelligence based on evidence, context, and editorial fit." readme = "README.md" requires-python = ">=3.11" @@ -35,5 +35,15 @@ Issues = "https://github.com/alok-vibe-code/backlink-intelligence/issues" [project.scripts] backlink-intelligence = "backlink_intelligence.cli:main" +[project.optional-dependencies] +api = [ + "fastapi>=0.116,<1", + "pydantic>=2.11,<3", + "uvicorn>=0.35,<1", +] +test = [ + "httpx>=0.28,<1", +] + [tool.setuptools.packages.find] include = ["backlink_intelligence*"] diff --git a/render.yaml b/render.yaml new file mode 100644 index 0000000..8431c49 --- /dev/null +++ b/render.yaml @@ -0,0 +1,40 @@ +services: + - type: web + name: backlink-intelligence-api + runtime: python + plan: free + buildCommand: pip install ".[api]" + startCommand: uvicorn backlink_intelligence.api:app --host 0.0.0.0 --port $PORT --workers 1 --no-access-log + healthCheckPath: /health + autoDeploy: true + envVars: + - key: BI_ENVIRONMENT + value: production + - key: BI_ALLOWED_ORIGINS + value: https://alokblog.com + - key: BI_ALLOWED_HOSTS + value: api.alokblog.com + - key: BI_TURNSTILE_HOSTNAME + value: alokblog.com + - key: BI_TURNSTILE_SECRET + sync: false + - key: BI_MIN_CONTEXT_SCORE + value: "0.15" + - key: BI_MIN_DESTINATION_SCORE + value: "0.08" + - key: BI_MAX_OPPORTUNITIES + value: "3" + - key: BI_RATE_LIMIT_PER_HOUR + value: "5" + - key: BI_MAX_CONCURRENCY + value: "2" + - key: BI_ANALYSIS_TIMEOUT + value: "35" + - key: BI_CONNECT_TIMEOUT + value: "5" + - key: BI_READ_TIMEOUT + value: "12" + - key: BI_MAX_COMPRESSED_BYTES + value: "1000000" + - key: BI_MAX_DECOMPRESSED_BYTES + value: "2000000" diff --git a/tests/test_analysis.py b/tests/test_analysis.py new file mode 100644 index 0000000..2a0b403 --- /dev/null +++ b/tests/test_analysis.py @@ -0,0 +1,58 @@ +import unittest +from unittest.mock import patch + +from backlink_intelligence.analysis import AnalysisConfig, PlacementAnalyzer +from backlink_intelligence.fetcher import FetchConfig +from backlink_intelligence.html_utils import parse_page + + +class AnalysisTests(unittest.TestCase): + def setUp(self): + self.source = parse_page( + "

A custom AI agent connects a website, CRM, email, database, and internal dashboard while supporting reliable business automation workflows.

", + requested_url="https://source.example/article/", + final_url="https://source.example/article/", + status_code=200, + ) + self.target = parse_page( + "AI Agent Cost and ROI

AI Agent Cost

Implementation pricing, integrations, operating costs, and ROI vary by project.

", + requested_url="https://target.example/guide/", + final_url="https://target.example/guide/", + status_code=200, + ) + + @patch("backlink_intelligence.analysis.fetch_page") + def test_completed_analysis_uses_structured_segments(self, fetch_page): + fetch_page.side_effect = [self.source, self.target] + analyzer = PlacementAnalyzer( + AnalysisConfig(0.0, 0.0, 3, FetchConfig(respect_robots=False)) + ) + result = analyzer.analyze( + "https://source.example/article/", + "https://target.example/guide/", + "AI Agent", + ) + self.assertEqual(result.status, "completed") + item = result.opportunities[0] + self.assertEqual(sum(segment.type == "link" for segment in item.after_segments), 1) + self.assertEqual("".join(segment.text for segment in item.after_segments), item.after_text) + self.assertEqual( + next(segment.url for segment in item.after_segments if segment.type == "link"), + "https://target.example/guide/", + ) + + @patch("backlink_intelligence.analysis.fetch_page") + def test_no_suitable_placement_is_successful_outcome(self, fetch_page): + fetch_page.side_effect = [self.source, self.target] + analyzer = PlacementAnalyzer(AnalysisConfig(1.0, 1.0, 3)) + result = analyzer.analyze( + "https://source.example/article/", + "https://target.example/guide/", + "AI Agent", + ) + self.assertEqual(result.status, "no_suitable_placement") + self.assertEqual(result.opportunities, []) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_api.py b/tests/test_api.py new file mode 100644 index 0000000..5607e65 --- /dev/null +++ b/tests/test_api.py @@ -0,0 +1,204 @@ +import unittest +from unittest.mock import patch + +from fastapi.testclient import TestClient + +from backlink_intelligence.analysis import PlacementAnalysis +from backlink_intelligence.api import ( + OpportunityResponse, + SETTINGS, + SegmentResponse, + app, + challenge_verifier, +) +from backlink_intelligence.html_utils import parse_page +from backlink_intelligence.models import PlacementSuggestion, TextSegment + + +class APITests(unittest.TestCase): + @classmethod + def setUpClass(cls): + app.dependency_overrides[challenge_verifier] = lambda: (lambda token: True) + cls.client = TestClient(app) + + @classmethod + def tearDownClass(cls): + app.dependency_overrides.clear() + + def pages(self): + source = parse_page( + "

Unicode testing includes an emoji 😀 and an AI agent in a complete editorial sentence for review.

", + requested_url="https://source.example/", + final_url="https://source.example/", + status_code=200, + ) + target = parse_page( + "AI Agent Guide

AI agent implementation guide.

", + requested_url="https://target.example/", + final_url="https://target.example/", + status_code=200, + ) + return source, target + + def payload(self): + return { + "source_url": "https://source.example/", + "target_url": "https://target.example/", + "anchor": "AI agent", + "challenge_token": "test-token", + } + + @patch("backlink_intelligence.api.PlacementAnalyzer.analyze") + def test_v1_completed_contract_uses_segments_not_offsets(self, analyze): + source, target = self.pages() + after_text = "Unicode testing includes an emoji 😀 and an AI agent in a complete sentence." + analyze.return_value = PlacementAnalysis( + status="completed", + source=source, + target=target, + opportunities=[ + PlacementSuggestion( + rank=1, + paragraph_index=1, + score=0.4, + context_level="high", + destination_score=0.2, + destination_fit="high", + requested_anchor="AI agent", + suggested_anchor="AI agent", + strategy="minimal_insertion", + before=after_text, + after="Unicode testing includes an emoji 😀 and an [AI agent](https://target.example/) in a complete sentence.", + after_text=after_text, + after_segments=[ + TextSegment("text", "Unicode testing includes an emoji 😀 and an "), + TextSegment("link", "AI agent", "https://target.example/"), + TextSegment("text", " in a complete sentence."), + ], + added_words=0, + preservation_percent=100.0, + intervention="low", + recommendation_status="recommended", + review_required=False, + reasons=["anchor_already_present_in_original_copy"], + ) + ], + ) + response = self.client.post("/v1/place", json=self.payload()) + self.assertEqual(response.status_code, 200) + body = response.json() + self.assertTrue(body["success"]) + self.assertNotIn("link_start", body["opportunities"][0]) + segments = body["opportunities"][0]["after_segments"] + self.assertEqual("".join(segment["text"] for segment in segments), after_text) + self.assertEqual(sum(segment["type"] == "link" for segment in segments), 1) + + @patch("backlink_intelligence.api.PlacementAnalyzer.analyze") + def test_no_suitable_placement_returns_http_200(self, analyze): + source, target = self.pages() + analyze.return_value = PlacementAnalysis( + status="no_suitable_placement", + source=source, + target=target, + opportunities=[], + ) + response = self.client.post("/v1/place", json=self.payload()) + self.assertEqual(response.status_code, 200) + self.assertEqual(response.json()["status"], "no_suitable_placement") + self.assertEqual(response.json()["opportunities"], []) + + def test_unknown_request_field_is_rejected(self): + payload = self.payload() + payload["unexpected"] = True + response = self.client.post("/v1/place", json=payload) + self.assertEqual(response.status_code, 422) + self.assertEqual(response.json()["error"]["code"], "invalid_request") + + def test_health_contract(self): + response = self.client.get("/health") + self.assertEqual(response.status_code, 200) + self.assertEqual(response.json()["api_version"], "1") + self.assertEqual(response.headers["cache-control"], "no-store") + + def test_unapproved_origin_is_rejected(self): + response = self.client.post( + "/v1/place", + json=self.payload(), + headers={"Origin": "https://attacker.example"}, + ) + self.assertEqual(response.status_code, 403) + + def test_unapproved_host_is_rejected_in_production(self): + with patch.object(SETTINGS, "environment", "production"): + response = self.client.post( + "/v1/place", + json=self.payload(), + headers={ + "Host": "attacker.example", + "Origin": "https://alokblog.com", + }, + ) + self.assertEqual(response.status_code, 400) + + def test_unicode_segments_reconstruct_combining_and_non_bmp_text(self): + after_text = "Cafe\u0301 😀 links to an AI agent guide." + opportunity = OpportunityResponse( + rank=1, + paragraph_index=1, + score=0.2, + context_level="medium", + destination_score=0.1, + destination_fit="medium", + requested_anchor="AI agent", + suggested_anchor="AI agent", + strategy="minimal_insertion", + recommendation_status="recommended", + review_required=False, + intervention="low", + preservation_percent=100.0, + before_text=after_text, + after_text=after_text, + after_segments=[ + SegmentResponse(type="text", text="Cafe\u0301 😀 links to an "), + SegmentResponse( + type="link", text="AI agent", url="https://target.example/" + ), + SegmentResponse(type="text", text=" guide."), + ], + reasons=[], + warnings=[], + ) + self.assertEqual( + "".join(segment.text for segment in opportunity.after_segments), after_text + ) + + def test_structured_segments_reject_mismatched_text(self): + with self.assertRaises(ValueError): + OpportunityResponse( + rank=1, + paragraph_index=1, + score=0.2, + context_level="medium", + destination_score=0.1, + destination_fit="medium", + requested_anchor="AI agent", + suggested_anchor="AI agent", + strategy="minimal_insertion", + recommendation_status="recommended", + review_required=False, + intervention="low", + preservation_percent=100.0, + before_text="Before", + after_text="Expected", + after_segments=[ + SegmentResponse( + type="link", text="Different", url="https://target.example/" + ) + ], + reasons=[], + warnings=[], + ) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_cli.py b/tests/test_cli.py index 9ae22e2..4594bf3 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -8,7 +8,7 @@ class CLITests(unittest.TestCase): def test_version_is_stable(self): - self.assertEqual(__version__, "1.0.1") + self.assertEqual(__version__, "1.1.0") def test_status_command(self): output = io.StringIO() diff --git a/tests/test_fetcher_security.py b/tests/test_fetcher_security.py new file mode 100644 index 0000000..be9e7ed --- /dev/null +++ b/tests/test_fetcher_security.py @@ -0,0 +1,26 @@ +import gzip +import unittest + +from backlink_intelligence.fetcher import FetchError, _decompress_limited +from backlink_intelligence.safety import UnsafeURLError, validate_public_url + + +class FetcherSecurityTests(unittest.TestCase): + def test_rejects_nonstandard_public_port(self): + with self.assertRaises(UnsafeURLError): + validate_public_url("https://example.com:8443/", resolve_dns=False) + + def test_gzip_expansion_is_bounded(self): + compressed = gzip.compress(b"A" * 20_000) + with self.assertRaises(FetchError): + _decompress_limited(compressed, "gzip", 1_000) + + def test_url_fragment_is_removed_before_fetch(self): + self.assertEqual( + validate_public_url("https://example.com/path#private", resolve_dns=False), + "https://example.com/path", + ) + + +if __name__ == "__main__": + unittest.main()