diff --git a/CHANGELOG.md b/CHANGELOG.md
index f90d37c..5b9fc6a 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -1,5 +1,14 @@
# Changelog
+## 1.1.3 - 2026-09-26
+
+- Removed navigation, sidebar, footer, and repeated promotional copy from page evidence.
+- Added anchor-aware target profiles built from topical headings and substantive body copy.
+- Separated broad topic alignment from specialized destination-purpose evidence.
+- Allowed strong course-topic matches to return editorial-review opportunities without lowering beta thresholds.
+- Prevented table and list introductions from becoming malformed contextual rewrites.
+- Added regression coverage for the DelightChat and Great Learning AI-agent course case.
+
## 1.1.2 - 2026-09-26
- Classify destination pages using on-page evidence with the final URL as a supporting signal.
diff --git a/backlink_intelligence/__init__.py b/backlink_intelligence/__init__.py
index e31a0f9..bc5bef7 100644
--- a/backlink_intelligence/__init__.py
+++ b/backlink_intelligence/__init__.py
@@ -1,3 +1,3 @@
"""Backlink Intelligence package."""
-__version__ = "1.1.2"
+__version__ = "1.1.3"
diff --git a/backlink_intelligence/html_utils.py b/backlink_intelligence/html_utils.py
index 07dc09f..4925243 100644
--- a/backlink_intelligence/html_utils.py
+++ b/backlink_intelligence/html_utils.py
@@ -34,6 +34,8 @@ def __init__(self, base_url: str) -> None:
self._current_link: dict | None = None
self._skip_depth = 0
self._landmark_stack: list[str] = []
+ self._seen_headings: set[str] = set()
+ self._seen_paragraphs: set[str] = set()
def _placement(self) -> str:
landmarks = set(self._landmark_stack)
@@ -90,16 +92,28 @@ def handle_endtag(self, tag: str) -> None:
self._h1_depth -= 1
if self._heading_tag == tag:
text = clean_text(" ".join(self._heading_parts))
- if text:
+ key = text.casefold()
+ if (
+ text
+ and self._placement() not in {"navigation", "sidebar", "footer"}
+ and key not in self._seen_headings
+ ):
self.headings.append(text)
+ self._seen_headings.add(key)
self._heading_tag = None
self._heading_parts = []
if tag == "p" and self._paragraph_depth:
self._paragraph_depth -= 1
if self._paragraph_depth == 0:
text = clean_text(" ".join(self._paragraph_parts))
- if text:
+ key = text.casefold()
+ if (
+ text
+ and self._placement() not in {"navigation", "sidebar", "footer"}
+ and key not in self._seen_paragraphs
+ ):
self.paragraphs.append(text)
+ self._seen_paragraphs.add(key)
self._paragraph_parts = []
if tag == "a" and self._current_link is not None:
text = clean_text(" ".join(self._current_link["text"]))
@@ -126,7 +140,7 @@ def handle_data(self, data: str) -> None:
return
if self._title_depth:
self.title_parts.append(text)
- if self._h1_depth:
+ if self._h1_depth and self._placement() not in {"navigation", "sidebar", "footer"}:
self.h1_parts.append(text)
if self._heading_tag:
self._heading_parts.append(text)
diff --git a/backlink_intelligence/placement.py b/backlink_intelligence/placement.py
index 310e121..c2bfa0c 100644
--- a/backlink_intelligence/placement.py
+++ b/backlink_intelligence/placement.py
@@ -113,8 +113,95 @@ def _anchor_with_article(anchor: str) -> str:
return f"{article} {anchor}"
+_PROFILE_BOILERPLATE_TERMS = {
+ "academy",
+ "certificate",
+ "certification",
+ "class",
+ "course",
+ "department",
+ "degree",
+ "education",
+ "institute",
+ "learn",
+ "learning",
+ "online",
+ "offered",
+ "program",
+ "programme",
+ "professional",
+ "professionals",
+ "school",
+ "student",
+ "students",
+ "training",
+ "university",
+ "working",
+}
+
+
+def _unique_text(items: list[str]) -> list[str]:
+ output: list[str] = []
+ seen: set[str] = set()
+ for item in items:
+ value = " ".join(item.split()).strip()
+ key = value.casefold()
+ if value and key not in seen:
+ seen.add(key)
+ output.append(value)
+ return output
+
+
def _target_profile(target: PageEvidence) -> str:
- return " ".join([target.title, target.h1, *target.headings, target.text[:12000]])
+ """Return the full cleaned page profile used by composition classifiers."""
+ return " ".join(
+ _unique_text([target.title, target.h1, *target.headings, *target.paragraphs])
+ )[:12000]
+
+
+def _target_topic_evidence(target: PageEvidence, anchor: str) -> tuple[str, list[str]]:
+ """Build a compact topical profile without navigation or destination wrappers."""
+ url_text = _target_url_text(target)
+ boilerplate_terms = {_stem(term) for term in _PROFILE_BOILERPLATE_TERMS}
+ anchor_subject_terms = _stems(anchor) - boilerplate_terms
+ fallback_seed = " ".join([target.title, target.h1, url_text]).strip()
+ subject_terms = anchor_subject_terms or (_stems(fallback_seed) - boilerplate_terms)
+ if not subject_terms:
+ subject_terms = _stems(anchor)
+ seed = " ".join(sorted(subject_terms))
+
+ headings = _unique_text(target.headings)
+ paragraphs = _unique_text(target.paragraphs)
+
+ def is_topical(value: str) -> bool:
+ return bool(_stems(value) & subject_terms)
+
+ ranked_headings = sorted(
+ (
+ (similarity(value, seed), index, value)
+ for index, value in enumerate(headings)
+ if len(value.split()) >= 3 and is_topical(value)
+ ),
+ key=lambda item: (-item[0], item[1]),
+ )
+ ranked_paragraphs = sorted(
+ (
+ (similarity(value, seed), index, value)
+ for index, value in enumerate(paragraphs)
+ if len(value.split()) >= 8 and is_topical(value)
+ ),
+ key=lambda item: (-item[0], item[1]),
+ )
+
+ selected_headings = [value for _, _, value in ranked_headings[:8]]
+ selected_paragraphs = [value for _, _, value in ranked_paragraphs[:12]]
+ topic_blocks = selected_paragraphs or selected_headings
+
+ primary = [target.title]
+ if is_topical(target.h1):
+ primary.append(target.h1)
+ profile_parts = _unique_text([*primary, url_text, *selected_headings, *selected_paragraphs])
+ return " ".join(profile_parts), topic_blocks
_TARGET_TYPE_SIGNALS: tuple[tuple[str, tuple[str, ...]], ...] = (
@@ -563,25 +650,38 @@ def _stems(text: str) -> set[str]:
return {_stem(term) for term in tokens(text)}
-def _destination_intent_score(paragraph: str, target: PageEvidence, anchor: str) -> float:
- """Measure fit to destination-specific intent, not just the requested anchor."""
- core_profile = " ".join([target.title, target.h1]).strip()
- if not core_profile:
- core_profile = " ".join(target.headings[:8]).strip()
- if not core_profile:
- return similarity(paragraph, target.text[:4000])
+def _destination_intent_score(
+ paragraph: str,
+ target: PageEvidence,
+ anchor: str,
+ topic_profile: str,
+ topic_blocks: list[str],
+) -> float:
+ """Measure topical fit separately from the destination page's purpose."""
+ profile_score = similarity(paragraph, topic_profile) if topic_profile else 0.0
+ block_score = max((similarity(paragraph, block) for block in topic_blocks), default=0.0)
+ topic_score = (0.70 * block_score) + (0.30 * profile_score)
+ intent = _target_intent(target)
+ if intent not in {"cost", "implementation", "governance"}:
+ # Learning, guide, service, and general resources can support a paragraph that
+ # shares their subject even when the source does not already mention a course,
+ # guide, or service. Generated copy remains subject to editorial review.
+ return round(topic_score, 4)
+
+ signals = next(
+ (values for kind, values in _TARGET_TYPE_SIGNALS if kind == intent),
+ (),
+ )
paragraph_terms = _stems(paragraph)
- core_terms = _stems(core_profile)
- anchor_terms = _stems(anchor)
-
- # Prefer terms that describe what makes the destination distinct from the anchor.
- intent_terms = core_terms - anchor_terms
- if len(intent_terms) < 2:
- intent_terms = core_terms
- intent_overlap = len(paragraph_terms & intent_terms) / max(len(intent_terms), 1)
- semantic = similarity(paragraph, core_profile)
- return round((0.35 * semantic) + (0.65 * intent_overlap), 4)
+ purpose_hits = sum(bool(paragraph_terms & _stems(value)) for value in signals)
+ if purpose_hits == 0:
+ # Specialized destinations need evidence of their specific purpose. This keeps
+ # generic topical paragraphs from outranking pricing, implementation, or
+ # governance context merely because they repeat the anchor.
+ return round(topic_score * 0.35, 4)
+ purpose_score = min(1.0, purpose_hits / 2.0)
+ return round((0.70 * topic_score) + (0.30 * purpose_score), 4)
def _destination_level(score: float) -> str:
@@ -608,15 +708,26 @@ def rank_placements(
return []
anchor, anchor_warnings = _select_anchor(preferred_anchor, target.title)
- target_profile = _target_profile(target)
+ target_profile, target_topic_blocks = _target_topic_evidence(target, anchor)
candidates: list[tuple[float, float, int, str]] = []
for i, paragraph in enumerate(source.paragraphs, start=1):
wc = len(paragraph.split())
if wc < 18 or wc > 260:
continue
+ # A standalone paragraph ending in a colon or semicolon normally introduces
+ # a table or list. Treating it as complete prose produces malformed rewrites
+ # and weak placements immediately before structured content.
+ if paragraph.rstrip().endswith((":", ";")):
+ continue
semantic_score = similarity(paragraph, target_profile)
- destination_score = _destination_intent_score(paragraph, target, anchor)
+ destination_score = _destination_intent_score(
+ paragraph,
+ target,
+ anchor,
+ target_profile,
+ target_topic_blocks,
+ )
anchor_terms = set(tokens(anchor))
anchor_overlap = len(anchor_terms & set(tokens(paragraph))) / max(len(anchor_terms), 1)
# Destination intent gets meaningful weight so a pricing/cost paragraph beats a
diff --git a/pyproject.toml b/pyproject.toml
index 65a6cc0..8dc47d8 100644
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
[project]
name = "backlink-intelligence"
-version = "1.1.2"
+version = "1.1.3"
description = "Open-source backlink intelligence based on evidence, context, and editorial fit."
readme = "README.md"
requires-python = ">=3.11"
diff --git a/tests/test_cli.py b/tests/test_cli.py
index 0ad19d7..158f8f3 100644
--- a/tests/test_cli.py
+++ b/tests/test_cli.py
@@ -8,7 +8,7 @@
class CLITests(unittest.TestCase):
def test_version_is_stable(self):
- self.assertEqual(__version__, "1.1.2")
+ self.assertEqual(__version__, "1.1.3")
def test_status_command(self):
output = io.StringIO()
diff --git a/tests/test_html_utils.py b/tests/test_html_utils.py
index 4d37ae0..d8f66a5 100644
--- a/tests/test_html_utils.py
+++ b/tests/test_html_utils.py
@@ -22,5 +22,27 @@ def test_link_context_and_placement(self):
target = next(l for l in self.page.links if "target.com" in l.href); self.assertEqual(target.text, "agentic AI course"); self.assertEqual(target.placement, "editorial_context"); self.assertIn("nofollow", target.rel)
footer = next(l for l in self.page.links if "social.example" in l.href); self.assertEqual(footer.placement, "footer")
+ def test_navigation_sidebar_footer_and_duplicates_do_not_pollute_page_copy(self):
+ page = parse_page(
+ """
+
+
+
Agentic AI Curriculum
+
Build autonomous AI agents with tools, retrieval, memory, and orchestration.
+
Agentic AI Curriculum
+
Build autonomous AI agents with tools, retrieval, memory, and orchestration.
+
+ """,
+ requested_url="https://example.com/course",
+ final_url="https://example.com/course",
+ status_code=200,
+ )
+ self.assertEqual(page.headings, ["Agentic AI Curriculum"])
+ self.assertEqual(page.h1, "")
+ self.assertEqual(
+ page.paragraphs,
+ ["Build autonomous AI agents with tools, retrieval, memory, and orchestration."],
+ )
+
if __name__ == "__main__": unittest.main()
diff --git a/tests/test_placement.py b/tests/test_placement.py
index b9ace6d..513a180 100644
--- a/tests/test_placement.py
+++ b/tests/test_placement.py
@@ -105,6 +105,79 @@ def test_destination_intent_prioritizes_cost_context(self, fetch):
for item in items[1:]:
self.assertEqual(item.destination_fit, "low")
+ def test_delightchat_agent_article_matches_agentic_ai_course_despite_menu_noise(self):
+ source = parse_page(
+ """Best Company Data APIs for AI Agents
+
The table below summarizes the main strengths of each provider and the AI agent use cases they are best suited for:
+
Choosing the right company data API for an AI agent requires more than comparing database size or the number of available fields. The API should provide reliable data in a format that autonomous workflows can retrieve, interpret, and use.
+
A company data API lets AI agents access structured business information such as firmographics, workforce data, funding, technologies, and company growth signals for research and automated decisions.
+ """,
+ requested_url="https://www.delightchat.io/blog/best-company-data-apis-for-ai-agents",
+ final_url="https://www.delightchat.io/blog/best-company-data-apis-for-ai-agents",
+ status_code=200,
+ )
+ target = parse_page(
+ """Agentic AI Course with Certificate by IIT Bombay for Working Professionals
+
+
Certificate in Agentic AI
+
Hands-on Agentic AI Curriculum
+
Learn to build autonomous AI agents that reason, act, and collaborate using retrieval augmented generation, Model Context Protocol, LangGraph, CrewAI, tools, memory, and orchestration.
+
What will you learn to build and apply?
+
Apply agentic AI techniques to real-world business use cases involving intelligent workflows, structured data, external tools, multi-agent systems, evaluation, and deployment.