Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
9 changes: 9 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
@@ -1,5 +1,14 @@
# Changelog

## 1.1.3 - 2026-09-26

- Removed navigation, sidebar, footer, and repeated promotional copy from page evidence.
- Added anchor-aware target profiles built from topical headings and substantive body copy.
- Separated broad topic alignment from specialized destination-purpose evidence.
- Allowed strong course-topic matches to return editorial-review opportunities without lowering beta thresholds.
- Prevented table and list introductions from becoming malformed contextual rewrites.
- Added regression coverage for the DelightChat and Great Learning AI-agent course case.

## 1.1.2 - 2026-09-26

- Classify destination pages using on-page evidence with the final URL as a supporting signal.
Expand Down
2 changes: 1 addition & 1 deletion backlink_intelligence/__init__.py
Original file line number Diff line number Diff line change
@@ -1,3 +1,3 @@
"""Backlink Intelligence package."""

__version__ = "1.1.2"
__version__ = "1.1.3"
20 changes: 17 additions & 3 deletions backlink_intelligence/html_utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -34,6 +34,8 @@ def __init__(self, base_url: str) -> None:
self._current_link: dict | None = None
self._skip_depth = 0
self._landmark_stack: list[str] = []
self._seen_headings: set[str] = set()
self._seen_paragraphs: set[str] = set()

def _placement(self) -> str:
landmarks = set(self._landmark_stack)
Expand Down Expand Up @@ -90,16 +92,28 @@ def handle_endtag(self, tag: str) -> None:
self._h1_depth -= 1
if self._heading_tag == tag:
text = clean_text(" ".join(self._heading_parts))
if text:
key = text.casefold()
if (
text
and self._placement() not in {"navigation", "sidebar", "footer"}
and key not in self._seen_headings
):
self.headings.append(text)
self._seen_headings.add(key)
self._heading_tag = None
self._heading_parts = []
if tag == "p" and self._paragraph_depth:
self._paragraph_depth -= 1
if self._paragraph_depth == 0:
text = clean_text(" ".join(self._paragraph_parts))
if text:
key = text.casefold()
if (
text
and self._placement() not in {"navigation", "sidebar", "footer"}
and key not in self._seen_paragraphs
):
self.paragraphs.append(text)
self._seen_paragraphs.add(key)
self._paragraph_parts = []
if tag == "a" and self._current_link is not None:
text = clean_text(" ".join(self._current_link["text"]))
Expand All @@ -126,7 +140,7 @@ def handle_data(self, data: str) -> None:
return
if self._title_depth:
self.title_parts.append(text)
if self._h1_depth:
if self._h1_depth and self._placement() not in {"navigation", "sidebar", "footer"}:
self.h1_parts.append(text)
if self._heading_tag:
self._heading_parts.append(text)
Expand Down
151 changes: 131 additions & 20 deletions backlink_intelligence/placement.py
Original file line number Diff line number Diff line change
Expand Up @@ -113,8 +113,95 @@ def _anchor_with_article(anchor: str) -> str:
return f"{article} {anchor}"


_PROFILE_BOILERPLATE_TERMS = {
"academy",
"certificate",
"certification",
"class",
"course",
"department",
"degree",
"education",
"institute",
"learn",
"learning",
"online",
"offered",
"program",
"programme",
"professional",
"professionals",
"school",
"student",
"students",
"training",
"university",
"working",
}


def _unique_text(items: list[str]) -> list[str]:
output: list[str] = []
seen: set[str] = set()
for item in items:
value = " ".join(item.split()).strip()
key = value.casefold()
if value and key not in seen:
seen.add(key)
output.append(value)
return output


def _target_profile(target: PageEvidence) -> str:
return " ".join([target.title, target.h1, *target.headings, target.text[:12000]])
"""Return the full cleaned page profile used by composition classifiers."""
return " ".join(
_unique_text([target.title, target.h1, *target.headings, *target.paragraphs])
)[:12000]


def _target_topic_evidence(target: PageEvidence, anchor: str) -> tuple[str, list[str]]:
"""Build a compact topical profile without navigation or destination wrappers."""
url_text = _target_url_text(target)
boilerplate_terms = {_stem(term) for term in _PROFILE_BOILERPLATE_TERMS}
anchor_subject_terms = _stems(anchor) - boilerplate_terms
fallback_seed = " ".join([target.title, target.h1, url_text]).strip()
subject_terms = anchor_subject_terms or (_stems(fallback_seed) - boilerplate_terms)
if not subject_terms:
subject_terms = _stems(anchor)
seed = " ".join(sorted(subject_terms))

headings = _unique_text(target.headings)
paragraphs = _unique_text(target.paragraphs)

def is_topical(value: str) -> bool:
return bool(_stems(value) & subject_terms)

ranked_headings = sorted(
(
(similarity(value, seed), index, value)
for index, value in enumerate(headings)
if len(value.split()) >= 3 and is_topical(value)
),
key=lambda item: (-item[0], item[1]),
)
ranked_paragraphs = sorted(
(
(similarity(value, seed), index, value)
for index, value in enumerate(paragraphs)
if len(value.split()) >= 8 and is_topical(value)
),
key=lambda item: (-item[0], item[1]),
)

selected_headings = [value for _, _, value in ranked_headings[:8]]
selected_paragraphs = [value for _, _, value in ranked_paragraphs[:12]]
topic_blocks = selected_paragraphs or selected_headings

primary = [target.title]
if is_topical(target.h1):
primary.append(target.h1)
profile_parts = _unique_text([*primary, url_text, *selected_headings, *selected_paragraphs])
return " ".join(profile_parts), topic_blocks


_TARGET_TYPE_SIGNALS: tuple[tuple[str, tuple[str, ...]], ...] = (
Expand Down Expand Up @@ -563,25 +650,38 @@ def _stems(text: str) -> set[str]:
return {_stem(term) for term in tokens(text)}


def _destination_intent_score(paragraph: str, target: PageEvidence, anchor: str) -> float:
"""Measure fit to destination-specific intent, not just the requested anchor."""
core_profile = " ".join([target.title, target.h1]).strip()
if not core_profile:
core_profile = " ".join(target.headings[:8]).strip()
if not core_profile:
return similarity(paragraph, target.text[:4000])
def _destination_intent_score(
paragraph: str,
target: PageEvidence,
anchor: str,
topic_profile: str,
topic_blocks: list[str],
) -> float:
"""Measure topical fit separately from the destination page's purpose."""
profile_score = similarity(paragraph, topic_profile) if topic_profile else 0.0
block_score = max((similarity(paragraph, block) for block in topic_blocks), default=0.0)
topic_score = (0.70 * block_score) + (0.30 * profile_score)

intent = _target_intent(target)
if intent not in {"cost", "implementation", "governance"}:
# Learning, guide, service, and general resources can support a paragraph that
# shares their subject even when the source does not already mention a course,
# guide, or service. Generated copy remains subject to editorial review.
return round(topic_score, 4)

signals = next(
(values for kind, values in _TARGET_TYPE_SIGNALS if kind == intent),
(),
)
paragraph_terms = _stems(paragraph)
core_terms = _stems(core_profile)
anchor_terms = _stems(anchor)

# Prefer terms that describe what makes the destination distinct from the anchor.
intent_terms = core_terms - anchor_terms
if len(intent_terms) < 2:
intent_terms = core_terms
intent_overlap = len(paragraph_terms & intent_terms) / max(len(intent_terms), 1)
semantic = similarity(paragraph, core_profile)
return round((0.35 * semantic) + (0.65 * intent_overlap), 4)
purpose_hits = sum(bool(paragraph_terms & _stems(value)) for value in signals)
if purpose_hits == 0:
# Specialized destinations need evidence of their specific purpose. This keeps
# generic topical paragraphs from outranking pricing, implementation, or
# governance context merely because they repeat the anchor.
return round(topic_score * 0.35, 4)
purpose_score = min(1.0, purpose_hits / 2.0)
return round((0.70 * topic_score) + (0.30 * purpose_score), 4)


def _destination_level(score: float) -> str:
Expand All @@ -608,15 +708,26 @@ def rank_placements(
return []

anchor, anchor_warnings = _select_anchor(preferred_anchor, target.title)
target_profile = _target_profile(target)
target_profile, target_topic_blocks = _target_topic_evidence(target, anchor)
candidates: list[tuple[float, float, int, str]] = []

for i, paragraph in enumerate(source.paragraphs, start=1):
wc = len(paragraph.split())
if wc < 18 or wc > 260:
continue
# A standalone paragraph ending in a colon or semicolon normally introduces
# a table or list. Treating it as complete prose produces malformed rewrites
# and weak placements immediately before structured content.
if paragraph.rstrip().endswith((":", ";")):
continue
semantic_score = similarity(paragraph, target_profile)
destination_score = _destination_intent_score(paragraph, target, anchor)
destination_score = _destination_intent_score(
paragraph,
target,
anchor,
target_profile,
target_topic_blocks,
)
anchor_terms = set(tokens(anchor))
anchor_overlap = len(anchor_terms & set(tokens(paragraph))) / max(len(anchor_terms), 1)
# Destination intent gets meaningful weight so a pricing/cost paragraph beats a
Expand Down
2 changes: 1 addition & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"

[project]
name = "backlink-intelligence"
version = "1.1.2"
version = "1.1.3"
description = "Open-source backlink intelligence based on evidence, context, and editorial fit."
readme = "README.md"
requires-python = ">=3.11"
Expand Down
2 changes: 1 addition & 1 deletion tests/test_cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,7 +8,7 @@

class CLITests(unittest.TestCase):
def test_version_is_stable(self):
self.assertEqual(__version__, "1.1.2")
self.assertEqual(__version__, "1.1.3")

def test_status_command(self):
output = io.StringIO()
Expand Down
22 changes: 22 additions & 0 deletions tests/test_html_utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -22,5 +22,27 @@ def test_link_context_and_placement(self):
target = next(l for l in self.page.links if "target.com" in l.href); self.assertEqual(target.text, "agentic AI course"); self.assertEqual(target.placement, "editorial_context"); self.assertIn("nofollow", target.rel)
footer = next(l for l in self.page.links if "social.example" in l.href); self.assertEqual(footer.placement, "footer")

def test_navigation_sidebar_footer_and_duplicates_do_not_pollute_page_copy(self):
page = parse_page(
"""<html><body>
<nav><h1>Navigation Heading</h1><h2>Repeated Course Menu</h2><p>Repeated promotional navigation copy.</p></nav>
<aside><h2>Related Programs</h2><p>Sidebar promotion for another course.</p></aside>
<main><h2>Agentic AI Curriculum</h2>
<p>Build autonomous AI agents with tools, retrieval, memory, and orchestration.</p>
<h2>Agentic AI Curriculum</h2>
<p>Build autonomous AI agents with tools, retrieval, memory, and orchestration.</p></main>
<footer><p>Footer promotional links and legal navigation.</p></footer>
</body></html>""",
requested_url="https://example.com/course",
final_url="https://example.com/course",
status_code=200,
)
self.assertEqual(page.headings, ["Agentic AI Curriculum"])
self.assertEqual(page.h1, "")
self.assertEqual(
page.paragraphs,
["Build autonomous AI agents with tools, retrieval, memory, and orchestration."],
)


if __name__ == "__main__": unittest.main()
73 changes: 73 additions & 0 deletions tests/test_placement.py
Original file line number Diff line number Diff line change
Expand Up @@ -105,6 +105,79 @@ def test_destination_intent_prioritizes_cost_context(self, fetch):
for item in items[1:]:
self.assertEqual(item.destination_fit, "low")

def test_delightchat_agent_article_matches_agentic_ai_course_despite_menu_noise(self):
source = parse_page(
"""<title>Best Company Data APIs for AI Agents</title><main><article>
<p>The table below summarizes the main strengths of each provider and the AI agent use cases they are best suited for:</p>
<p>Choosing the right company data API for an AI agent requires more than comparing database size or the number of available fields. The API should provide reliable data in a format that autonomous workflows can retrieve, interpret, and use.</p>
<p>A company data API lets AI agents access structured business information such as firmographics, workforce data, funding, technologies, and company growth signals for research and automated decisions.</p>
</article></main>""",
requested_url="https://www.delightchat.io/blog/best-company-data-apis-for-ai-agents",
final_url="https://www.delightchat.io/blog/best-company-data-apis-for-ai-agents",
status_code=200,
)
target = parse_page(
"""<title>Agentic AI Course with Certificate by IIT Bombay for Working Professionals</title>
<body><nav>
<h2>PG Program in Artificial Intelligence and Machine Learning</h2>
<p>Browse online degrees, certificates, bootcamps, and professional programs.</p>
<h2>Certificate Program in Data Science</h2>
<p>Browse online degrees, certificates, bootcamps, and professional programs.</p>
</nav><div class="main">
<h1>Certificate in Agentic AI</h1>
<h2>Hands-on Agentic AI Curriculum</h2>
<p>Learn to build autonomous AI agents that reason, act, and collaborate using retrieval augmented generation, Model Context Protocol, LangGraph, CrewAI, tools, memory, and orchestration.</p>
<h2>What will you learn to build and apply?</h2>
<p>Apply agentic AI techniques to real-world business use cases involving intelligent workflows, structured data, external tools, multi-agent systems, evaluation, and deployment.</p>
</div></body>""",
requested_url="https://www.mygreatlearning.com/iit-bombay-certificate-in-agentic-ai",
final_url="https://www.mygreatlearning.com/iit-bombay-certificate-in-agentic-ai",
status_code=200,
)
items = rank_placements(
source,
target,
"AI agent course",
target.final_url,
top_n=3,
min_context_score=0.15,
min_destination_score=0.08,
)
self.assertGreaterEqual(len(items), 1)
self.assertTrue(items[0].review_required)
self.assertEqual(items[0].recommendation_status, "manual_review")
self.assertIn("AI agent", items[0].before)
self.assertFalse(items[0].before.rstrip().endswith((':', ';')))
self.assertGreaterEqual(items[0].score, 0.15)
self.assertGreaterEqual(items[0].destination_score, 0.08)

def test_unrelated_course_does_not_pass_on_generic_course_language(self):
source = parse_page(
"""<main><p>AI agents retrieve company data, coordinate tools, monitor business changes, and support automated research decisions across connected workflows.</p></main>""",
requested_url="https://source.example/ai-agents",
final_url="https://source.example/ai-agents",
status_code=200,
)
target = parse_page(
"""<title>Professional Watercolor Painting Course</title><main>
<h1>Learn Watercolor Painting</h1>
<p>Study color mixing, brush control, paper selection, washes, composition, landscapes, and portrait painting through guided studio exercises.</p>
</main>""",
requested_url="https://target.example/watercolor-course",
final_url="https://target.example/watercolor-course",
status_code=200,
)
items = rank_placements(
source,
target,
"AI agent course",
target.final_url,
top_n=3,
min_context_score=0.15,
min_destination_score=0.08,
)
self.assertEqual(items, [])


@patch("backlink_intelligence.placement.fetch_page")
def test_contextual_sentence_avoids_target_title_dump(self, fetch):
Expand Down
Loading