From 8e852a8643a59347ca7b5d9499d0a895b3db6e8c Mon Sep 17 00:00:00 2001 From: Codex Date: Sat, 26 Sep 2026 23:14:41 +0530 Subject: [PATCH] Reduce false no-match placement results for v1.1.3 --- CHANGELOG.md | 9 ++ backlink_intelligence/__init__.py | 2 +- backlink_intelligence/html_utils.py | 20 +++- backlink_intelligence/placement.py | 151 ++++++++++++++++++++++++---- pyproject.toml | 2 +- tests/test_cli.py | 2 +- tests/test_html_utils.py | 22 ++++ tests/test_placement.py | 73 ++++++++++++++ 8 files changed, 255 insertions(+), 26 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index f90d37c..5b9fc6a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,14 @@ # Changelog +## 1.1.3 - 2026-09-26 + +- Removed navigation, sidebar, footer, and repeated promotional copy from page evidence. +- Added anchor-aware target profiles built from topical headings and substantive body copy. +- Separated broad topic alignment from specialized destination-purpose evidence. +- Allowed strong course-topic matches to return editorial-review opportunities without lowering beta thresholds. +- Prevented table and list introductions from becoming malformed contextual rewrites. +- Added regression coverage for the DelightChat and Great Learning AI-agent course case. + ## 1.1.2 - 2026-09-26 - Classify destination pages using on-page evidence with the final URL as a supporting signal. diff --git a/backlink_intelligence/__init__.py b/backlink_intelligence/__init__.py index e31a0f9..bc5bef7 100644 --- a/backlink_intelligence/__init__.py +++ b/backlink_intelligence/__init__.py @@ -1,3 +1,3 @@ """Backlink Intelligence package.""" -__version__ = "1.1.2" +__version__ = "1.1.3" diff --git a/backlink_intelligence/html_utils.py b/backlink_intelligence/html_utils.py index 07dc09f..4925243 100644 --- a/backlink_intelligence/html_utils.py +++ b/backlink_intelligence/html_utils.py @@ -34,6 +34,8 @@ def __init__(self, base_url: str) -> None: self._current_link: dict | None = None self._skip_depth = 0 self._landmark_stack: list[str] = [] + self._seen_headings: set[str] = set() + self._seen_paragraphs: set[str] = set() def _placement(self) -> str: landmarks = set(self._landmark_stack) @@ -90,16 +92,28 @@ def handle_endtag(self, tag: str) -> None: self._h1_depth -= 1 if self._heading_tag == tag: text = clean_text(" ".join(self._heading_parts)) - if text: + key = text.casefold() + if ( + text + and self._placement() not in {"navigation", "sidebar", "footer"} + and key not in self._seen_headings + ): self.headings.append(text) + self._seen_headings.add(key) self._heading_tag = None self._heading_parts = [] if tag == "p" and self._paragraph_depth: self._paragraph_depth -= 1 if self._paragraph_depth == 0: text = clean_text(" ".join(self._paragraph_parts)) - if text: + key = text.casefold() + if ( + text + and self._placement() not in {"navigation", "sidebar", "footer"} + and key not in self._seen_paragraphs + ): self.paragraphs.append(text) + self._seen_paragraphs.add(key) self._paragraph_parts = [] if tag == "a" and self._current_link is not None: text = clean_text(" ".join(self._current_link["text"])) @@ -126,7 +140,7 @@ def handle_data(self, data: str) -> None: return if self._title_depth: self.title_parts.append(text) - if self._h1_depth: + if self._h1_depth and self._placement() not in {"navigation", "sidebar", "footer"}: self.h1_parts.append(text) if self._heading_tag: self._heading_parts.append(text) diff --git a/backlink_intelligence/placement.py b/backlink_intelligence/placement.py index 310e121..c2bfa0c 100644 --- a/backlink_intelligence/placement.py +++ b/backlink_intelligence/placement.py @@ -113,8 +113,95 @@ def _anchor_with_article(anchor: str) -> str: return f"{article} {anchor}" +_PROFILE_BOILERPLATE_TERMS = { + "academy", + "certificate", + "certification", + "class", + "course", + "department", + "degree", + "education", + "institute", + "learn", + "learning", + "online", + "offered", + "program", + "programme", + "professional", + "professionals", + "school", + "student", + "students", + "training", + "university", + "working", +} + + +def _unique_text(items: list[str]) -> list[str]: + output: list[str] = [] + seen: set[str] = set() + for item in items: + value = " ".join(item.split()).strip() + key = value.casefold() + if value and key not in seen: + seen.add(key) + output.append(value) + return output + + def _target_profile(target: PageEvidence) -> str: - return " ".join([target.title, target.h1, *target.headings, target.text[:12000]]) + """Return the full cleaned page profile used by composition classifiers.""" + return " ".join( + _unique_text([target.title, target.h1, *target.headings, *target.paragraphs]) + )[:12000] + + +def _target_topic_evidence(target: PageEvidence, anchor: str) -> tuple[str, list[str]]: + """Build a compact topical profile without navigation or destination wrappers.""" + url_text = _target_url_text(target) + boilerplate_terms = {_stem(term) for term in _PROFILE_BOILERPLATE_TERMS} + anchor_subject_terms = _stems(anchor) - boilerplate_terms + fallback_seed = " ".join([target.title, target.h1, url_text]).strip() + subject_terms = anchor_subject_terms or (_stems(fallback_seed) - boilerplate_terms) + if not subject_terms: + subject_terms = _stems(anchor) + seed = " ".join(sorted(subject_terms)) + + headings = _unique_text(target.headings) + paragraphs = _unique_text(target.paragraphs) + + def is_topical(value: str) -> bool: + return bool(_stems(value) & subject_terms) + + ranked_headings = sorted( + ( + (similarity(value, seed), index, value) + for index, value in enumerate(headings) + if len(value.split()) >= 3 and is_topical(value) + ), + key=lambda item: (-item[0], item[1]), + ) + ranked_paragraphs = sorted( + ( + (similarity(value, seed), index, value) + for index, value in enumerate(paragraphs) + if len(value.split()) >= 8 and is_topical(value) + ), + key=lambda item: (-item[0], item[1]), + ) + + selected_headings = [value for _, _, value in ranked_headings[:8]] + selected_paragraphs = [value for _, _, value in ranked_paragraphs[:12]] + topic_blocks = selected_paragraphs or selected_headings + + primary = [target.title] + if is_topical(target.h1): + primary.append(target.h1) + profile_parts = _unique_text([*primary, url_text, *selected_headings, *selected_paragraphs]) + return " ".join(profile_parts), topic_blocks _TARGET_TYPE_SIGNALS: tuple[tuple[str, tuple[str, ...]], ...] = ( @@ -563,25 +650,38 @@ def _stems(text: str) -> set[str]: return {_stem(term) for term in tokens(text)} -def _destination_intent_score(paragraph: str, target: PageEvidence, anchor: str) -> float: - """Measure fit to destination-specific intent, not just the requested anchor.""" - core_profile = " ".join([target.title, target.h1]).strip() - if not core_profile: - core_profile = " ".join(target.headings[:8]).strip() - if not core_profile: - return similarity(paragraph, target.text[:4000]) +def _destination_intent_score( + paragraph: str, + target: PageEvidence, + anchor: str, + topic_profile: str, + topic_blocks: list[str], +) -> float: + """Measure topical fit separately from the destination page's purpose.""" + profile_score = similarity(paragraph, topic_profile) if topic_profile else 0.0 + block_score = max((similarity(paragraph, block) for block in topic_blocks), default=0.0) + topic_score = (0.70 * block_score) + (0.30 * profile_score) + intent = _target_intent(target) + if intent not in {"cost", "implementation", "governance"}: + # Learning, guide, service, and general resources can support a paragraph that + # shares their subject even when the source does not already mention a course, + # guide, or service. Generated copy remains subject to editorial review. + return round(topic_score, 4) + + signals = next( + (values for kind, values in _TARGET_TYPE_SIGNALS if kind == intent), + (), + ) paragraph_terms = _stems(paragraph) - core_terms = _stems(core_profile) - anchor_terms = _stems(anchor) - - # Prefer terms that describe what makes the destination distinct from the anchor. - intent_terms = core_terms - anchor_terms - if len(intent_terms) < 2: - intent_terms = core_terms - intent_overlap = len(paragraph_terms & intent_terms) / max(len(intent_terms), 1) - semantic = similarity(paragraph, core_profile) - return round((0.35 * semantic) + (0.65 * intent_overlap), 4) + purpose_hits = sum(bool(paragraph_terms & _stems(value)) for value in signals) + if purpose_hits == 0: + # Specialized destinations need evidence of their specific purpose. This keeps + # generic topical paragraphs from outranking pricing, implementation, or + # governance context merely because they repeat the anchor. + return round(topic_score * 0.35, 4) + purpose_score = min(1.0, purpose_hits / 2.0) + return round((0.70 * topic_score) + (0.30 * purpose_score), 4) def _destination_level(score: float) -> str: @@ -608,15 +708,26 @@ def rank_placements( return [] anchor, anchor_warnings = _select_anchor(preferred_anchor, target.title) - target_profile = _target_profile(target) + target_profile, target_topic_blocks = _target_topic_evidence(target, anchor) candidates: list[tuple[float, float, int, str]] = [] for i, paragraph in enumerate(source.paragraphs, start=1): wc = len(paragraph.split()) if wc < 18 or wc > 260: continue + # A standalone paragraph ending in a colon or semicolon normally introduces + # a table or list. Treating it as complete prose produces malformed rewrites + # and weak placements immediately before structured content. + if paragraph.rstrip().endswith((":", ";")): + continue semantic_score = similarity(paragraph, target_profile) - destination_score = _destination_intent_score(paragraph, target, anchor) + destination_score = _destination_intent_score( + paragraph, + target, + anchor, + target_profile, + target_topic_blocks, + ) anchor_terms = set(tokens(anchor)) anchor_overlap = len(anchor_terms & set(tokens(paragraph))) / max(len(anchor_terms), 1) # Destination intent gets meaningful weight so a pricing/cost paragraph beats a diff --git a/pyproject.toml b/pyproject.toml index 65a6cc0..8dc47d8 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "backlink-intelligence" -version = "1.1.2" +version = "1.1.3" description = "Open-source backlink intelligence based on evidence, context, and editorial fit." readme = "README.md" requires-python = ">=3.11" diff --git a/tests/test_cli.py b/tests/test_cli.py index 0ad19d7..158f8f3 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -8,7 +8,7 @@ class CLITests(unittest.TestCase): def test_version_is_stable(self): - self.assertEqual(__version__, "1.1.2") + self.assertEqual(__version__, "1.1.3") def test_status_command(self): output = io.StringIO() diff --git a/tests/test_html_utils.py b/tests/test_html_utils.py index 4d37ae0..d8f66a5 100644 --- a/tests/test_html_utils.py +++ b/tests/test_html_utils.py @@ -22,5 +22,27 @@ def test_link_context_and_placement(self): target = next(l for l in self.page.links if "target.com" in l.href); self.assertEqual(target.text, "agentic AI course"); self.assertEqual(target.placement, "editorial_context"); self.assertIn("nofollow", target.rel) footer = next(l for l in self.page.links if "social.example" in l.href); self.assertEqual(footer.placement, "footer") + def test_navigation_sidebar_footer_and_duplicates_do_not_pollute_page_copy(self): + page = parse_page( + """ + + +

Agentic AI Curriculum

+

Build autonomous AI agents with tools, retrieval, memory, and orchestration.

+

Agentic AI Curriculum

+

Build autonomous AI agents with tools, retrieval, memory, and orchestration.

+ + """, + requested_url="https://example.com/course", + final_url="https://example.com/course", + status_code=200, + ) + self.assertEqual(page.headings, ["Agentic AI Curriculum"]) + self.assertEqual(page.h1, "") + self.assertEqual( + page.paragraphs, + ["Build autonomous AI agents with tools, retrieval, memory, and orchestration."], + ) + if __name__ == "__main__": unittest.main() diff --git a/tests/test_placement.py b/tests/test_placement.py index b9ace6d..513a180 100644 --- a/tests/test_placement.py +++ b/tests/test_placement.py @@ -105,6 +105,79 @@ def test_destination_intent_prioritizes_cost_context(self, fetch): for item in items[1:]: self.assertEqual(item.destination_fit, "low") + def test_delightchat_agent_article_matches_agentic_ai_course_despite_menu_noise(self): + source = parse_page( + """Best Company Data APIs for AI Agents
+

The table below summarizes the main strengths of each provider and the AI agent use cases they are best suited for:

+

Choosing the right company data API for an AI agent requires more than comparing database size or the number of available fields. The API should provide reliable data in a format that autonomous workflows can retrieve, interpret, and use.

+

A company data API lets AI agents access structured business information such as firmographics, workforce data, funding, technologies, and company growth signals for research and automated decisions.

+
""", + requested_url="https://www.delightchat.io/blog/best-company-data-apis-for-ai-agents", + final_url="https://www.delightchat.io/blog/best-company-data-apis-for-ai-agents", + status_code=200, + ) + target = parse_page( + """Agentic AI Course with Certificate by IIT Bombay for Working Professionals +
+

Certificate in Agentic AI

+

Hands-on Agentic AI Curriculum

+

Learn to build autonomous AI agents that reason, act, and collaborate using retrieval augmented generation, Model Context Protocol, LangGraph, CrewAI, tools, memory, and orchestration.

+

What will you learn to build and apply?

+

Apply agentic AI techniques to real-world business use cases involving intelligent workflows, structured data, external tools, multi-agent systems, evaluation, and deployment.

+
""", + requested_url="https://www.mygreatlearning.com/iit-bombay-certificate-in-agentic-ai", + final_url="https://www.mygreatlearning.com/iit-bombay-certificate-in-agentic-ai", + status_code=200, + ) + items = rank_placements( + source, + target, + "AI agent course", + target.final_url, + top_n=3, + min_context_score=0.15, + min_destination_score=0.08, + ) + self.assertGreaterEqual(len(items), 1) + self.assertTrue(items[0].review_required) + self.assertEqual(items[0].recommendation_status, "manual_review") + self.assertIn("AI agent", items[0].before) + self.assertFalse(items[0].before.rstrip().endswith((':', ';'))) + self.assertGreaterEqual(items[0].score, 0.15) + self.assertGreaterEqual(items[0].destination_score, 0.08) + + def test_unrelated_course_does_not_pass_on_generic_course_language(self): + source = parse_page( + """

AI agents retrieve company data, coordinate tools, monitor business changes, and support automated research decisions across connected workflows.

""", + requested_url="https://source.example/ai-agents", + final_url="https://source.example/ai-agents", + status_code=200, + ) + target = parse_page( + """Professional Watercolor Painting Course
+

Learn Watercolor Painting

+

Study color mixing, brush control, paper selection, washes, composition, landscapes, and portrait painting through guided studio exercises.

+
""", + requested_url="https://target.example/watercolor-course", + final_url="https://target.example/watercolor-course", + status_code=200, + ) + items = rank_placements( + source, + target, + "AI agent course", + target.final_url, + top_n=3, + min_context_score=0.15, + min_destination_score=0.08, + ) + self.assertEqual(items, []) + @patch("backlink_intelligence.placement.fetch_page") def test_contextual_sentence_avoids_target_title_dump(self, fetch):