From f3d1628d3b725f5bdd3f8e65a90940db11645823 Mon Sep 17 00:00:00 2001 From: Codex Date: Sat, 26 Sep 2026 23:37:12 +0530 Subject: [PATCH] Add context-aware deterministic placement copy --- CHANGELOG.md | 8 + backlink_intelligence/__init__.py | 2 +- backlink_intelligence/placement.py | 244 ++++++++++++++++++++++------- pyproject.toml | 2 +- tests/test_cli.py | 2 +- tests/test_placement.py | 83 +++++++--- 6 files changed, 259 insertions(+), 82 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 5b9fc6a..7dcd2a6 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,13 @@ # Changelog +## 1.1.4 - 2026-09-26 + +- Replace the repeated learning-placement clause with sentence-function-aware copy. +- Use distinct deterministic wording for definitions, requirements, workflows, benefits, skills, and general context. +- Introduce target audiences only when the source paragraph addresses the same audience. +- Avoid unsafe rewrites of long, definitional, or evaluation-focused sentences. +- Diversify ranked opportunities while preserving deterministic output and API compatibility. + ## 1.1.3 - 2026-09-26 - Removed navigation, sidebar, footer, and repeated promotional copy from page evidence. diff --git a/backlink_intelligence/__init__.py b/backlink_intelligence/__init__.py index bc5bef7..ab6e6b1 100644 --- a/backlink_intelligence/__init__.py +++ b/backlink_intelligence/__init__.py @@ -1,3 +1,3 @@ """Backlink Intelligence package.""" -__version__ = "1.1.3" +__version__ = "1.1.4" diff --git a/backlink_intelligence/placement.py b/backlink_intelligence/placement.py index c2bfa0c..aa741a6 100644 --- a/backlink_intelligence/placement.py +++ b/backlink_intelligence/placement.py @@ -316,11 +316,6 @@ def _target_audience(target: PageEvidence) -> str | None: return None -def _audience_subject(audience: str) -> str: - """Uppercase a sentence-initial audience without damaging acronyms such as SEO.""" - return audience[:1].upper() + audience[1:] - - def _context_kind(text: str) -> str: lower = text.casefold() groups = ( @@ -336,6 +331,78 @@ def _context_kind(text: str) -> str: return "general" +def _sentence_function(text: str) -> str: + """Classify what the publisher's sentence is doing, not merely its topic. + + This deliberately runs before broad topical markers such as ``agent``. The old + composer treated nearly every AI-agent sentence as an application and attached + the same ``a practical application ...`` clause, even to definitions and + evaluation criteria. + """ + lower = " ".join(text.casefold().split()) + groups = ( + ( + "requirement", + ( + " should ", + " must ", + " requires ", + " require ", + " needs ", + " need ", + "choosing ", + "evaluating ", + "consider ", + "criteria", + "right ", + ), + ), + ( + "definition", + ( + " is an ", + " is a ", + " are an ", + " are a ", + "refers to", + "means ", + "defined as", + ), + ), + ( + "workflow", + ( + "workflow", + "process", + "step", + "retrieve", + "monitor", + "coordinate", + "route", + "interpret", + ), + ), + ( + "benefit", + ("benefit", "outcome", "improve", "reduce", "increase", "enable", "support"), + ), + ( + "application", + ("application", "use case", "in practice", "perform", "automate", "apply"), + ), + ) + padded = f" {lower} " + for function, markers in groups: + if any(marker in padded for marker in markers): + return function + return "general" + + +def _audience_fits_source(audience: str | None, paragraph: str) -> bool: + """Do not introduce an audience label that the publisher did not address.""" + return bool(audience and _contains_term(paragraph, audience)) + + def _sentence_spans(paragraph: str) -> list[tuple[int, int, str]]: """Split prose conservatively while retaining exact offsets and punctuation.""" boundaries = list(re.finditer(r"(?<=[.!?])\s+(?=[A-Z0-9\"'\u201c\u2018])", paragraph)) @@ -356,6 +423,7 @@ def _rewrite_candidate( anchor: str, target_url: str, target: PageEvidence, + variant: int = 0, ) -> tuple[str, str, list[TextSegment], str, list[str]] | None: """Integrate an anchor into one supported sentence without replacing source words.""" intent = _target_intent(target) @@ -366,9 +434,16 @@ def _rewrite_candidate( for start, end, sentence in _sentence_spans(paragraph): overlap = _meaningful_overlap(sentence, target, anchor) kind = _context_kind(sentence) - if overlap < 2 or kind not in {"application", "skills", "implementation"}: + function = _sentence_function(sentence) + if ( + overlap < 2 + or start != 0 + or len(sentence.split()) > 18 + or kind not in {"application", "skills", "implementation"} + or function in {"definition", "requirement"} + ): continue - ranked.append((overlap, similarity(sentence, _target_profile(target)), start, end, kind)) + ranked.append((overlap, similarity(sentence, _target_profile(target)), start, end, function)) if not ranked: return None @@ -385,7 +460,7 @@ def _rewrite_candidate( article_prefix = linked_phrase[:anchor_offset] audience = _target_audience(target) if intent == "learning" and kind == "application": - descriptor, active, passive = "a practical application", "examine more deeply", "examined more deeply" + descriptor, active, passive = "an application", "examine in greater depth", "examined in greater depth" elif intent == "learning" and kind == "skills": descriptor, active, passive = "a topic", "study more deeply", "studied more deeply" elif intent in {"service", "implementation"}: @@ -394,10 +469,18 @@ def _rewrite_candidate( descriptor, active, passive = "a consideration", "examine further", "examined further" else: descriptor, active, passive = "a topic", "explore further", "explored further" - if audience: - clause = f", {descriptor} {audience} can {active} through " + # Audience labels are not grammatical decorations. They are used only when the + # publisher already addresses the same audience; otherwise neutral copy is safer. + use_audience = _audience_fits_source(audience, paragraph) + neutral_clauses = ( + f", {descriptor} {passive} through ", + f", an example explored in greater depth through ", + f", a related concept covered in ", + ) + if use_audience: + clause = f", {descriptor} that {audience} can {active} through " else: - clause = f", {descriptor} that can be {passive} through " + clause = neutral_clauses[variant % len(neutral_clauses)] prefix = paragraph[:start] + sentence_body + clause + article_prefix suffix = terminal + sentence[len(stripped) :] + paragraph[end:] @@ -411,7 +494,7 @@ def _rewrite_candidate( ] notes.append( "target_audience_used_for_contextual_sentence" - if audience + if use_audience else "neutral_audience_wording_used" ) if placed_anchor != anchor: @@ -423,12 +506,14 @@ def _contextual_fallback_sentence( paragraph: str, anchor: str, target: PageEvidence, + variant: int = 0, ) -> tuple[str, str, str, list[str]]: """Create concise deterministic fallback copy without dumping the target title.""" placed_anchor = _fallback_anchor_case(anchor) intent = _target_intent(target) audience = _target_audience(target) context = _context_kind(paragraph) + function = _sentence_function(paragraph) paragraph_lower = paragraph.lower() cost_intent = intent == "cost" @@ -459,56 +544,100 @@ def _contextual_fallback_sentence( anchor_phrase = _anchor_with_article(placed_anchor) anchor_offset = anchor_phrase.rfind(placed_anchor) article_prefix = anchor_phrase[:anchor_offset] - if intent == "learning" and context == "application": - notes.append("destination_intent_used_for_contextual_sentence") - if audience: - notes.append("target_audience_used_for_contextual_sentence") - return ( - f"{_audience_subject(audience)} can examine these applications more deeply through {article_prefix}", - ".", - placed_anchor, - notes, - ) - notes.append("neutral_audience_wording_used") - return ( - f"These applications can be examined more deeply through {article_prefix}", - ".", - placed_anchor, - notes, - ) - - if intent == "learning" and context == "skills": - notes.append("destination_intent_used_for_contextual_sentence") - if audience: - notes.append("target_audience_used_for_contextual_sentence") - return ( - f"{_audience_subject(audience)} can develop these skills further through {article_prefix}", - ".", - placed_anchor, - notes, - ) - notes.append("neutral_audience_wording_used") - return ( - f"These skills can be developed further through {article_prefix}", - ".", - placed_anchor, - notes, - ) + sentence_article_prefix = article_prefix[:1].upper() + article_prefix[1:] + use_audience = _audience_fits_source(audience, paragraph) if intent == "learning": notes.append("destination_intent_used_for_contextual_sentence") - if audience: + if use_audience: notes.append("target_audience_used_for_contextual_sentence") return ( - f"{_audience_subject(audience)} can explore this subject further through {article_prefix}", - ".", + f"For {audience}, {article_prefix}", + " provides a structured way to explore this subject in greater depth.", placed_anchor, notes, ) + notes.append("neutral_audience_wording_used") + if function == "definition": + options = ( + ( + f"{sentence_article_prefix}", + " can provide additional context for how this technology works in practice.", + ), + ( + "The role of this technology in agent systems can be explored further through " + article_prefix, + ".", + ), + ( + f"{sentence_article_prefix}", + " can help connect this definition with practical agent workflows.", + ), + ) + elif function == "requirement": + options = ( + ( + f"{sentence_article_prefix}", + " can provide additional context for evaluating these requirements in practice.", + ), + ( + "These criteria can also be examined through " + article_prefix, + ".", + ), + ( + f"{sentence_article_prefix}", + " can help explain how these considerations affect real-world agent systems.", + ), + ) + elif function == "workflow" or context == "implementation": + options = ( + ( + f"{sentence_article_prefix}", + " can help explain how these workflows are designed and applied.", + ), + ( + "These workflows can be studied in greater depth through " + article_prefix, + ".", + ), + ( + f"{sentence_article_prefix}", + " offers a structured way to explore the methods behind these workflows.", + ), + ) + elif function == "benefit": + options = ( + ( + "The methods behind these outcomes can be studied further through " + article_prefix, + ".", + ), + ( + f"{sentence_article_prefix}", + " can provide additional context for understanding these outcomes.", + ), + ( + "The concepts supporting these benefits can be explored through " + article_prefix, + ".", + ), + ) + elif context == "skills": + options = ( + ("These skills can be developed further through " + article_prefix, "."), + (f"{sentence_article_prefix}", " can provide a structured path for developing these skills."), + ("A deeper treatment of these capabilities is available through " + article_prefix, "."), + ) + else: + options = ( + ( + f"{sentence_article_prefix}", + " can provide a structured way to explore how these concepts work in practice.", + ), + ("These concepts can be examined in greater depth through " + article_prefix, "."), + (f"{sentence_article_prefix}", " offers additional context for applying these ideas."), + ) + sentence_prefix, sentence_suffix = options[variant % len(options)] return ( - f"This subject can be explored further through {article_prefix}", - ".", + sentence_prefix, + sentence_suffix, placed_anchor, notes, ) @@ -564,6 +693,7 @@ def _compose_after( anchor: str, target_url: str, target: PageEvidence, + variant: int = 0, ) -> tuple[str, str, str, list[TextSegment], str, list[str]]: """Compose the draft while preserving source grammar/capitalization when possible.""" exact = _find_complete_phrase(paragraph, anchor) @@ -582,8 +712,8 @@ def _compose_after( # If the exact requested form is not present, prefer a complete natural word-form # already in the publisher copy instead of creating artifacts such as [AI Agent]s. - for variant in _simple_anchor_variants(anchor): - match = _find_complete_phrase(paragraph, variant) + for anchor_variant in _simple_anchor_variants(anchor): + match = _find_complete_phrase(paragraph, anchor_variant) if match is not None: placed_anchor = match.group(0) linked = f"[{placed_anchor}]({target_url})" @@ -601,13 +731,13 @@ def _compose_after( ["anchor_adapted_to_source_grammar", "requested_anchor_not_used_verbatim"], ) - rewrite = _rewrite_candidate(paragraph, anchor, target_url, target) + rewrite = _rewrite_candidate(paragraph, anchor, target_url, target, variant) if rewrite is not None: after, after_text, segments, placed_anchor, notes = rewrite return "contextual_sentence", after, after_text, segments, placed_anchor, notes sentence_prefix, sentence_suffix, placed_anchor, notes = _contextual_fallback_sentence( - paragraph, anchor, target + paragraph, anchor, target, variant ) prefix = paragraph.rstrip() + " " + sentence_prefix after_text = prefix + placed_anchor + sentence_suffix @@ -740,7 +870,7 @@ def rank_placements( suggestions: list[PlacementSuggestion] = [] for rank, (score, destination_score, index, paragraph) in enumerate(candidates[: max(top_n, 1)], start=1): strategy, after, after_text, after_segments, placed_anchor, compose_notes = _compose_after( - paragraph, anchor, target_url, target + paragraph, anchor, target_url, target, rank - 1 ) original_words = max(len(paragraph.split()), 1) after_words = len(after_text.split()) diff --git a/pyproject.toml b/pyproject.toml index 8dc47d8..a76e34a 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "backlink-intelligence" -version = "1.1.3" +version = "1.1.4" description = "Open-source backlink intelligence based on evidence, context, and editorial fit." readme = "README.md" requires-python = ">=3.11" diff --git a/tests/test_cli.py b/tests/test_cli.py index 158f8f3..1b900d2 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -8,7 +8,7 @@ class CLITests(unittest.TestCase): def test_version_is_stable(self): - self.assertEqual(__version__, "1.1.3") + self.assertEqual(__version__, "1.1.4") def test_status_command(self): output = io.StringIO() diff --git a/tests/test_placement.py b/tests/test_placement.py index 513a180..4cf840c 100644 --- a/tests/test_placement.py +++ b/tests/test_placement.py @@ -230,8 +230,8 @@ def test_finance_course_anchor_is_integrated_into_one_supported_sentence(self): item = rank_placements(source, target, "ai in finance course", "https://t.com/course", top_n=1)[0] expected = ( - "Financial Agents now perform Continuous Accounting, a practical application " - "that can be examined more deeply through an ai in finance course. " + "Financial Agents now perform Continuous Accounting, an application " + "examined in greater depth through an ai in finance course. " "Instead of waiting for month-end, agents monitor every transaction in real time across global entities." ) self.assertEqual(item.after_text, expected) @@ -259,10 +259,8 @@ def test_rewrite_changes_only_one_sentence_and_preserves_protected_facts(self): self.assertIn("FinanceCo", item.after_text) self.assertIn('"continuous accounting"', item.after_text) self.assertIn("14 entities", item.after_text) - self.assertEqual( - item.after_text.count("a practical application that can be examined more deeply"), - 1, - ) + self.assertNotIn("a practical application", item.after_text.casefold()) + self.assertEqual(len([segment for segment in item.after_segments if segment.type == "link"]), 1) self.assertGreaterEqual(item.preservation_percent, 99.0) def test_unsupported_learning_relationship_uses_context_specific_append(self): @@ -280,8 +278,8 @@ def test_unsupported_learning_relationship_uses_context_specific_append(self): self.assertTrue(item.after_text.startswith(source.paragraphs[0])) self.assertNotIn("source_sentence_lightly_rewritten", item.reasons) self.assertNotIn("Readers who want additional context", item.after_text) - self.assertIn("Finance professionals can examine", item.after_text) - self.assertIn("target_audience_used_for_contextual_sentence", item.reasons) + self.assertNotIn("finance professionals", item.after_text.casefold()) + self.assertIn("neutral_audience_wording_used", item.reasons) self.assertEqual(item.preservation_percent, 100.0) def test_segments_reconstruct_rewrite_and_contain_one_safe_link(self): @@ -331,9 +329,10 @@ def test_rewrite_normalizes_space_before_terminal_punctuation(self): ) item = rank_placements(source, target, "ai in finance course", "https://t.com", top_n=1)[0] self.assertNotIn(" ,", item.after_text) - self.assertIn("operations, a practical application", item.after_text) + self.assertIn("ai in finance course", item.after_text) + self.assertNotIn("a practical application", item.after_text.casefold()) - def test_learning_rewrite_uses_explicit_student_audience(self): + def test_learning_rewrite_does_not_inject_target_audience_into_source(self): source = parse_page( """

Financial agents support continuous accounting and transaction monitoring across connected business systems, regional teams, and carefully governed reporting workflows.

""", requested_url="https://s.com", final_url="https://s.com", status_code=200, @@ -344,27 +343,31 @@ def test_learning_rewrite_uses_explicit_student_audience(self): ) item = rank_placements(source, target, "ai in finance course", "https://t.com/course", top_n=1)[0] - self.assertIn("a practical application students can examine more deeply", item.after_text) + self.assertNotIn("students", item.after_text.casefold()) self.assertNotIn("professionals", item.after_text.casefold()) - self.assertIn("target_audience_used_for_contextual_sentence", item.reasons) + self.assertIn("neutral_audience_wording_used", item.reasons) - def test_learning_rewrite_uses_other_explicit_target_audiences(self): + def test_learning_rewrite_uses_audience_only_when_source_also_addresses_it(self): source = parse_page( """

Financial agents support continuous accounting and transaction monitoring across connected business systems, regional teams, and carefully governed reporting workflows.

""", requested_url="https://s.com", final_url="https://s.com", status_code=200, ) - for audience, expected in ( - ("Developers", "developers can examine"), - ("Marketers", "marketers can examine"), - ("Executives", "executives can examine"), + for audience in ( + "Developers", + "Marketers", + "Executives", ): with self.subTest(audience=audience): + source = parse_page( + f"""

{audience} can use financial agents for continuous accounting, transaction monitoring, connected business systems, and carefully governed reporting workflows.

""", + requested_url="https://s.com", final_url="https://s.com", status_code=200, + ) target = parse_page( f"""AI in Finance Course for {audience}

Finance AI Applications

Study financial agents, continuous accounting, and transaction monitoring.

""", requested_url="https://t.com/course", final_url="https://t.com/course", status_code=200, ) item = rank_placements(source, target, "ai in finance course", "https://t.com/course", top_n=1)[0] - self.assertIn(expected, item.after_text) + self.assertIn(f"{audience.casefold()} can explore further", item.after_text.casefold()) self.assertIn("target_audience_used_for_contextual_sentence", item.reasons) def test_learning_rewrite_uses_neutral_wording_without_explicit_audience(self): @@ -378,7 +381,8 @@ def test_learning_rewrite_uses_neutral_wording_without_explicit_audience(self): ) item = rank_placements(source, target, "ai in finance course", "https://t.com/course", top_n=1)[0] - self.assertIn("a practical application that can be examined more deeply", item.after_text) + self.assertIn("ai in finance course", item.after_text) + self.assertNotIn("a practical application", item.after_text.casefold()) self.assertNotIn("professionals", item.after_text.casefold()) self.assertIn("neutral_audience_wording_used", item.reasons) @@ -393,7 +397,8 @@ def test_passing_audience_mention_does_not_define_target_audience(self): ) item = rank_placements(source, target, "ai in finance course", "https://t.com/course", top_n=1)[0] - self.assertIn("a practical application that can be examined more deeply", item.after_text) + self.assertIn("ai in finance course", item.after_text) + self.assertNotIn("a practical application", item.after_text.casefold()) self.assertNotIn("professionals can", item.after_text.casefold()) self.assertIn("neutral_audience_wording_used", item.reasons) @@ -423,7 +428,7 @@ def test_service_target_uses_implementation_wording_not_course_audience(self): ) item = rank_placements(source, target, "workflow automation platform", "https://t.com/products/workflow-automation", top_n=1)[0] - self.assertIn("an approach that can be explored further", item.after_text) + self.assertIn("implementation guidance for similar approaches", item.after_text) self.assertNotIn("professionals", item.after_text.casefold()) self.assertNotIn("course", item.after_text.casefold()) @@ -453,9 +458,43 @@ def test_course_url_supports_ambiguous_on_page_copy_without_guessing_audience(se ) item = rank_placements(source, target, "finance AI education", "https://t.com/education/course/finance-ai", top_n=1)[0] - self.assertIn("a practical application that can be examined more deeply", item.after_text) + self.assertIn("finance AI education", item.after_text) + self.assertNotIn("a practical application", item.after_text.casefold()) self.assertNotIn("professionals", item.after_text.casefold()) + def test_ranked_learning_suggestions_are_contextual_and_not_boilerplate(self): + source = parse_page( + """
+

AI agents need more than access to a large company database. The API should make business information easy to retrieve, interpret, and use within automated workflows, while providing enough flexibility for different research, monitoring, and decision-making tasks.

+

A company data API is an interface that lets AI agents access structured business information such as firmographics, workforce data, funding, technologies, and company growth signals. Agents can use this data for research, enrichment, monitoring, scoring, and automated decision-making.

+

Choosing the right company data API for an AI agent requires more than comparing database size or the number of available fields. The API should provide reliable data in a format that is easy for an agent to retrieve, interpret, and use within automated workflows. Below, you can find five actionable steps for selecting a company data API for AI agents.

+
""", + requested_url="https://source.example/company-data-apis", + final_url="https://source.example/company-data-apis", + status_code=200, + ) + target = parse_page( + """Agentic AI Course for Working Professionals
+

Certificate in Agentic AI

+

Learn to build autonomous AI agents using structured data, external tools, memory, orchestration, and automated workflows.

+
""", + requested_url="https://target.example/agentic-ai-course", + final_url="https://target.example/agentic-ai-course", + status_code=200, + ) + + items = rank_placements(source, target, "AI agent course", target.final_url, top_n=3) + self.assertEqual(len(items), 3) + generated_sentences = [item.after_text[len(item.before):].strip() for item in items] + self.assertEqual(len(set(generated_sentences)), 3) + for item in items: + lower = item.after_text.casefold() + self.assertNotIn("a practical application", lower) + self.assertNotIn("working professionals", lower) + self.assertNotIn("course resource", lower) + self.assertTrue(item.review_required) + self.assertEqual("".join(segment.text for segment in item.after_segments), item.after_text) + if __name__ == "__main__":