mirror of
https://github.com/gbessoni/seobuild-onpage.git
synced 2026-06-23 11:58:37 +02:00
scripts/research.py:
- New --differentiators CLI flag accepts a comma-separated list of brand
USPs (e.g. "women-owned, 24/7 service, no hidden fees"). Flows through
to research output as research.differentiators for the writing agent
to enforce verbatim in body content + AI Summary Nugget.
- New extract_missing_spokes() walks the top 3 ranking competitors'
internal-link anchors, filters out generic navigation (Home, Contact,
Privacy, FAQ, Login, social media, etc.) and broken-markdown image-
link leakage ([](href) nesting), and outputs a ranked
missing_spokes list. The list is the client's build-order priority
for filling the topical-silo gap.
- Compact + brief outputs both surface differentiators and missing_spokes.
scripts/lib/massive.py:
- _parse_markdown now returns a links list ([{text, url}]) from standard
[text](url) syntax. Skips image links, hash anchors, mailto/tel/javascript.
scripts/lib/dataforseo.py:
- New _extract_links static method walks page_content.main_topic and
secondary_topic, pulling anchor_text + url from primary_content[].urls.
- content_parse output now includes links field matching MassiveClient.
SKILL.md (v1.9.0 -> v1.9.1):
- Execution Protocol step 2 brief template gains Brand Differentiators /
USPs field. New paragraph instructs the agent to STOP and ASK the user
for differentiators if not provided up front.
- Section 12 Hub & Spoke Internal Linking gets a new Missing Spoke
Detection subsection requiring every generated page to append a
"## Recommended Spoke Pages" block built from missing_spokes data.
- Section 14 checklist expands 48 -> 51 points:
#49 Decision Fit (heading structure maps to buyer stage)
#50 Brand Identity (differentiators verbatim in chunks + nugget)
#51 Topical Silo (Recommended Spoke Pages block appended)
Passing threshold raised to 42/51.
references/quality-checklist.md: new v1.9.1 section detailing the three
new checks. Top-section reference updated to 51-point.
README.md, CHANGELOG.md, CLAUDE.md: version bumped, release-notes block
added, capability list updated. Historical version blocks restored to
their version-of-the-time checklist sizes (28, 34, 38, 41, 45, 48)
after over-greedy replace_all in prior commits.
Tests:
- 16 new tests in tests/test_research_v191.py covering --differentiators
parsing, domain normalization, generic-anchor filtering (including
nested-image-link leakage regression test), missing-spokes extraction
(same-domain filter, top-N respect, empty-input safety), markdown
link parsing in MassiveClient, and topic-tree link extraction in
DataForSEOClient.
- All 6 test files green.
Live smoke-tested against airport parking JFK with both flags:
- differentiators populated in compact output
- missing_spokes returned 12 semantic anchors after filtering
(SpotHero for Business, Reserve your spot, Parking details by lot,
EV charging stations, Learn about the JFK AirTrain, etc.)
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
133 lines
4.8 KiB
Python
133 lines
4.8 KiB
Python
"""
|
|
Massive Web Render API client for seo-agi.
|
|
|
|
Massive (render.joinmassive.com) is used for COMPETITOR CONTENT PARSING --
|
|
fetching a URL and getting back clean rendered markdown that includes
|
|
JS-loaded content, which DataForSEO's content_parsing/live endpoint misses.
|
|
|
|
Scope intentionally narrow:
|
|
- Massive /browser endpoint: URL -> markdown (used here)
|
|
- Massive /search endpoint: NOT used. As of v1.9.0 it returns only
|
|
'also-searched' query suggestions, not organic SERP results. SERP
|
|
data continues to come from DataForSEO.
|
|
|
|
The client outputs the same shape as DataForSEOClient._extract_content
|
|
(`title`, `word_count`, `headings`, `plain_text_size`) so it is a
|
|
drop-in replacement for the content_parse() step in research.py.
|
|
"""
|
|
|
|
import json
|
|
import re
|
|
import urllib.parse
|
|
import urllib.request
|
|
import urllib.error
|
|
from typing import Optional
|
|
|
|
|
|
class MassiveClient:
|
|
"""Client for Massive Web Render API (render.joinmassive.com)."""
|
|
|
|
BASE_URL = "https://render.joinmassive.com"
|
|
|
|
def __init__(self, api_token: str, default_country: str = "US"):
|
|
self.api_token = api_token
|
|
self.default_country = default_country
|
|
|
|
def _headers(self) -> dict:
|
|
return {"Authorization": f"Bearer {self.api_token}"}
|
|
|
|
def _get(self, path: str, params: dict, timeout: int = 60) -> str:
|
|
"""Make a GET request and return the raw response body as text."""
|
|
qs = urllib.parse.urlencode(params)
|
|
url = f"{self.BASE_URL}{path}?{qs}"
|
|
req = urllib.request.Request(url, headers=self._headers())
|
|
try:
|
|
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
|
return resp.read().decode("utf-8", errors="ignore")
|
|
except urllib.error.HTTPError as e:
|
|
body = e.read().decode("utf-8", errors="ignore") if e.fp else ""
|
|
raise RuntimeError(
|
|
f"Massive API error {e.code}: {body[:300]}"
|
|
) from e
|
|
except urllib.error.URLError as e:
|
|
raise RuntimeError(f"Massive connection error: {e.reason}") from e
|
|
|
|
def content_parse(
|
|
self, url: str, country: Optional[str] = None
|
|
) -> Optional[dict]:
|
|
"""Fetch a URL via Massive's renderer and extract content structure.
|
|
|
|
Returns the same shape as DataForSEOClient.content_parse():
|
|
{ "title": str, "word_count": int, "headings": [str],
|
|
"plain_text_size": int }
|
|
where each heading is formatted as "H{level}: {text}".
|
|
|
|
Returns None if the fetch returns empty content.
|
|
"""
|
|
try:
|
|
markdown = self._get(
|
|
"/browser",
|
|
{
|
|
"url": url,
|
|
"format": "markdown",
|
|
"country": country or self.default_country,
|
|
},
|
|
)
|
|
except RuntimeError:
|
|
return None
|
|
|
|
if not markdown or not markdown.strip():
|
|
return None
|
|
|
|
return self._parse_markdown(markdown)
|
|
|
|
@staticmethod
|
|
def _parse_markdown(md: str) -> dict:
|
|
"""Parse rendered markdown into the seo-agi content shape.
|
|
|
|
Headings: lines starting with `#`, `##`, ..., `######` -> H1..H6.
|
|
Title: first H1 found, otherwise empty.
|
|
Word count: rough split of body text after stripping markdown
|
|
syntax (consistent with how dataforseo.py counts).
|
|
Links: markdown `[text](url)` extraction for spoke detection.
|
|
"""
|
|
headings: list[str] = []
|
|
title = ""
|
|
for line in md.splitlines():
|
|
m = re.match(r"^(#{1,6})\s+(.+?)\s*$", line)
|
|
if not m:
|
|
continue
|
|
level = len(m.group(1))
|
|
text = m.group(2).strip()
|
|
if not text:
|
|
continue
|
|
if level == 1 and not title:
|
|
title = text
|
|
headings.append(f"H{level}: {text}")
|
|
|
|
# Word count: strip markdown punctuation, split on whitespace.
|
|
text = re.sub(r"[`*_#>\[\]()!|-]+", " ", md)
|
|
word_count = len([w for w in text.split() if w.strip()])
|
|
|
|
# Links (v1.9.1): standard markdown [text](url). Skip image links
|
|
#  -- those don't carry semantic anchor signal.
|
|
links: list[dict] = []
|
|
link_re = re.compile(r"(?<!!)\[([^\]]+)\]\(([^)\s]+)(?:\s+\"[^\"]*\")?\)")
|
|
for m in link_re.finditer(md):
|
|
anchor = m.group(1).strip()
|
|
url = m.group(2).strip()
|
|
if not anchor or not url:
|
|
continue
|
|
# Skip pure-hash/javascript/empty anchors -- no spoke value
|
|
if url.startswith(("#", "javascript:", "mailto:", "tel:")):
|
|
continue
|
|
links.append({"text": anchor, "url": url})
|
|
|
|
return {
|
|
"title": title,
|
|
"word_count": word_count,
|
|
"headings": headings,
|
|
"plain_text_size": len(md),
|
|
"links": links,
|
|
}
|