mirror of
https://github.com/gbessoni/seobuild-onpage.git
synced 2026-06-23 11:58:37 +02:00
Adds Massive (render.joinmassive.com) as the primary competitor content
parser when MASSIVE_API_TOKEN is configured. Returns clean rendered
markdown including JS-loaded content, which DataForSEO's
content_parsing/live endpoint has always missed.
Architecture:
- scripts/lib/massive.py wraps the /browser endpoint and outputs the
same shape as DataForSEOClient._extract_content (title, word_count,
headings, plain_text_size) so it's a drop-in replacement.
- scripts/research.py branches on Massive availability per URL. If
Massive errors or returns empty for any single URL, that URL falls
back to DataForSEO -- a partial Massive outage cannot break a run.
- env.py and .env.example pick up MASSIVE_API_TOKEN. New
creds["has_massive"] flag.
Scope intentionally narrow:
- SERP organic, PAA, and keyword data continue to come from DataForSEO.
- Massive's /search endpoint as of v1.9.0 only returns "also-searched"
query suggestions, not organic results. Tested directly to confirm
before scoping the integration.
Observability:
- research output now carries a content_parsers field summarizing
which parser handled each URL (e.g. {"massive": 4,
"dataforseo-fallback": 1}).
- Per-URL stderr log lines tag the parser: "Parsing content
(1/5, via massive)" or "via dataforseo".
Tests:
- 8 new unit tests in tests/test_massive.py covering the markdown
parser, the shape contract with DataForSEOClient, and client
construction.
- All 5 test files pass.
Live smoke-tested against airport parking JFK:
- With token: content_parsers = {"massive": 5}, real word counts on
all 5 competitors (Massive sees full body content; e.g. SpotHero
1132 words vs DataForSEO's previous 245).
- Without token: content_parsers = {"dataforseo": 4}, graceful
fallback to existing behavior.
No real token committed -- .env.example uses an empty placeholder.
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
119 lines
3.4 KiB
Python
119 lines
3.4 KiB
Python
"""Tests for the Massive Web Render client (v1.9.0)."""
|
|
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
sys.path.insert(0, str(Path(__file__).parent.parent / "scripts"))
|
|
|
|
from lib.massive import MassiveClient
|
|
|
|
|
|
# --- markdown parser ----------------------------------------------------------
|
|
|
|
def test_parse_markdown_extracts_headings_with_levels():
|
|
md = """# Page Title
|
|
|
|
Some intro paragraph.
|
|
|
|
## First Section
|
|
|
|
Content one.
|
|
|
|
### Subsection A
|
|
|
|
More content.
|
|
|
|
## Second Section
|
|
|
|
Content two.
|
|
"""
|
|
out = MassiveClient._parse_markdown(md)
|
|
assert out["title"] == "Page Title"
|
|
assert "H1: Page Title" in out["headings"]
|
|
assert "H2: First Section" in out["headings"]
|
|
assert "H3: Subsection A" in out["headings"]
|
|
assert "H2: Second Section" in out["headings"]
|
|
# Word count is rough but should be in a reasonable range
|
|
assert 10 <= out["word_count"] <= 30
|
|
assert out["plain_text_size"] == len(md)
|
|
|
|
|
|
def test_parse_markdown_no_h1_leaves_title_empty():
|
|
md = "## Only H2 here\n\nSome text body content."
|
|
out = MassiveClient._parse_markdown(md)
|
|
assert out["title"] == ""
|
|
assert "H2: Only H2 here" in out["headings"]
|
|
|
|
|
|
def test_parse_markdown_empty():
|
|
out = MassiveClient._parse_markdown("")
|
|
assert out["title"] == ""
|
|
assert out["headings"] == []
|
|
assert out["word_count"] == 0
|
|
assert out["plain_text_size"] == 0
|
|
|
|
|
|
def test_parse_markdown_ignores_hash_in_body():
|
|
"""A `#` mid-line is not a heading -- only line-start `#` counts."""
|
|
md = "Some text with # symbol in body\n\n## Real Heading\n"
|
|
out = MassiveClient._parse_markdown(md)
|
|
assert "H2: Real Heading" in out["headings"]
|
|
# The body `#` should not become an H1
|
|
assert out["title"] == ""
|
|
assert len(out["headings"]) == 1
|
|
|
|
|
|
def test_parse_markdown_all_six_levels():
|
|
md = """# H1
|
|
## H2
|
|
### H3
|
|
#### H4
|
|
##### H5
|
|
###### H6
|
|
"""
|
|
out = MassiveClient._parse_markdown(md)
|
|
for level in range(1, 7):
|
|
assert f"H{level}: H{level}" in out["headings"]
|
|
assert out["title"] == "H1"
|
|
|
|
|
|
# --- contract with DataForSEOClient ------------------------------------------
|
|
|
|
def test_output_shape_matches_dataforseo():
|
|
"""MassiveClient.content_parse() must return the same keys as
|
|
DataForSEOClient._extract_content() so it's a drop-in replacement
|
|
in research.py."""
|
|
out = MassiveClient._parse_markdown("# Title\n## Section\nbody words here")
|
|
required_keys = {"title", "word_count", "headings", "plain_text_size"}
|
|
assert required_keys.issubset(out.keys())
|
|
assert isinstance(out["title"], str)
|
|
assert isinstance(out["word_count"], int)
|
|
assert isinstance(out["headings"], list)
|
|
assert isinstance(out["plain_text_size"], int)
|
|
|
|
|
|
# --- client init --------------------------------------------------------------
|
|
|
|
def test_client_constructs_with_token():
|
|
c = MassiveClient("test-token")
|
|
assert c.api_token == "test-token"
|
|
assert c.default_country == "US"
|
|
assert c._headers()["Authorization"] == "Bearer test-token"
|
|
|
|
|
|
def test_client_custom_country():
|
|
c = MassiveClient("test-token", default_country="GB")
|
|
assert c.default_country == "GB"
|
|
|
|
|
|
if __name__ == "__main__":
|
|
test_parse_markdown_extracts_headings_with_levels()
|
|
test_parse_markdown_no_h1_leaves_title_empty()
|
|
test_parse_markdown_empty()
|
|
test_parse_markdown_ignores_hash_in_body()
|
|
test_parse_markdown_all_six_levels()
|
|
test_output_shape_matches_dataforseo()
|
|
test_client_constructs_with_token()
|
|
test_client_custom_country()
|
|
print("All tests passed.")
|