mirror of
https://github.com/gbessoni/seobuild-onpage.git
synced 2026-06-23 11:58:37 +02:00
seo-agi: The GEO framework skill for Claude Code + OpenClaw
GEO framework that writes pages ranking on Google AND getting cited by LLMs. 500-token chunk architecture, Reddit Test quality gates, verification tags, Not For You blocks, information gain enforcement. Data layer: DataForSEO, GSC, Ahrefs MCP, SEMRush MCP. 21 files, all tests passing. Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,96 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
SEO-AGI Google Search Console data puller.
|
||||
Retrieves query performance data and detects cannibalization.
|
||||
|
||||
Usage:
|
||||
python3 gsc_pull.py "<site_url>" [options]
|
||||
|
||||
Options:
|
||||
--keyword=KEYWORD Filter queries containing this keyword
|
||||
--days=N Lookback period (default: 90)
|
||||
--min-impressions=N Minimum impressions threshold (default: 10)
|
||||
--output=FORMAT Output: json|compact (default: compact)
|
||||
--cannibalization Run cannibalization detection for the keyword
|
||||
"""
|
||||
|
||||
import sys
|
||||
import os
|
||||
import json
|
||||
import argparse
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
from lib.env import get_credentials
|
||||
from lib.gsc_client import GSCClient
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="SEO-AGI GSC Pull")
|
||||
parser.add_argument("site_url", help="GSC site URL (e.g., https://example.com)")
|
||||
parser.add_argument("--keyword", default=None, help="Keyword filter")
|
||||
parser.add_argument("--days", type=int, default=90, help="Lookback days")
|
||||
parser.add_argument("--min-impressions", type=int, default=10, help="Min impressions")
|
||||
parser.add_argument("--output", choices=["json", "compact"], default="compact")
|
||||
parser.add_argument("--cannibalization", action="store_true", help="Detect cannibalization")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def format_compact(data: list[dict], mode: str = "performance") -> str:
|
||||
lines = []
|
||||
|
||||
if mode == "cannibalization":
|
||||
lines.append("# Cannibalization Report")
|
||||
for item in data:
|
||||
lines.append(f"\nQuery: {item['query']} ({item['page_count']} pages, {item['total_impressions']} impressions)")
|
||||
for page in item["pages"]:
|
||||
lines.append(f" pos {page['position']}: {page['page']} ({page['clicks']} clicks, {page['ctr']}% CTR)")
|
||||
else:
|
||||
lines.append("# Query Performance")
|
||||
lines.append(f"{'Query':<40} {'Clicks':>7} {'Impr':>7} {'CTR':>7} {'Pos':>5} Page")
|
||||
lines.append("-" * 110)
|
||||
for row in data[:50]:
|
||||
lines.append(
|
||||
f"{row['query'][:39]:<40} {row['clicks']:>7} {row['impressions']:>7} "
|
||||
f"{row['ctr']:>6.1f}% {row['position']:>5.1f} {row['page'][:50]}"
|
||||
)
|
||||
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def main():
|
||||
args = parse_args()
|
||||
creds = get_credentials()
|
||||
|
||||
if not creds["has_gsc"]:
|
||||
print("ERROR: Google Search Console credentials not found.", file=sys.stderr)
|
||||
print("Add GSC_SERVICE_ACCOUNT_PATH to ~/.config/seo-agi/.env", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
client = GSCClient(credentials_path=creds["gsc_service_account_path"])
|
||||
|
||||
if args.cannibalization and args.keyword:
|
||||
data = client.detect_cannibalization(
|
||||
site_url=args.site_url,
|
||||
keyword=args.keyword,
|
||||
days=args.days,
|
||||
)
|
||||
if args.output == "json":
|
||||
print(json.dumps(data, indent=2))
|
||||
else:
|
||||
print(format_compact(data, mode="cannibalization"))
|
||||
else:
|
||||
data = client.query_performance(
|
||||
site_url=args.site_url,
|
||||
keyword=args.keyword,
|
||||
days=args.days,
|
||||
min_impressions=args.min_impressions,
|
||||
)
|
||||
if args.output == "json":
|
||||
print(json.dumps(data, indent=2))
|
||||
else:
|
||||
print(format_compact(data))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1 @@
|
||||
# seo-agi scripts library
|
||||
@@ -0,0 +1,244 @@
|
||||
"""
|
||||
DataForSEO API client for SEO-AGI.
|
||||
Handles SERP results, keyword data, People Also Ask, and content parsing.
|
||||
"""
|
||||
|
||||
import json
|
||||
import base64
|
||||
import urllib.request
|
||||
import urllib.error
|
||||
from typing import Optional
|
||||
|
||||
|
||||
class DataForSEOClient:
|
||||
"""Client for DataForSEO REST API v3."""
|
||||
|
||||
BASE_URL = "https://api.dataforseo.com/v3"
|
||||
|
||||
def __init__(self, login: str, password: str):
|
||||
self.login = login
|
||||
self.password = password
|
||||
self._auth_header = self._make_auth_header(login, password)
|
||||
|
||||
@staticmethod
|
||||
def _make_auth_header(login: str, password: str) -> str:
|
||||
token = base64.b64encode(f"{login}:{password}".encode()).decode()
|
||||
return f"Basic {token}"
|
||||
|
||||
def _request(self, endpoint: str, payload: list[dict]) -> dict:
|
||||
"""Make a POST request to DataForSEO API."""
|
||||
url = f"{self.BASE_URL}{endpoint}"
|
||||
data = json.dumps(payload).encode("utf-8")
|
||||
|
||||
req = urllib.request.Request(
|
||||
url,
|
||||
data=data,
|
||||
headers={
|
||||
"Authorization": self._auth_header,
|
||||
"Content-Type": "application/json",
|
||||
},
|
||||
method="POST",
|
||||
)
|
||||
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=30) as resp:
|
||||
return json.loads(resp.read().decode())
|
||||
except urllib.error.HTTPError as e:
|
||||
body = e.read().decode() if e.fp else ""
|
||||
raise RuntimeError(
|
||||
f"DataForSEO API error {e.code}: {body}"
|
||||
) from e
|
||||
except urllib.error.URLError as e:
|
||||
raise RuntimeError(
|
||||
f"DataForSEO connection error: {e.reason}"
|
||||
) from e
|
||||
|
||||
def serp_live(
|
||||
self,
|
||||
keyword: str,
|
||||
location_code: int = 2840,
|
||||
language_code: str = "en",
|
||||
depth: int = 10,
|
||||
) -> dict:
|
||||
"""
|
||||
Get live SERP results for a keyword.
|
||||
Returns organic results with position, URL, title, description.
|
||||
"""
|
||||
payload = [
|
||||
{
|
||||
"keyword": keyword,
|
||||
"location_code": location_code,
|
||||
"language_code": language_code,
|
||||
"depth": depth,
|
||||
"se_type": "organic",
|
||||
}
|
||||
]
|
||||
result = self._request(
|
||||
"/serp/google/organic/live/advanced", payload
|
||||
)
|
||||
return self._extract_serp(result)
|
||||
|
||||
def related_keywords(
|
||||
self,
|
||||
keyword: str,
|
||||
location_code: int = 2840,
|
||||
language_code: str = "en",
|
||||
limit: int = 30,
|
||||
) -> list[dict]:
|
||||
"""Get related keywords with search volume and difficulty."""
|
||||
payload = [
|
||||
{
|
||||
"keyword": keyword,
|
||||
"location_code": location_code,
|
||||
"language_code": language_code,
|
||||
"limit": limit,
|
||||
}
|
||||
]
|
||||
result = self._request(
|
||||
"/dataforseo_labs/google/related_keywords/live", payload
|
||||
)
|
||||
return self._extract_keywords(result)
|
||||
|
||||
def keyword_suggestions(
|
||||
self,
|
||||
keyword: str,
|
||||
location_code: int = 2840,
|
||||
language_code: str = "en",
|
||||
limit: int = 30,
|
||||
) -> list[dict]:
|
||||
"""Get keyword suggestions (broader ideation)."""
|
||||
payload = [
|
||||
{
|
||||
"keyword": keyword,
|
||||
"location_code": location_code,
|
||||
"language_code": language_code,
|
||||
"limit": limit,
|
||||
}
|
||||
]
|
||||
result = self._request(
|
||||
"/dataforseo_labs/google/keyword_suggestions/live", payload
|
||||
)
|
||||
return self._extract_keywords(result)
|
||||
|
||||
def content_parse(self, url: str) -> Optional[dict]:
|
||||
"""Parse content from a URL (headings, word count, structure)."""
|
||||
payload = [{"url": url}]
|
||||
try:
|
||||
result = self._request(
|
||||
"/on_page/content_parsing/live", payload
|
||||
)
|
||||
return self._extract_content(result)
|
||||
except RuntimeError:
|
||||
return None
|
||||
|
||||
def _extract_serp(self, raw: dict) -> dict:
|
||||
"""Extract clean SERP data from API response."""
|
||||
tasks = raw.get("tasks", [])
|
||||
if not tasks:
|
||||
return {"organic": [], "paa": [], "featured_snippet": None}
|
||||
|
||||
result = tasks[0].get("result", [])
|
||||
if not result:
|
||||
return {"organic": [], "paa": [], "featured_snippet": None}
|
||||
|
||||
items = result[0].get("items", [])
|
||||
|
||||
organic = []
|
||||
paa_questions = []
|
||||
featured_snippet = None
|
||||
|
||||
for item in items:
|
||||
item_type = item.get("type", "")
|
||||
|
||||
if item_type == "organic":
|
||||
organic.append(
|
||||
{
|
||||
"position": item.get("rank_absolute", 0),
|
||||
"url": item.get("url", ""),
|
||||
"domain": item.get("domain", ""),
|
||||
"title": item.get("title", ""),
|
||||
"description": item.get("description", ""),
|
||||
}
|
||||
)
|
||||
|
||||
elif item_type == "people_also_ask":
|
||||
for paa_item in item.get("items", []):
|
||||
q = paa_item.get("title", "")
|
||||
if q:
|
||||
paa_questions.append(q)
|
||||
|
||||
elif item_type == "featured_snippet":
|
||||
featured_snippet = {
|
||||
"url": item.get("url", ""),
|
||||
"title": item.get("title", ""),
|
||||
"description": item.get("description", ""),
|
||||
}
|
||||
|
||||
return {
|
||||
"organic": organic,
|
||||
"paa": paa_questions,
|
||||
"featured_snippet": featured_snippet,
|
||||
"total_results": result[0].get("se_results_count", 0),
|
||||
}
|
||||
|
||||
def _extract_keywords(self, raw: dict) -> list[dict]:
|
||||
"""Extract keyword data from labs API response."""
|
||||
tasks = raw.get("tasks", [])
|
||||
if not tasks:
|
||||
return []
|
||||
|
||||
result = tasks[0].get("result", [])
|
||||
if not result:
|
||||
return []
|
||||
|
||||
items = result[0].get("items", [])
|
||||
keywords = []
|
||||
|
||||
for item in items:
|
||||
kw_data = item.get("keyword_data", item)
|
||||
keyword_info = kw_data.get("keyword_info", {})
|
||||
keywords.append(
|
||||
{
|
||||
"keyword": kw_data.get("keyword", ""),
|
||||
"volume": keyword_info.get("search_volume", 0),
|
||||
"cpc": keyword_info.get("cpc", 0),
|
||||
"competition": keyword_info.get("competition", 0),
|
||||
"difficulty": kw_data.get(
|
||||
"keyword_properties", {}
|
||||
).get("keyword_difficulty", 0),
|
||||
}
|
||||
)
|
||||
|
||||
return sorted(keywords, key=lambda x: x["volume"], reverse=True)
|
||||
|
||||
def _extract_content(self, raw: dict) -> Optional[dict]:
|
||||
"""Extract content structure from on-page parsing."""
|
||||
tasks = raw.get("tasks", [])
|
||||
if not tasks:
|
||||
return None
|
||||
|
||||
result = tasks[0].get("result", [])
|
||||
if not result:
|
||||
return None
|
||||
|
||||
items = result[0].get("items", [])
|
||||
if not items:
|
||||
return None
|
||||
|
||||
page = items[0].get("page_content", {})
|
||||
|
||||
return {
|
||||
"title": page.get("header", {}).get("title", ""),
|
||||
"word_count": page.get("plain_text_word_count", 0),
|
||||
"headings": self._extract_headings(page),
|
||||
"plain_text_size": page.get("plain_text_size", 0),
|
||||
}
|
||||
|
||||
@staticmethod
|
||||
def _extract_headings(page_content: dict) -> list[str]:
|
||||
"""Pull heading tags from parsed content."""
|
||||
headings = []
|
||||
for level in ["h1", "h2", "h3"]:
|
||||
for heading in page_content.get(level, []):
|
||||
headings.append(f"{level.upper()}: {heading}")
|
||||
return headings
|
||||
@@ -0,0 +1,147 @@
|
||||
"""
|
||||
seo-agi environment and configuration loader.
|
||||
Reads API keys from ~/.config/seo-agi/.env or os.environ.
|
||||
Resolves paths relative to the skill installation directory.
|
||||
"""
|
||||
|
||||
import os
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
# Skill root is two levels up from this file (scripts/lib/env.py -> .)
|
||||
SKILL_DIR = Path(__file__).resolve().parent.parent.parent
|
||||
|
||||
OUTPUT_DIR = Path.home() / "Documents" / "SEO-AGI"
|
||||
DATA_DIR = Path.home() / ".local" / "share" / "seo-agi"
|
||||
CONFIG_DIR = Path.home() / ".config" / "seo-agi"
|
||||
ENV_FILE = CONFIG_DIR / ".env"
|
||||
|
||||
DEFAULT_CONFIG = {
|
||||
"default_location": 2840,
|
||||
"default_language": "en",
|
||||
"default_site": "",
|
||||
"serp_depth": 10,
|
||||
"save_research": True,
|
||||
"output_dir": str(OUTPUT_DIR),
|
||||
}
|
||||
|
||||
|
||||
def load_env() -> dict:
|
||||
"""
|
||||
Load environment variables.
|
||||
Reads from ~/.config/seo-agi/.env first, then overlays os.environ.
|
||||
"""
|
||||
env = {}
|
||||
|
||||
# First, try the config file
|
||||
if ENV_FILE.exists():
|
||||
with open(ENV_FILE, "r") as f:
|
||||
for line in f:
|
||||
line = line.strip()
|
||||
if not line or line.startswith("#"):
|
||||
continue
|
||||
if "=" in line:
|
||||
key, _, value = line.partition("=")
|
||||
key = key.strip()
|
||||
value = value.strip().strip('"').strip("'")
|
||||
if value:
|
||||
env[key] = value
|
||||
|
||||
# Then overlay with os.environ
|
||||
for key in [
|
||||
"DATAFORSEO_LOGIN", "DATAFORSEO_PASSWORD",
|
||||
"GSC_SERVICE_ACCOUNT_PATH", "GSC_CLIENT_ID",
|
||||
"GSC_CLIENT_SECRET", "GSC_REFRESH_TOKEN",
|
||||
"AHREFS_API_KEY", "SEMRUSH_API_KEY",
|
||||
]:
|
||||
val = os.environ.get(key)
|
||||
if val:
|
||||
env[key] = val
|
||||
|
||||
return env
|
||||
|
||||
|
||||
def load_config() -> dict:
|
||||
"""Load user config, merged with defaults."""
|
||||
config = DEFAULT_CONFIG.copy()
|
||||
config_file = CONFIG_DIR / "config.json"
|
||||
if config_file.exists():
|
||||
try:
|
||||
with open(config_file, "r") as f:
|
||||
user_config = json.load(f)
|
||||
config.update(user_config)
|
||||
except (json.JSONDecodeError, IOError):
|
||||
pass
|
||||
return config
|
||||
|
||||
|
||||
def get_credentials() -> dict:
|
||||
"""Get API credentials with availability flags."""
|
||||
env = load_env()
|
||||
|
||||
creds = {
|
||||
"dataforseo_login": env.get("DATAFORSEO_LOGIN", ""),
|
||||
"dataforseo_password": env.get("DATAFORSEO_PASSWORD", ""),
|
||||
"gsc_service_account_path": env.get("GSC_SERVICE_ACCOUNT_PATH", ""),
|
||||
"gsc_client_id": env.get("GSC_CLIENT_ID", ""),
|
||||
"gsc_client_secret": env.get("GSC_CLIENT_SECRET", ""),
|
||||
"gsc_refresh_token": env.get("GSC_REFRESH_TOKEN", ""),
|
||||
"ahrefs_api_key": env.get("AHREFS_API_KEY", ""),
|
||||
"semrush_api_key": env.get("SEMRUSH_API_KEY", ""),
|
||||
}
|
||||
|
||||
creds["has_dataforseo"] = bool(
|
||||
creds["dataforseo_login"] and creds["dataforseo_password"]
|
||||
)
|
||||
creds["has_gsc"] = bool(
|
||||
creds["gsc_service_account_path"]
|
||||
or (creds["gsc_client_id"] and creds["gsc_client_secret"])
|
||||
)
|
||||
creds["has_ahrefs"] = bool(creds["ahrefs_api_key"])
|
||||
creds["has_semrush"] = bool(creds["semrush_api_key"])
|
||||
|
||||
return creds
|
||||
|
||||
|
||||
def ensure_dirs():
|
||||
"""Create output directories if they don't exist."""
|
||||
config = load_config()
|
||||
output_dir = Path(config["output_dir"]).expanduser()
|
||||
|
||||
for subdir in ["research", "briefs", "pages", "rewrites"]:
|
||||
(output_dir / subdir).mkdir(parents=True, exist_ok=True)
|
||||
|
||||
(DATA_DIR / "research").mkdir(parents=True, exist_ok=True)
|
||||
(DATA_DIR / "cache").mkdir(parents=True, exist_ok=True)
|
||||
|
||||
|
||||
def check_setup() -> dict:
|
||||
"""Check setup status and return a summary."""
|
||||
creds = get_credentials()
|
||||
config = load_config()
|
||||
|
||||
return {
|
||||
"runtime": "claude-code",
|
||||
"skill_dir": str(SKILL_DIR),
|
||||
"config_dir_exists": CONFIG_DIR.exists(),
|
||||
"env_file_exists": ENV_FILE.exists(),
|
||||
"has_dataforseo": creds["has_dataforseo"],
|
||||
"has_gsc": creds["has_gsc"],
|
||||
"has_ahrefs": creds["has_ahrefs"],
|
||||
"has_semrush": creds["has_semrush"],
|
||||
"default_location": config["default_location"],
|
||||
"default_language": config["default_language"],
|
||||
"mode": _determine_mode(creds),
|
||||
}
|
||||
|
||||
|
||||
def _determine_mode(creds: dict) -> str:
|
||||
"""Determine operational mode based on available credentials."""
|
||||
if creds["has_dataforseo"] and creds["has_gsc"]:
|
||||
return "full"
|
||||
elif creds["has_dataforseo"]:
|
||||
return "dataforseo-only"
|
||||
elif creds["has_gsc"]:
|
||||
return "gsc-only"
|
||||
else:
|
||||
return "fallback"
|
||||
@@ -0,0 +1,179 @@
|
||||
"""
|
||||
Google Search Console API client for SEO-AGI.
|
||||
Pulls query performance data, cannibalization detection, and indexing status.
|
||||
"""
|
||||
|
||||
import json
|
||||
from typing import Optional
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
class GSCClient:
|
||||
"""Client for Google Search Console API."""
|
||||
|
||||
def __init__(self, credentials_path: str = None, oauth_creds: dict = None):
|
||||
"""
|
||||
Initialize GSC client.
|
||||
|
||||
Args:
|
||||
credentials_path: Path to service account JSON file
|
||||
oauth_creds: Dict with client_id, client_secret, refresh_token
|
||||
"""
|
||||
self.credentials_path = credentials_path
|
||||
self.oauth_creds = oauth_creds
|
||||
self._service = None
|
||||
|
||||
def _get_service(self):
|
||||
"""Lazy-initialize the GSC API service."""
|
||||
if self._service is not None:
|
||||
return self._service
|
||||
|
||||
try:
|
||||
from google.oauth2 import service_account
|
||||
from googleapiclient.discovery import build
|
||||
except ImportError:
|
||||
raise RuntimeError(
|
||||
"GSC requires google-auth and google-api-python-client. "
|
||||
"Install with: pip install google-auth google-api-python-client"
|
||||
)
|
||||
|
||||
if self.credentials_path:
|
||||
creds = service_account.Credentials.from_service_account_file(
|
||||
self.credentials_path,
|
||||
scopes=["https://www.googleapis.com/auth/webmasters.readonly"],
|
||||
)
|
||||
else:
|
||||
raise RuntimeError(
|
||||
"OAuth2 flow not yet implemented. Use a service account."
|
||||
)
|
||||
|
||||
self._service = build("searchconsole", "v1", credentials=creds)
|
||||
return self._service
|
||||
|
||||
def query_performance(
|
||||
self,
|
||||
site_url: str,
|
||||
keyword: str = None,
|
||||
days: int = 90,
|
||||
min_impressions: int = 10,
|
||||
row_limit: int = 100,
|
||||
) -> list[dict]:
|
||||
"""
|
||||
Pull query performance data from GSC.
|
||||
|
||||
Args:
|
||||
site_url: The GSC property URL (e.g., "https://example.com")
|
||||
keyword: Optional keyword filter (partial match)
|
||||
days: Lookback period in days
|
||||
min_impressions: Minimum impressions threshold
|
||||
row_limit: Max rows to return
|
||||
|
||||
Returns:
|
||||
List of dicts with query, page, clicks, impressions, ctr, position
|
||||
"""
|
||||
from datetime import datetime, timedelta
|
||||
|
||||
service = self._get_service()
|
||||
|
||||
end_date = datetime.now().strftime("%Y-%m-%d")
|
||||
start_date = (datetime.now() - timedelta(days=days)).strftime("%Y-%m-%d")
|
||||
|
||||
request_body = {
|
||||
"startDate": start_date,
|
||||
"endDate": end_date,
|
||||
"dimensions": ["query", "page"],
|
||||
"rowLimit": row_limit,
|
||||
"dimensionFilterGroups": [],
|
||||
}
|
||||
|
||||
if keyword:
|
||||
request_body["dimensionFilterGroups"].append(
|
||||
{
|
||||
"filters": [
|
||||
{
|
||||
"dimension": "query",
|
||||
"operator": "contains",
|
||||
"expression": keyword,
|
||||
}
|
||||
]
|
||||
}
|
||||
)
|
||||
|
||||
response = (
|
||||
service.searchanalytics()
|
||||
.query(siteUrl=site_url, body=request_body)
|
||||
.execute()
|
||||
)
|
||||
|
||||
rows = response.get("rows", [])
|
||||
results = []
|
||||
|
||||
for row in rows:
|
||||
impressions = row.get("impressions", 0)
|
||||
if impressions < min_impressions:
|
||||
continue
|
||||
|
||||
results.append(
|
||||
{
|
||||
"query": row["keys"][0],
|
||||
"page": row["keys"][1],
|
||||
"clicks": row.get("clicks", 0),
|
||||
"impressions": impressions,
|
||||
"ctr": round(row.get("ctr", 0) * 100, 2),
|
||||
"position": round(row.get("position", 0), 1),
|
||||
}
|
||||
)
|
||||
|
||||
return sorted(results, key=lambda x: x["impressions"], reverse=True)
|
||||
|
||||
def detect_cannibalization(
|
||||
self,
|
||||
site_url: str,
|
||||
keyword: str,
|
||||
days: int = 90,
|
||||
) -> list[dict]:
|
||||
"""
|
||||
Detect keyword cannibalization: multiple pages ranking for same query.
|
||||
|
||||
Returns:
|
||||
List of queries where 2+ pages from the site appear,
|
||||
sorted by total impressions.
|
||||
"""
|
||||
results = self.query_performance(
|
||||
site_url=site_url,
|
||||
keyword=keyword,
|
||||
days=days,
|
||||
min_impressions=5,
|
||||
row_limit=500,
|
||||
)
|
||||
|
||||
# Group by query
|
||||
query_pages = {}
|
||||
for row in results:
|
||||
q = row["query"]
|
||||
if q not in query_pages:
|
||||
query_pages[q] = []
|
||||
query_pages[q].append(row)
|
||||
|
||||
# Find queries with multiple pages
|
||||
cannibalized = []
|
||||
for query, pages in query_pages.items():
|
||||
if len(pages) > 1:
|
||||
cannibalized.append(
|
||||
{
|
||||
"query": query,
|
||||
"page_count": len(pages),
|
||||
"pages": sorted(
|
||||
pages, key=lambda x: x["position"]
|
||||
),
|
||||
"total_impressions": sum(
|
||||
p["impressions"] for p in pages
|
||||
),
|
||||
}
|
||||
)
|
||||
|
||||
return sorted(
|
||||
cannibalized,
|
||||
key=lambda x: x["total_impressions"],
|
||||
reverse=True,
|
||||
)
|
||||
@@ -0,0 +1,261 @@
|
||||
"""
|
||||
SERP content analyzer for SEO-AGI.
|
||||
Analyzes competitor content structure and identifies content gaps.
|
||||
"""
|
||||
|
||||
import statistics
|
||||
from typing import Optional
|
||||
|
||||
|
||||
def analyze_serp(
|
||||
serp_data: dict,
|
||||
content_data: list[Optional[dict]],
|
||||
target_keyword: str,
|
||||
) -> dict:
|
||||
"""
|
||||
Analyze SERP results + parsed competitor content.
|
||||
|
||||
Args:
|
||||
serp_data: Output from DataForSEOClient.serp_live()
|
||||
content_data: List of DataForSEOClient.content_parse() results
|
||||
(one per organic result, may contain None for failed parses)
|
||||
target_keyword: The keyword being targeted
|
||||
|
||||
Returns:
|
||||
Analysis dict with intent, word count stats, topic coverage, gaps.
|
||||
"""
|
||||
organic = serp_data.get("organic", [])
|
||||
paa = serp_data.get("paa", [])
|
||||
|
||||
# Word count analysis
|
||||
word_counts = []
|
||||
all_headings = []
|
||||
all_topics = set()
|
||||
|
||||
for i, content in enumerate(content_data):
|
||||
if content is None:
|
||||
continue
|
||||
|
||||
wc = content.get("word_count", 0)
|
||||
if wc > 0:
|
||||
word_counts.append(wc)
|
||||
|
||||
headings = content.get("headings", [])
|
||||
all_headings.extend(headings)
|
||||
|
||||
# Extract topics from H2/H3 headings
|
||||
for h in headings:
|
||||
if h.startswith(("H2:", "H3:")):
|
||||
topic = h.split(":", 1)[1].strip().lower()
|
||||
all_topics.add(topic)
|
||||
|
||||
# Word count statistics
|
||||
wc_stats = {}
|
||||
if word_counts:
|
||||
wc_stats = {
|
||||
"min": min(word_counts),
|
||||
"max": max(word_counts),
|
||||
"median": int(statistics.median(word_counts)),
|
||||
"mean": int(statistics.mean(word_counts)),
|
||||
"recommended_min": int(statistics.median(word_counts) * 0.8),
|
||||
"recommended_max": int(statistics.median(word_counts) * 1.3),
|
||||
"count_analyzed": len(word_counts),
|
||||
}
|
||||
|
||||
# Intent detection
|
||||
intent = detect_intent(target_keyword, organic, serp_data)
|
||||
|
||||
# Topic frequency (which topics appear in multiple competitors)
|
||||
topic_freq = _count_topic_frequency(content_data)
|
||||
|
||||
# Heading patterns
|
||||
heading_patterns = _analyze_heading_patterns(content_data)
|
||||
|
||||
return {
|
||||
"keyword": target_keyword,
|
||||
"intent": intent,
|
||||
"word_count_stats": wc_stats,
|
||||
"paa_questions": paa,
|
||||
"topic_frequency": topic_freq,
|
||||
"heading_patterns": heading_patterns,
|
||||
"competitors_analyzed": len(
|
||||
[c for c in content_data if c is not None]
|
||||
),
|
||||
"total_organic_results": len(organic),
|
||||
"featured_snippet": serp_data.get("featured_snippet"),
|
||||
}
|
||||
|
||||
|
||||
def detect_intent(
|
||||
keyword: str, organic: list[dict], serp_data: dict
|
||||
) -> str:
|
||||
"""
|
||||
Detect search intent from keyword and SERP features.
|
||||
|
||||
Returns: informational, commercial, transactional, or navigational
|
||||
"""
|
||||
kw_lower = keyword.lower()
|
||||
|
||||
# Navigational signals
|
||||
nav_signals = [
|
||||
"login",
|
||||
"sign in",
|
||||
"website",
|
||||
"official",
|
||||
".com",
|
||||
".org",
|
||||
]
|
||||
if any(s in kw_lower for s in nav_signals):
|
||||
return "navigational"
|
||||
|
||||
# Transactional signals
|
||||
transactional_signals = [
|
||||
"buy",
|
||||
"purchase",
|
||||
"order",
|
||||
"download",
|
||||
"subscribe",
|
||||
"deal",
|
||||
"discount",
|
||||
"coupon",
|
||||
"price",
|
||||
"pricing",
|
||||
"cost",
|
||||
"cheap",
|
||||
"free trial",
|
||||
]
|
||||
if any(s in kw_lower for s in transactional_signals):
|
||||
return "transactional"
|
||||
|
||||
# Commercial investigation signals
|
||||
commercial_signals = [
|
||||
"best",
|
||||
"top",
|
||||
"review",
|
||||
"comparison",
|
||||
"vs",
|
||||
"versus",
|
||||
"alternative",
|
||||
"vs.",
|
||||
"compared to",
|
||||
"pros and cons",
|
||||
]
|
||||
if any(s in kw_lower for s in commercial_signals):
|
||||
return "commercial"
|
||||
|
||||
# Informational signals
|
||||
informational_signals = [
|
||||
"how to",
|
||||
"what is",
|
||||
"what are",
|
||||
"why",
|
||||
"guide",
|
||||
"tutorial",
|
||||
"learn",
|
||||
"example",
|
||||
"definition",
|
||||
"meaning",
|
||||
"explain",
|
||||
]
|
||||
if any(s in kw_lower for s in informational_signals):
|
||||
return "informational"
|
||||
|
||||
# Default: check SERP patterns
|
||||
if serp_data.get("featured_snippet"):
|
||||
return "informational"
|
||||
|
||||
# If titles contain pricing/comparison language
|
||||
titles = [r.get("title", "").lower() for r in organic[:5]]
|
||||
title_text = " ".join(titles)
|
||||
|
||||
if any(s in title_text for s in ["best", "top", "review", "vs"]):
|
||||
return "commercial"
|
||||
if any(
|
||||
s in title_text for s in ["how to", "guide", "what", "tutorial"]
|
||||
):
|
||||
return "informational"
|
||||
|
||||
return "commercial" # default for ambiguous
|
||||
|
||||
|
||||
def _count_topic_frequency(
|
||||
content_data: list[Optional[dict]],
|
||||
) -> list[dict]:
|
||||
"""Count how often topics (H2/H3 headings) appear across competitors."""
|
||||
topic_counts = {}
|
||||
|
||||
for content in content_data:
|
||||
if content is None:
|
||||
continue
|
||||
|
||||
seen_in_page = set()
|
||||
for heading in content.get("headings", []):
|
||||
if heading.startswith(("H2:", "H3:")):
|
||||
topic = heading.split(":", 1)[1].strip().lower()
|
||||
# Normalize common variations
|
||||
topic = _normalize_topic(topic)
|
||||
if topic and topic not in seen_in_page:
|
||||
seen_in_page.add(topic)
|
||||
topic_counts[topic] = topic_counts.get(topic, 0) + 1
|
||||
|
||||
# Sort by frequency
|
||||
sorted_topics = sorted(
|
||||
topic_counts.items(), key=lambda x: x[1], reverse=True
|
||||
)
|
||||
return [
|
||||
{"topic": t, "competitor_count": c} for t, c in sorted_topics[:30]
|
||||
]
|
||||
|
||||
|
||||
def _normalize_topic(topic: str) -> str:
|
||||
"""Basic topic normalization."""
|
||||
# Remove common filler words at start
|
||||
for prefix in [
|
||||
"the ",
|
||||
"a ",
|
||||
"an ",
|
||||
"our ",
|
||||
"your ",
|
||||
"my ",
|
||||
"about ",
|
||||
]:
|
||||
if topic.startswith(prefix):
|
||||
topic = topic[len(prefix) :]
|
||||
|
||||
# Strip trailing punctuation
|
||||
topic = topic.rstrip(".:!?")
|
||||
|
||||
return topic.strip()
|
||||
|
||||
|
||||
def _analyze_heading_patterns(
|
||||
content_data: list[Optional[dict]],
|
||||
) -> dict:
|
||||
"""Analyze heading patterns across competitors."""
|
||||
h2_counts = []
|
||||
h3_counts = []
|
||||
|
||||
for content in content_data:
|
||||
if content is None:
|
||||
continue
|
||||
|
||||
headings = content.get("headings", [])
|
||||
h2_count = sum(1 for h in headings if h.startswith("H2:"))
|
||||
h3_count = sum(1 for h in headings if h.startswith("H3:"))
|
||||
h2_counts.append(h2_count)
|
||||
h3_counts.append(h3_count)
|
||||
|
||||
return {
|
||||
"avg_h2_count": (
|
||||
round(statistics.mean(h2_counts), 1) if h2_counts else 0
|
||||
),
|
||||
"avg_h3_count": (
|
||||
round(statistics.mean(h3_counts), 1) if h3_counts else 0
|
||||
),
|
||||
"median_h2_count": (
|
||||
int(statistics.median(h2_counts)) if h2_counts else 0
|
||||
),
|
||||
"median_h3_count": (
|
||||
int(statistics.median(h3_counts)) if h3_counts else 0
|
||||
),
|
||||
}
|
||||
@@ -0,0 +1,334 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
seo-agi research orchestrator.
|
||||
Pulls SERP data, keyword data, PAA questions, and competitor content analysis
|
||||
into a single research.json file that feeds content generation.
|
||||
|
||||
Usage:
|
||||
python3 research.py "<keyword>" [options]
|
||||
|
||||
Options:
|
||||
--serp-depth=N Number of SERP results to analyze (default: 10)
|
||||
--include-paa Include People Also Ask extraction (default: true)
|
||||
--location=CODE DataForSEO location code (default: 2840 = US)
|
||||
--language=CODE Language code (default: en)
|
||||
--output=FORMAT Output: json|compact|brief (default: compact)
|
||||
--save-dir=PATH Save raw data (default: ~/.local/share/seo-agi/research/)
|
||||
--content-depth=N Number of top results to parse for content (default: 5)
|
||||
--mock Use fixture data instead of live API calls
|
||||
"""
|
||||
|
||||
import sys
|
||||
import os
|
||||
import json
|
||||
import argparse
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
|
||||
# Add parent dir to path for lib imports
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
from lib.env import load_env, load_config, get_credentials, ensure_dirs
|
||||
from lib.dataforseo import DataForSEOClient
|
||||
from lib.serp_analyze import analyze_serp
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="SEO-AGI Research")
|
||||
parser.add_argument("keyword", help="Target keyword or topic")
|
||||
parser.add_argument(
|
||||
"--serp-depth", type=int, default=10, help="SERP depth"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--content-depth",
|
||||
type=int,
|
||||
default=5,
|
||||
help="Number of competitors to parse content from",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--location", type=int, default=None, help="Location code"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--language", default=None, help="Language code"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--output",
|
||||
choices=["json", "compact", "brief"],
|
||||
default="compact",
|
||||
help="Output format",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--save-dir", default=None, help="Directory to save research data"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--mock",
|
||||
action="store_true",
|
||||
help="Use fixture data (no API calls)",
|
||||
)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def load_mock_data(keyword: str) -> dict:
|
||||
"""Load fixture data for testing without API calls."""
|
||||
fixtures_dir = (
|
||||
Path(__file__).parent.parent / "fixtures"
|
||||
)
|
||||
|
||||
serp_fixture = fixtures_dir / "serp_sample.json"
|
||||
keywords_fixture = fixtures_dir / "keywords_sample.json"
|
||||
|
||||
serp_data = {"organic": [], "paa": [], "featured_snippet": None}
|
||||
related_kw = []
|
||||
|
||||
if serp_fixture.exists():
|
||||
with open(serp_fixture) as f:
|
||||
serp_data = json.load(f)
|
||||
|
||||
if keywords_fixture.exists():
|
||||
with open(keywords_fixture) as f:
|
||||
related_kw = json.load(f)
|
||||
|
||||
return {
|
||||
"keyword": keyword,
|
||||
"timestamp": datetime.now(timezone.utc).isoformat(),
|
||||
"source": "mock",
|
||||
"serp": serp_data,
|
||||
"related_keywords": related_kw,
|
||||
"analysis": {
|
||||
"intent": "commercial",
|
||||
"word_count_stats": {
|
||||
"min": 800,
|
||||
"max": 3200,
|
||||
"median": 1800,
|
||||
"recommended_min": 1440,
|
||||
"recommended_max": 2340,
|
||||
},
|
||||
"paa_questions": serp_data.get("paa", []),
|
||||
"topic_frequency": [],
|
||||
"heading_patterns": {
|
||||
"avg_h2_count": 6,
|
||||
"avg_h3_count": 8,
|
||||
"median_h2_count": 5,
|
||||
"median_h3_count": 7,
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def run_research(args) -> dict:
|
||||
"""Execute the full research pipeline."""
|
||||
creds = get_credentials()
|
||||
config = load_config()
|
||||
|
||||
location = args.location or config["default_location"]
|
||||
language = args.language or config["default_language"]
|
||||
|
||||
if args.mock:
|
||||
return load_mock_data(args.keyword)
|
||||
|
||||
if not creds["has_dataforseo"]:
|
||||
print(
|
||||
"ERROR: DataForSEO credentials not found.",
|
||||
file=sys.stderr,
|
||||
)
|
||||
print(
|
||||
"Run: python3 scripts/setup.py",
|
||||
file=sys.stderr,
|
||||
)
|
||||
print(
|
||||
"Or add DATAFORSEO_LOGIN and DATAFORSEO_PASSWORD to "
|
||||
"~/.config/seo-agi/.env",
|
||||
file=sys.stderr,
|
||||
)
|
||||
sys.exit(1)
|
||||
|
||||
client = DataForSEOClient(
|
||||
creds["dataforseo_login"], creds["dataforseo_password"]
|
||||
)
|
||||
|
||||
# Step 1: SERP results
|
||||
print(f"Fetching SERP for: {args.keyword}", file=sys.stderr)
|
||||
serp_data = client.serp_live(
|
||||
args.keyword, location, language, args.serp_depth
|
||||
)
|
||||
|
||||
# Step 2: Related keywords
|
||||
print("Fetching related keywords...", file=sys.stderr)
|
||||
related_kw = client.related_keywords(args.keyword, location, language)
|
||||
|
||||
# Step 3: Parse competitor content (top N)
|
||||
content_data = []
|
||||
organic = serp_data.get("organic", [])
|
||||
parse_count = min(args.content_depth, len(organic))
|
||||
|
||||
for i in range(parse_count):
|
||||
url = organic[i].get("url", "")
|
||||
if url:
|
||||
print(
|
||||
f"Parsing content ({i+1}/{parse_count}): {url[:80]}...",
|
||||
file=sys.stderr,
|
||||
)
|
||||
content = client.content_parse(url)
|
||||
content_data.append(content)
|
||||
|
||||
# Merge content data back into organic result
|
||||
if content:
|
||||
organic[i]["word_count"] = content.get("word_count", 0)
|
||||
organic[i]["headings"] = content.get("headings", [])
|
||||
else:
|
||||
content_data.append(None)
|
||||
|
||||
# Step 4: Analyze
|
||||
print("Analyzing competitive landscape...", file=sys.stderr)
|
||||
analysis = analyze_serp(serp_data, content_data, args.keyword)
|
||||
|
||||
# Assemble research output
|
||||
research = {
|
||||
"keyword": args.keyword,
|
||||
"timestamp": datetime.now(timezone.utc).isoformat(),
|
||||
"location": location,
|
||||
"language": language,
|
||||
"source": "dataforseo",
|
||||
"serp": serp_data,
|
||||
"related_keywords": related_kw[:20],
|
||||
"analysis": analysis,
|
||||
}
|
||||
|
||||
return research
|
||||
|
||||
|
||||
def save_research(research: dict, save_dir: str = None):
|
||||
"""Save research data to disk."""
|
||||
ensure_dirs()
|
||||
config = load_config()
|
||||
|
||||
if save_dir:
|
||||
out_dir = Path(save_dir).expanduser()
|
||||
else:
|
||||
out_dir = Path.home() / ".local" / "share" / "seo-agi" / "research"
|
||||
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
# Filename from keyword + date
|
||||
slug = (
|
||||
research["keyword"]
|
||||
.lower()
|
||||
.replace(" ", "-")
|
||||
.replace("/", "-")[:50]
|
||||
)
|
||||
date_str = datetime.now().strftime("%Y%m%d")
|
||||
filename = f"{slug}-{date_str}.json"
|
||||
|
||||
filepath = out_dir / filename
|
||||
with open(filepath, "w") as f:
|
||||
json.dump(research, f, indent=2)
|
||||
|
||||
print(f"Research saved: {filepath}", file=sys.stderr)
|
||||
return str(filepath)
|
||||
|
||||
|
||||
def format_compact(research: dict) -> str:
|
||||
"""Format research as compact human-readable output."""
|
||||
lines = []
|
||||
kw = research["keyword"]
|
||||
analysis = research.get("analysis", {})
|
||||
serp = research.get("serp", {})
|
||||
organic = serp.get("organic", [])
|
||||
|
||||
lines.append(f"# Research: {kw}")
|
||||
lines.append(f"Intent: {analysis.get('intent', 'unknown')}")
|
||||
|
||||
# Word count
|
||||
wc = analysis.get("word_count_stats", {})
|
||||
if wc:
|
||||
lines.append(
|
||||
f"Competitor word count: {wc.get('min', '?')}-{wc.get('max', '?')} "
|
||||
f"(median: {wc.get('median', '?')})"
|
||||
)
|
||||
lines.append(
|
||||
f"Recommended range: {wc.get('recommended_min', '?')}-"
|
||||
f"{wc.get('recommended_max', '?')} words"
|
||||
)
|
||||
|
||||
# Top results
|
||||
lines.append(f"\n## Top {len(organic)} Results")
|
||||
for r in organic[:10]:
|
||||
wc_str = (
|
||||
f" ({r.get('word_count', '?')} words)"
|
||||
if r.get("word_count")
|
||||
else ""
|
||||
)
|
||||
lines.append(f" {r['position']}. {r['title']}{wc_str}")
|
||||
lines.append(f" {r['url']}")
|
||||
|
||||
# PAA
|
||||
paa = analysis.get("paa_questions", serp.get("paa", []))
|
||||
if paa:
|
||||
lines.append(f"\n## People Also Ask ({len(paa)})")
|
||||
for q in paa:
|
||||
lines.append(f" - {q}")
|
||||
|
||||
# Related keywords
|
||||
related = research.get("related_keywords", [])
|
||||
if related:
|
||||
lines.append(f"\n## Related Keywords (top 10)")
|
||||
for kw_data in related[:10]:
|
||||
lines.append(
|
||||
f" - {kw_data['keyword']} "
|
||||
f"(vol: {kw_data['volume']}, "
|
||||
f"diff: {kw_data.get('difficulty', '?')})"
|
||||
)
|
||||
|
||||
# Topics
|
||||
topics = analysis.get("topic_frequency", [])
|
||||
if topics:
|
||||
lines.append(f"\n## Common Topics Across Competitors")
|
||||
for t in topics[:15]:
|
||||
lines.append(
|
||||
f" - {t['topic']} (in {t['competitor_count']} pages)"
|
||||
)
|
||||
|
||||
# Heading patterns
|
||||
hp = analysis.get("heading_patterns", {})
|
||||
if hp:
|
||||
lines.append(f"\n## Heading Structure")
|
||||
lines.append(
|
||||
f" Avg H2s: {hp.get('avg_h2_count', '?')}, "
|
||||
f"Avg H3s: {hp.get('avg_h3_count', '?')}"
|
||||
)
|
||||
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def main():
|
||||
args = parse_args()
|
||||
research = run_research(args)
|
||||
|
||||
# Save
|
||||
filepath = save_research(research, args.save_dir)
|
||||
|
||||
# Output
|
||||
if args.output == "json":
|
||||
print(json.dumps(research, indent=2))
|
||||
elif args.output == "brief":
|
||||
# Minimal output for piping into content generation
|
||||
analysis = research.get("analysis", {})
|
||||
brief_data = {
|
||||
"keyword": research["keyword"],
|
||||
"intent": analysis.get("intent"),
|
||||
"word_count_stats": analysis.get("word_count_stats"),
|
||||
"paa_questions": analysis.get(
|
||||
"paa_questions",
|
||||
research.get("serp", {}).get("paa", []),
|
||||
),
|
||||
"topic_frequency": analysis.get("topic_frequency", [])[:10],
|
||||
"heading_patterns": analysis.get("heading_patterns"),
|
||||
"research_file": filepath,
|
||||
}
|
||||
print(json.dumps(brief_data, indent=2))
|
||||
else:
|
||||
print(format_compact(research))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,160 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
SEO-AGI setup script.
|
||||
Creates config directory, prompts for API keys, installs dependencies.
|
||||
|
||||
Usage:
|
||||
python3 setup.py
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
import json
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
|
||||
CONFIG_DIR = Path.home() / ".config" / "seo-agi"
|
||||
ENV_FILE = CONFIG_DIR / ".env"
|
||||
CONFIG_FILE = CONFIG_DIR / "config.json"
|
||||
OUTPUT_DIR = Path.home() / "Documents" / "SEO-AGI"
|
||||
|
||||
|
||||
def main():
|
||||
print("=" * 50)
|
||||
print(" SEO-AGI Setup")
|
||||
print("=" * 50)
|
||||
print()
|
||||
|
||||
# Create directories
|
||||
CONFIG_DIR.mkdir(parents=True, exist_ok=True)
|
||||
for subdir in ["research", "briefs", "pages", "rewrites"]:
|
||||
(OUTPUT_DIR / subdir).mkdir(parents=True, exist_ok=True)
|
||||
|
||||
data_dir = Path.home() / ".local" / "share" / "seo-agi"
|
||||
for subdir in ["research", "cache"]:
|
||||
(data_dir / subdir).mkdir(parents=True, exist_ok=True)
|
||||
|
||||
print("[OK] Directories created")
|
||||
|
||||
# Install Python dependencies
|
||||
print("\nInstalling dependencies...")
|
||||
deps = ["requests"]
|
||||
try:
|
||||
subprocess.check_call(
|
||||
[sys.executable, "-m", "pip", "install", "--quiet"]
|
||||
+ deps
|
||||
+ ["--break-system-packages"],
|
||||
stdout=subprocess.DEVNULL,
|
||||
stderr=subprocess.DEVNULL,
|
||||
)
|
||||
print("[OK] Core dependencies installed (requests)")
|
||||
except subprocess.CalledProcessError:
|
||||
try:
|
||||
subprocess.check_call(
|
||||
[sys.executable, "-m", "pip", "install", "--quiet"] + deps,
|
||||
stdout=subprocess.DEVNULL,
|
||||
stderr=subprocess.DEVNULL,
|
||||
)
|
||||
print("[OK] Core dependencies installed (requests)")
|
||||
except subprocess.CalledProcessError:
|
||||
print("[WARN] Could not install requests. Install manually:")
|
||||
print(" pip install requests")
|
||||
|
||||
# API Keys
|
||||
print("\n--- API Keys ---")
|
||||
print("DataForSEO is REQUIRED. Get credentials at:")
|
||||
print(" https://app.dataforseo.com/api-dashboard\n")
|
||||
|
||||
existing_env = {}
|
||||
if ENV_FILE.exists():
|
||||
with open(ENV_FILE) as f:
|
||||
for line in f:
|
||||
line = line.strip()
|
||||
if line and not line.startswith("#") and "=" in line:
|
||||
k, _, v = line.partition("=")
|
||||
existing_env[k.strip()] = v.strip()
|
||||
|
||||
# DataForSEO
|
||||
dfs_login = existing_env.get("DATAFORSEO_LOGIN", "")
|
||||
dfs_pass = existing_env.get("DATAFORSEO_PASSWORD", "")
|
||||
|
||||
if dfs_login:
|
||||
print(f"DataForSEO login found: {dfs_login[:4]}***")
|
||||
change = input("Update DataForSEO credentials? [y/N]: ").strip().lower()
|
||||
if change == "y":
|
||||
dfs_login = input("DataForSEO login (email): ").strip()
|
||||
dfs_pass = input("DataForSEO password: ").strip()
|
||||
else:
|
||||
dfs_login = input("DataForSEO login (email): ").strip()
|
||||
dfs_pass = input("DataForSEO password: ").strip()
|
||||
|
||||
# GSC (optional)
|
||||
print("\nGoogle Search Console (OPTIONAL - press Enter to skip)")
|
||||
gsc_path = existing_env.get("GSC_SERVICE_ACCOUNT_PATH", "")
|
||||
if gsc_path:
|
||||
print(f"GSC service account found: {gsc_path}")
|
||||
change = input("Update? [y/N]: ").strip().lower()
|
||||
if change == "y":
|
||||
gsc_path = input("Path to service account JSON: ").strip()
|
||||
else:
|
||||
gsc_path = input("Path to GSC service account JSON (or Enter to skip): ").strip()
|
||||
|
||||
# Write .env
|
||||
env_lines = [
|
||||
"# SEO-AGI Configuration",
|
||||
f"# Generated by setup.py on {__import__('datetime').datetime.now().isoformat()[:10]}",
|
||||
"",
|
||||
"# DataForSEO",
|
||||
f"DATAFORSEO_LOGIN={dfs_login}",
|
||||
f"DATAFORSEO_PASSWORD={dfs_pass}",
|
||||
"",
|
||||
"# Google Search Console",
|
||||
f"GSC_SERVICE_ACCOUNT_PATH={gsc_path}",
|
||||
"",
|
||||
"# Future integrations",
|
||||
f"AHREFS_API_KEY={existing_env.get('AHREFS_API_KEY', '')}",
|
||||
f"SEMRUSH_API_KEY={existing_env.get('SEMRUSH_API_KEY', '')}",
|
||||
"",
|
||||
"# Defaults",
|
||||
f"DEFAULT_LOCATION={existing_env.get('DEFAULT_LOCATION', '2840')}",
|
||||
f"DEFAULT_LANGUAGE={existing_env.get('DEFAULT_LANGUAGE', 'en')}",
|
||||
]
|
||||
|
||||
with open(ENV_FILE, "w") as f:
|
||||
f.write("\n".join(env_lines) + "\n")
|
||||
|
||||
print(f"\n[OK] Credentials saved to {ENV_FILE}")
|
||||
|
||||
# Write default config.json if not exists
|
||||
if not CONFIG_FILE.exists():
|
||||
default_config = {
|
||||
"default_location": 2840,
|
||||
"default_language": "en",
|
||||
"default_site": "",
|
||||
"serp_depth": 10,
|
||||
"save_research": True,
|
||||
"output_dir": str(OUTPUT_DIR),
|
||||
}
|
||||
with open(CONFIG_FILE, "w") as f:
|
||||
json.dump(default_config, f, indent=2)
|
||||
print(f"[OK] Config saved to {CONFIG_FILE}")
|
||||
|
||||
# Verify
|
||||
print("\n--- Setup Summary ---")
|
||||
print(f" DataForSEO: {'configured' if dfs_login else 'NOT SET'}")
|
||||
print(f" GSC: {'configured' if gsc_path else 'not configured (optional)'}")
|
||||
print(f" Output dir: {OUTPUT_DIR}")
|
||||
print(f" Config dir: {CONFIG_DIR}")
|
||||
|
||||
if dfs_login:
|
||||
print("\n[READY] Run a test:")
|
||||
print(f" python3 {Path(__file__).parent}/research.py \"test keyword\"")
|
||||
else:
|
||||
print("\n[WARN] DataForSEO credentials missing. The skill will fall")
|
||||
print(" back to Claude's web search for basic research.")
|
||||
|
||||
print()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user