Add browsable category folders

This commit is contained in:
cporter202
2026-07-23 13:24:08 -04:00
parent 405ca77fd1
commit e3542f1ea7
18 changed files with 847 additions and 17 deletions
+252
View File
@@ -0,0 +1,252 @@
#!/usr/bin/env python3
"""Generate browsable category pages from the canonical Worker CSV."""
from __future__ import annotations
import argparse
import csv
import re
import sys
from collections import defaultdict
from pathlib import Path
ROOT = Path(__file__).resolve().parents[1]
CSV_PATH = ROOT / "data" / "workers.csv"
CATEGORY_ROOT = ROOT / "categories"
CATEGORIES = (
(
"E-Commerce & Marketplaces",
"e-commerce-marketplaces",
"🛍️",
"Products, pricing, reviews, stores, suppliers and marketplace intelligence.",
),
(
"Social & Creator Data",
"social-creator-data",
"🌐",
"Profiles, posts, comments, creators, events and audience signals.",
),
(
"Search, Maps & SEO",
"search-maps-seo",
"🔎",
"SERPs, maps, local business data, keywords and SEO intelligence.",
),
(
"Jobs & Recruiting",
"jobs-recruiting",
"💼",
"Job listings, candidate discovery, employers, salaries and recruiting data.",
),
(
"Lead Generation & Company Intelligence",
"lead-generation-company-intelligence",
"🎯",
"Companies, decision-makers, suppliers, business emails and sales prospects.",
),
(
"AI & Research",
"ai-research",
"",
"AI answers, cited sources and structured research workflows.",
),
(
"Developer & Web Utilities",
"developer-web-utilities",
"🧰",
"Browser automation, extraction, screenshots, parsing and data utilities.",
),
(
"Finance & Markets",
"finance-markets",
"📈",
"Public-market, company and financial time-series data.",
),
(
"Real Estate",
"real-estate",
"🏠",
"Property listings, valuations, history and location data.",
),
(
"Education & Knowledge",
"education-knowledge",
"📚",
"Courses, books, metadata and learning catalog intelligence.",
),
(
"News & Media",
"news-media",
"📰",
"News articles, media pages and publication monitoring.",
),
)
def table_text(value: str) -> str:
return " ".join(value.split()).replace("|", "\\|")
def link_text(value: str) -> str:
return table_text(value).replace("[", "\\[").replace("]", "\\]")
def anchor_slug(value: str) -> str:
return re.sub(r"-+", "-", re.sub(r"[^a-z0-9]+", "-", value.lower())).strip("-")
def load_workers() -> list[dict[str, str]]:
with CSV_PATH.open(encoding="utf-8", newline="") as handle:
workers = list(csv.DictReader(handle))
workers.sort(key=lambda row: (row["category"], row["source"], row["title"].casefold()))
return workers
def render_index(grouped: dict[str, list[dict[str, str]]]) -> str:
lines = [
'<p align="center"><a href="../README.md">← Main directory</a></p>',
"",
"# CoreClaw API Categories",
"",
"Browse all **118 CoreClaw Worker APIs** through 11 focused categories. "
"Open a category to see every matching API, grouped by source.",
"",
"| | Category | APIs | Sources |",
"|---:|---|---:|---:|",
]
for name, slug, emoji, _summary in CATEGORIES:
rows = grouped[name]
source_count = len({row["source"] for row in rows})
lines.append(
f"| {emoji} | [**{name}**]({slug}/README.md) | "
f"**{len(rows)}** | **{source_count}** |"
)
lines.extend(
[
"",
"> [!NOTE]",
"> Worker links include the maintainers `fpr=chris69` referral parameter. "
"Purchases through these links may support this directory at no additional cost to you.",
"",
'<p align="center"><a href="../README.md#directory"><strong>View the complete directory →</strong></a></p>',
"",
]
)
return "\n".join(lines)
def render_category(
name: str,
emoji: str,
summary: str,
workers: list[dict[str, str]],
) -> str:
sources: dict[str, list[dict[str, str]]] = defaultdict(list)
for worker in workers:
sources[worker["source"]].append(worker)
lines = [
'<p align="center"><a href="../README.md">← All categories</a> · '
'<a href="../../README.md">Main directory</a></p>',
"",
f"# {emoji} {name}",
"",
summary,
"",
f"**{len(workers)} APIs** across **{len(sources)} data sources**",
"",
"## Sources",
"",
" · ".join(
f"[{source}](#{anchor_slug(source)})"
for source in sorted(sources, key=str.casefold)
),
"",
]
for source in sorted(sources, key=str.casefold):
source_workers = sorted(sources[source], key=lambda row: row["title"].casefold())
lines.extend(
[
f'<a id="{anchor_slug(source)}"></a>',
f"## {source} ({len(source_workers)})",
"",
"| API | Worker ID | What it does |",
"|---|---|---|",
]
)
for worker in source_workers:
lines.append(
f'| [**{link_text(worker["title"])}**]({worker["url"]}) '
f'| `{table_text(worker["path"])}` '
f'| {table_text(worker["description"])} |'
)
lines.append("")
lines.extend(
[
"> [!NOTE]",
"> Links open the exact CoreClaw Worker page and include the maintainers "
"affiliate attribution. See the main README for the full disclosure.",
"",
'<p align="center"><a href="../README.md">← Browse all categories</a> · '
'<a href="../../README.md#directory">Complete API directory</a></p>',
"",
]
)
return "\n".join(lines)
def expected_pages() -> dict[Path, str]:
workers = load_workers()
grouped: dict[str, list[dict[str, str]]] = defaultdict(list)
for worker in workers:
grouped[worker["category"]].append(worker)
known = {name for name, *_rest in CATEGORIES}
unknown = sorted(set(grouped) - known)
if unknown:
raise SystemExit(f"Unknown categories in CSV: {', '.join(unknown)}")
pages = {CATEGORY_ROOT / "README.md": render_index(grouped)}
for name, slug, emoji, summary in CATEGORIES:
pages[CATEGORY_ROOT / slug / "README.md"] = render_category(
name, emoji, summary, grouped[name]
)
return pages
def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument(
"--check",
action="store_true",
help="fail when generated category pages are missing or stale",
)
args = parser.parse_args()
stale = []
for path, content in expected_pages().items():
content = content.replace("\r\n", "\n")
if args.check:
if not path.is_file() or path.read_text(encoding="utf-8").replace("\r\n", "\n") != content:
stale.append(path.relative_to(ROOT))
continue
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(content, encoding="utf-8", newline="\n")
if stale:
print("Generated category pages are stale:", file=sys.stderr)
for path in stale:
print(f" {path}", file=sys.stderr)
print("Run: python scripts/generate_category_pages.py", file=sys.stderr)
raise SystemExit(1)
action = "Verified" if args.check else "Generated"
print(f"{action} {len(expected_pages())} category pages.")
if __name__ == "__main__":
main()
+20
View File
@@ -7,11 +7,15 @@ import xml.etree.ElementTree as ET
from collections import Counter
from pathlib import Path
from generate_category_pages import CATEGORIES
ROOT = Path(__file__).resolve().parents[1]
CSV_PATH = ROOT / "data" / "workers.csv"
README_PATH = ROOT / "README.md"
BANNER_PATH = ROOT / "assets" / "coreclaw-api-directory-banner.svg"
CATEGORY_ROOT = ROOT / "categories"
CATEGORY_SLUGS = {name: slug for name, slug, _emoji, _summary in CATEGORIES}
REQUIRED_FIELDS = {
"title",
@@ -69,6 +73,22 @@ def main() -> None:
if missing_from_readme:
fail(f"README is missing {len(missing_from_readme)} Worker paths")
category_index = CATEGORY_ROOT / "README.md"
if not category_index.is_file():
fail("missing categories/README.md")
for row in rows:
category_slug = CATEGORY_SLUGS.get(row["category"])
if category_slug is None:
fail(f"unknown category: {row['category']}")
category_page = CATEGORY_ROOT / category_slug / "README.md"
if not category_page.is_file():
fail(f"missing category page: {category_page.relative_to(ROOT)}")
if row["path"] not in category_page.read_text(encoding="utf-8"):
fail(
f"{category_page.relative_to(ROOT)} is missing Worker path: "
f"{row['path']}"
)
try:
root = ET.parse(BANNER_PATH).getroot()
except ET.ParseError as error: