From 1fee3314d2330073b6eb869327397cd471700111 Mon Sep 17 00:00:00 2001 From: Christoph Dyllick-Brenzinger Date: Mon, 5 Oct 2026 16:47:59 +0200 Subject: [PATCH] Use ReadMe page slugs in sitemap.xml ReadMe appends -1, -2, ... to the slug when an operationId exists in more than one API definition (e.g. deleteBase). The sitemap assumed the slug is always the lowercase operationId, so 16 URLs returned 404 and 65 pages were missing. generate.py now reads the actual slugs from the sidebar data embedded in the public reference page (operationId + API definition ID -> slug) and falls back to the lowercase operationId with a warning if that fails. Co-Authored-By: Claude Opus 5.5 (1M context) --- custom-domain/generate.py | 99 +++++++++++++++++++++++++++++++++++++-- 1 file changed, 96 insertions(+), 3 deletions(-) diff --git a/custom-domain/generate.py b/custom-domain/generate.py index f73cd78..78a294e 100644 --- a/custom-domain/generate.py +++ b/custom-domain/generate.py @@ -12,19 +12,23 @@ """ import glob +import json import os import re import subprocess import sys +import urllib.request from datetime import datetime, timezone import yaml BASE_URL = "https://api.seatable.com" +README_HUB_URL = "https://seatable.readme.io" REPO_ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) OUTPUT_DIR = os.path.join(os.path.dirname(os.path.abspath(__file__)), "output") SPEC_DIR = REPO_ROOT INTRO_DIR = os.path.join(REPO_ROOT, "intro") +README_IDS_FILE = os.path.join(REPO_ROOT, ".github", "readme-ids.json") SPEC_FILES = [ "authentication.yaml", @@ -203,6 +207,72 @@ def resolve_parameter(spec, param): return param +def load_readme_ids(): + """Load the ReadMe version and API definition IDs from .github/readme-ids.json.""" + try: + with open(README_IDS_FILE) as f: + return json.load(f) + except (OSError, ValueError) as e: + print(f"Warning: could not read {README_IDS_FILE}: {e}", file=sys.stderr) + return {} + + +def readme_api_id(readme_ids, spec_file): + """Return the ReadMe API definition ID for a spec file (python-scheduler.yaml -> python_scheduler).""" + key = os.path.splitext(spec_file)[0].replace("-", "_") + return readme_ids.get(key, "") + + +def fetch_readme_slugs(version): + """Fetch the page slugs ReadMe assigned to each operation. + + ReadMe derives the slug from the operationId, but appends -1, -2, ... when the + same operationId exists in more than one API definition (e.g. deleteBase). + The public reference page embeds the full sidebar as JSON, including the + operationId ("sync_unique") and the API definition ID ("apiSetting") of each page. + + Returns dict: (api_definition_id, operationId) -> slug. Empty dict on failure. + """ + url = f"{README_HUB_URL}/v{version}/reference/introduction" + request = urllib.request.Request(url, headers={"User-Agent": "seatable-openapi-generate/1.0"}) + try: + with urllib.request.urlopen(request, timeout=30) as response: + html = response.read().decode("utf-8") + except OSError as e: + print(f"Warning: could not fetch {url}: {e}", file=sys.stderr) + return {} + + match = re.search( + r'', html, re.S + ) + if not match: + print(f"Warning: no sidebar data found in {url}", file=sys.stderr) + return {} + try: + sidebar = json.loads(match.group(1)).get("sidebars", {}).get("refs", []) + except ValueError as e: + print(f"Warning: could not parse sidebar data from {url}: {e}", file=sys.stderr) + return {} + + slugs = {} + + def walk(node): + if isinstance(node, dict): + api = node.get("api") + if node.get("type") == "endpoint" and isinstance(api, dict): + key = (api.get("apiSetting", ""), node.get("sync_unique", "")) + if all(key) and node.get("slug"): + slugs[key] = node["slug"] + for value in node.values(): + walk(value) + elif isinstance(node, list): + for value in node: + walk(value) + + walk(sidebar) + return slugs + + def get_intro_pages(): """Extract slug and title from intro/*.md frontmatter.""" pages = [] @@ -277,9 +347,10 @@ def get_operations(spec): # --------------------------------------------------------------------------- # sitemap.xml # --------------------------------------------------------------------------- -def generate_sitemap(intro_pages, specs_data): +def generate_sitemap(intro_pages, specs_data, readme_ids, readme_slugs): # Collect all URLs with metadata: (url, lastmod, priority) entries = [] + unresolved = [] # Intro / doc pages for page in intro_pages: @@ -292,15 +363,32 @@ def generate_sitemap(intro_pages, specs_data): spec_path = os.path.join(SPEC_DIR, spec_file) op_lastmods, fallback = get_operation_lastmods(spec_path) priority = SPEC_PRIORITY.get(spec_file, 0.3) + api_id = readme_api_id(readme_ids, spec_file) for op in operations: if op["operationId"]: - url = f"{BASE_URL}/reference/{op['operationId'].lower()}" + slug = readme_slugs.get((api_id, op["operationId"])) + if not slug: + # Fallback: ReadMe's default slug, wrong for duplicate operationIds + slug = op["operationId"].lower() + unresolved.append(f"{spec_file}: {op['operationId']}") + url = f"{BASE_URL}/reference/{slug}" if op_lastmods: lastmod = op_lastmods.get(op["operationId"], fallback) else: lastmod = fallback entries.append((url, lastmod, priority)) + if unresolved: + print( + f"Warning: no ReadMe slug found for {len(unresolved)} operations, " + "using lowercase operationId:", + file=sys.stderr, + ) + # List names only on a partial mismatch; if the fetch failed, all are missing + if readme_slugs: + for name in unresolved: + print(f" {name}", file=sys.stderr) + # Deduplicate by URL (keep highest priority) seen = {} for url, lastmod, priority in entries: @@ -523,7 +611,12 @@ def main(): specs_data.append((spec_file, spec, operations)) all_operations.extend(operations) - sitemap = generate_sitemap(intro_pages, specs_data) + readme_ids = load_readme_ids() + version = readme_ids.get("version", "") + readme_slugs = fetch_readme_slugs(version) if version else {} + print(f"ReadMe slugs — {len(readme_slugs)} endpoints (version {version or 'unknown'})") + + sitemap = generate_sitemap(intro_pages, specs_data, readme_ids, readme_slugs) with open(os.path.join(OUTPUT_DIR, "sitemap.xml"), "w") as f: f.write(sitemap) print(f"sitemap.xml — {len(intro_pages)} doc pages + {len(all_operations)} operations")