diff --git a/External/robots-sitemap-validator/SKILL.md b/External/robots-sitemap-validator/SKILL.md new file mode 100644 index 0000000..02ea47b --- /dev/null +++ b/External/robots-sitemap-validator/SKILL.md @@ -0,0 +1,39 @@ +--- +name: robots-sitemap-validator +description: Check robots.txt and sitemap.xml for crawl-blocking mistakes. Use when the user says "check my robots.txt", "validate my sitemap", or "is my site crawlable". +license: MIT +metadata: + author: JustHandled Labs + category: External + credits: 0 + price: $0 + triggers: ["check my robots.txt","validate my sitemap","is my site crawlable"] + version: 1.0.2 +--- + +# Robots Sitemap Validator + +Detect common local robots.txt and sitemap.xml mistakes that can hurt crawling. This detector does not fetch, submit, or change live sites. + +## Workflow + +1. Run scripts/scan_robots_sitemap.py with a folder, files, or --stdin. +2. Review findings by rule id, severity, file, and line. +3. Treat missing sibling sitemap files as review flags when robots.txt references them. + +## Guardrails + +- Read local files or stdin only. +- Do not fetch live URLs or submit sitemaps. +- Do not modify robots.txt or sitemap files. + +## Verification + +Run `python scripts/scan_robots_sitemap.py --help` first. Then use a disposable folder to verify both outcomes: + +1. A `robots.txt` containing `User-agent: *` and `Disallow: /` must report `RSV001`. +2. A `robots.txt` with an absolute `Sitemap:` URL plus a matching, well-formed local `sitemap.xml` must produce zero findings. + +## Commercial Terms + +Price: $0 / 0 credits. Free commercial use for personal or internal team workflows. Do not resell or redistribute the package as-is. diff --git a/External/robots-sitemap-validator/references/audit-checklist.md b/External/robots-sitemap-validator/references/audit-checklist.md new file mode 100644 index 0000000..81ca929 --- /dev/null +++ b/External/robots-sitemap-validator/references/audit-checklist.md @@ -0,0 +1,20 @@ +# Audit Checklist - Robots Sitemap Validator + +## Claimed Capability Coverage + +- Whole-site robots block: backed by rule RSV001. +- Malformed or unknown robots directive: backed by rule RSV002. +- Missing Sitemap directive: backed by rule RSV003. +- Sitemap directive points to absent sibling file: backed by rule RSV004. +- Sitemap XML parse failure: backed by rule RSV005. +- loc URL missing scheme or absolute URL: backed by rule RSV006. +- Sitemap count or size guidance issue: backed by rule RSV007. +- Mixed http and https locs: backed by rule RSV008. + +Unbacked claims to resolve: none. + +## Manual Review Checklist + +- Confirm the local files match the production domain. +- Confirm intentional private-area disallow rules. +- Confirm referenced sitemap files exist in the deploy output. diff --git a/External/robots-sitemap-validator/references/remediation-snippets.md b/External/robots-sitemap-validator/references/remediation-snippets.md new file mode 100644 index 0000000..6b4d751 --- /dev/null +++ b/External/robots-sitemap-validator/references/remediation-snippets.md @@ -0,0 +1,13 @@ +# Remediation Snippets - Robots Sitemap Validator + +## Robots Block + +Remove full-site blocking unless the site is intentionally private or pre-launch. + +## Missing Sitemap + +Add a Sitemap directive pointing to the canonical sitemap URL. + +## Bad Sitemap loc + +Use absolute URLs with http or https schemes. diff --git a/External/robots-sitemap-validator/scripts/scan_robots_sitemap.py b/External/robots-sitemap-validator/scripts/scan_robots_sitemap.py new file mode 100644 index 0000000..ea27db7 --- /dev/null +++ b/External/robots-sitemap-validator/scripts/scan_robots_sitemap.py @@ -0,0 +1,100 @@ +#!/usr/bin/env python3 +from __future__ import annotations +import argparse +import json +import re +import sys +import xml.etree.ElementTree as ET +from pathlib import Path +from urllib.parse import urlparse + +IGNORE_DIRS={".git","node_modules","dist","build","vendor","__pycache__"} +KNOWN={"user-agent","allow","disallow","sitemap","crawl-delay","host"} + +def should_read(path: Path)->bool: + name=path.name.lower() + return name=="robots.txt" or (name.startswith("sitemap") and name.endswith(".xml")) + +def iter_files(paths, stdin_text=None): + if stdin_text is not None: + yield Path("stdin"), stdin_text, None + for raw in paths: + p=Path(raw) + if not p.exists(): yield p,"","missing"; continue + if p.is_file(): + if should_read(p): yield p,p.read_text(encoding="utf-8",errors="replace"),None + continue + for child in p.rglob("*"): + if child.is_file() and should_read(child) and not any(part in IGNORE_DIRS for part in child.parts): + yield child,child.read_text(encoding="utf-8",errors="replace"),None + +def add(findings,code,severity,path,line,message,excerpt): + findings.append({"code":code,"severity":severity,"file":str(path),"line":line,"message":message,"excerpt":" ".join(str(excerpt).split())[:220]}) + +def scan_robots(path,text,findings): + has_sitemap=False + for lineno,line in enumerate(text.splitlines(),1): + stripped=line.strip() + if not stripped or stripped.startswith("#"): continue + if ":" not in stripped: + add(findings,"RSV002","medium",path,lineno,"Robots directive line is malformed.",stripped); continue + key,value=stripped.split(":",1); low=key.strip().lower(); val=value.strip() + if low not in KNOWN: + add(findings,"RSV002","low",path,lineno,"Unknown robots directive.",stripped) + if low=="disallow" and val=="/": + add(findings,"RSV001","high",path,lineno,"robots.txt blocks the whole site.",stripped) + if low=="sitemap": + has_sitemap=True + local=Path(urlparse(val).path).name + if local and path.name!="stdin" and not (path.parent/local).exists(): + add(findings,"RSV004","low",path,lineno,"Sitemap directive points to a sibling file not present locally.",val) + if not has_sitemap: + add(findings,"RSV003","medium",path,1,"robots.txt has no Sitemap directive.","missing Sitemap") + +def loc_texts(root): + for node in root.iter(): + if node.tag.endswith("loc") and node.text: + yield node.text.strip() + +def scan_sitemap(path,text,findings): + try: + root=ET.fromstring(text) + except ET.ParseError as exc: + add(findings,"RSV005","high",path,getattr(exc.position,"__getitem__",lambda i:1)(0) if hasattr(exc,"position") else 1,"Sitemap XML is not well formed.",str(exc)); return + locs=list(loc_texts(root)); schemes=set() + for loc in locs: + parsed=urlparse(loc) + if not parsed.scheme or not parsed.netloc: + add(findings,"RSV006","medium",path,1,"Sitemap loc is not an absolute URL with a scheme.",loc) + if parsed.scheme in {"http","https"}: schemes.add(parsed.scheme) + if len(locs)>50000 or len(text.encode("utf-8"))>50_000_000: + add(findings,"RSV007","medium",path,1,"Sitemap exceeds common size or URL count guidance.",f"{len(locs)} URLs") + if len(schemes)>1: + add(findings,"RSV008","low",path,1,"Sitemap mixes http and https loc URLs.",", ".join(sorted(schemes))) + +def scan(paths=None, stdin_text=None): + paths=paths or [] + findings=[]; missing=[]; scanned=0 + for path,text,error in iter_files(paths, stdin_text): + if error: missing.append(str(path)); continue + scanned+=1 + if path.name.lower()=="robots.txt" or path.name=="stdin": scan_robots(path,text,findings) + if path.name.lower().startswith("sitemap") or "