From 10388e68f4b966ad512a74d04eb252bbd5ca91a6 Mon Sep 17 00:00:00 2001 From: rivassec Date: Fri, 11 Sep 2026 15:23:48 -0700 Subject: [PATCH] llms: full-content index, markdown mirrors, topical grouping, .well-known alias - llms-full.txt now inlines each post's full text (was metadata-only; the 400-char truncate was inert since all summaries are <=160) - new md_mirror plugin publishes {slug}.md next to every published article; both llms files link the markdown variants - llms.txt: posts grouped by category with per-entry dates, expanded key pages (IAM Blast Radius tool, accessibility), tools/code section, meta section (author, security.txt, feeds, citation), provenance line - header description single-sourced via LLMS_DESCRIPTION and refreshed to cover AI security, threat intel, and hiring-security threads - .well-known/llms.txt alias via include shim - robots.txt: explicit allow for GPTBot/ClaudeBot/Claude-Web/ PerplexityBot/Google-Extended, each with /drafts/ excluded - check_wellknown_llms: guard draft-URL leaks, llms-full parity, markdown mirrors, .well-known alias; soft warn at 384KB --- content/extra/robots.txt | 22 ++++++++++ pelicanconf.py | 16 +++++++ plugins/md_mirror.py | 44 +++++++++++++++++++ scripts/check_wellknown_llms.py | 32 +++++++++++--- themes/Flex/templates/llms_full_txt.html | 17 +++++-- themes/Flex/templates/llms_txt.html | 37 +++++++++++----- themes/Flex/templates/llms_txt_wellknown.html | 1 + 7 files changed, 149 insertions(+), 20 deletions(-) create mode 100644 plugins/md_mirror.py create mode 100644 themes/Flex/templates/llms_txt_wellknown.html diff --git a/content/extra/robots.txt b/content/extra/robots.txt index 58403fd9..1bbefea2 100644 --- a/content/extra/robots.txt +++ b/content/extra/robots.txt @@ -2,4 +2,26 @@ User-agent: * Allow: / Disallow: /drafts/ +# AI crawlers are explicitly welcome; a curated index is published for them at +# https://rivassec.com/llms.txt (full content: https://rivassec.com/llms-full.txt). +User-agent: GPTBot +Allow: / +Disallow: /drafts/ + +User-agent: ClaudeBot +Allow: / +Disallow: /drafts/ + +User-agent: Claude-Web +Allow: / +Disallow: /drafts/ + +User-agent: PerplexityBot +Allow: / +Disallow: /drafts/ + +User-agent: Google-Extended +Allow: / +Disallow: /drafts/ + Sitemap: https://rivassec.com/sitemap.xml diff --git a/pelicanconf.py b/pelicanconf.py index 78a3f969..a91f15c4 100644 --- a/pelicanconf.py +++ b/pelicanconf.py @@ -1,3 +1,4 @@ +import datetime import os import subprocess @@ -84,6 +85,7 @@ def _asset_version() -> str: 'related_posts', 'extract_toc', 'img_hygiene', + 'md_mirror', ] # related_posts configuration @@ -188,11 +190,25 @@ def _asset_version() -> str: DIRECT_TEMPLATES = ['index', 'categories', 'tags', 'archives'] # Render llms.txt / llms-full.txt for AI-crawler discovery from the article set. +# The .well-known/ copy exists because some crawlers probe there first. TEMPLATE_PAGES = { 'llms_txt.html': 'llms.txt', 'llms_full_txt.html': 'llms-full.txt', + 'llms_txt_wellknown.html': '.well-known/llms.txt', } +# Single source of truth for the llms.txt header blockquote, referenced by all +# llms templates so the short and full variants cannot drift apart. +LLMS_DESCRIPTION = ( + 'DevSecOps, cloud, and platform security notes by Oliver Rivas. ' + 'Threat-model-driven writing on AWS IAM, Kubernetes, incident response ' + 'and forensics, AI security, threat intelligence and OSINT, hiring ' + 'security, and controls that hold up in production.' +) + +# Build-time stamp for the llms.txt provenance line. +LLMS_GENERATED = datetime.date.today().isoformat() + # One-line intros rendered at the top of each /category/.html page. # Keys match Category: frontmatter values exactly. CATEGORY_INTROS = { diff --git a/plugins/md_mirror.py b/plugins/md_mirror.py new file mode 100644 index 00000000..0908b3d4 --- /dev/null +++ b/plugins/md_mirror.py @@ -0,0 +1,44 @@ +# -*- coding: utf-8 -*- +""" +Markdown mirrors for published articles +======================================= + +A Pelican plugin that copies each published article's Markdown source into the +output root as ``{slug}.md``, so language models and other tooling can fetch a +token-cheap plain-text version of any post (the llms.txt convention of linking +``.md`` variants next to HTML pages). + +Only published articles are mirrored: drafts live on ``generator.drafts`` and +are never touched, so nothing under ``Status: draft`` can leak. The copy runs +on the ``finalized`` signal, after every generator (including the sitemap) has +written, so mirrors never appear in the sitemap. +""" +import os +import shutil + +from pelican import signals + +_articles = [] + + +def _grab_articles(generator): + global _articles + _articles = list(generator.articles) + + +def _write_mirrors(pelican): + out = pelican.settings['OUTPUT_PATH'] + count = 0 + for article in _articles: + src = getattr(article, 'source_path', None) + if not src or not src.endswith('.md') or not os.path.isfile(src): + continue + dst = os.path.join(out, '{0}.md'.format(article.slug)) + shutil.copyfile(src, dst) + count += 1 + print('md_mirror: wrote {0} markdown mirrors'.format(count)) + + +def register(): + signals.article_generator_finalized.connect(_grab_articles) + signals.finalized.connect(_write_mirrors) diff --git a/scripts/check_wellknown_llms.py b/scripts/check_wellknown_llms.py index 70ff748f..9d4229f2 100755 --- a/scripts/check_wellknown_llms.py +++ b/scripts/check_wellknown_llms.py @@ -15,6 +15,7 @@ REPO = Path(__file__).resolve().parent.parent EXPIRES_MIN_DAYS = 30 LLMS_FULL_MAX_BYTES = 512 * 1024 +LLMS_FULL_WARN_BYTES = 384 * 1024 def find_out(): @@ -60,21 +61,40 @@ def main() -> int: except ValueError as e: print(f"::error::security.txt Expires unparseable: {e}"); v += 1 llms = out / "llms.txt"; llmsf = out / "llms-full.txt" + llmswk = out / ".well-known" / "llms.txt" if not llms.is_file(): print("::error::missing llms.txt"); v += 1 + if not llmswk.is_file(): + print("::error::missing .well-known/llms.txt alias"); v += 1 if not llmsf.is_file(): print("::error::missing llms-full.txt"); v += 1 - elif llmsf.stat().st_size > LLMS_FULL_MAX_BYTES: - print(f"::error::llms-full.txt too large ({llmsf.stat().st_size}B)"); v += 1 - if llms.is_file(): - body = llms.read_text(encoding="utf-8", errors="replace") + else: + size = llmsf.stat().st_size + if size > LLMS_FULL_MAX_BYTES: + print(f"::error::llms-full.txt too large ({size}B)"); v += 1 + elif size > LLMS_FULL_WARN_BYTES: + print(f"::warning::llms-full.txt at {size}B, approaching the " + f"{LLMS_FULL_MAX_BYTES}B cap; consider trimming Content excerpts") + body = llms.read_text(encoding="utf-8", errors="replace") if llms.is_file() else "" + fbody = llmsf.read_text(encoding="utf-8", errors="replace") if llmsf.is_file() else "" + # Drafts must never leak into any AI index, whatever the template does. + for name, text in (("llms.txt", body), ("llms-full.txt", fbody)): + if "/drafts/" in text: + print(f"::error::{name} references a /drafts/ URL"); v += 1 + if body: for html in sorted(out.rglob("*.html")): rel = html.relative_to(out).as_posix() if rel.startswith("drafts/") or "/drafts/" in rel: continue # drafts are noindex and intentionally not in llms.txt t = html.read_text(encoding="utf-8", errors="replace") - if "schema.org/BlogPosting" in t and html.stem not in body: - print(f"::error::llms.txt does not reference post: {rel}"); v += 1 + if "schema.org/BlogPosting" in t: + if html.stem not in body: + print(f"::error::llms.txt does not reference post: {rel}"); v += 1 + if fbody and html.stem not in fbody: + print(f"::error::llms-full.txt does not reference post: {rel}"); v += 1 + md = out / f"{html.stem}.md" + if not md.is_file(): + print(f"::error::missing markdown mirror: {html.stem}.md"); v += 1 print(f"security.txt + llms guard: {v} issue(s)", file=sys.stderr) return 1 if v else 0 diff --git a/themes/Flex/templates/llms_full_txt.html b/themes/Flex/templates/llms_full_txt.html index a7f469c6..4e1f539c 100644 --- a/themes/Flex/templates/llms_full_txt.html +++ b/themes/Flex/templates/llms_full_txt.html @@ -1,18 +1,27 @@ {% autoescape false %}# {{ SITENAME }} - full index -> DevSecOps, cloud, and platform security notes by Oliver Rivas. Threat-model-driven writing on AWS, Kubernetes, IAM, hardening, and incident retrospectives. +> {{ LLMS_DESCRIPTION }} -Expanded index for language models. Canonical URL list: {{ SITEURL }}/sitemap.xml +Expanded index for language models: every published post with its full text (tags and markup stripped; code blocks and tables flattened - fetch the Markdown URL for exact formatting). Generated: {{ LLMS_GENERATED }} | {{ articles|length }} posts. Curated index: {{ SITEURL }}/llms.txt | Canonical URL list: {{ SITEURL }}/sitemap.xml ## Posts {% for a in articles|sort(attribute='date', reverse=true) %} ### {{ a.title|striptags }} URL: {{ SITEURL }}/{{ a.url }} +Markdown: {{ SITEURL }}/{{ a.slug }}.md Date: {{ a.date.strftime('%Y-%m-%d') }}{{ (' (updated ' + a.modified.strftime('%Y-%m-%d') + ')') if (a.modified and a.modified.date() != a.date.date()) else '' }} Category: {{ a.category }}{{ (' | Tags: ' + (a.tags|join(', '))) if a.tags else '' }} -Summary: {{ (a.description or a.summary)|striptags|replace('\n', ' ')|truncate(400, true) }} +Summary: {{ (a.description or a.summary)|striptags|replace('\n', ' ') }} +Content: {{ a.content|striptags|truncate(24000, true) }} {% endfor %}## Key pages - About: {{ SITEURL }}/pages/about.html - DevSecOps Guide: {{ SITEURL }}/devsecops-guide.html -{% endautoescape %} +- IAM Blast Radius tool: {{ SITEURL }}/tools/iam-blast-radius/ +- Accessibility: {{ SITEURL }}/accessibility/ + +## Meta +- Author: {{ AUTHOR }} +- Security contact: {{ SITEURL }}/.well-known/security.txt +- Feeds: {{ SITEURL }}/feeds/all.atom.xml and {{ SITEURL }}/feeds/all.rss.xml +{% endautoescape %} \ No newline at end of file diff --git a/themes/Flex/templates/llms_txt.html b/themes/Flex/templates/llms_txt.html index 07035314..2abf8fea 100644 --- a/themes/Flex/templates/llms_txt.html +++ b/themes/Flex/templates/llms_txt.html @@ -1,14 +1,31 @@ {% autoescape false %}# {{ SITENAME }} -> DevSecOps, cloud, and platform security notes by Oliver Rivas. Threat-model-driven writing on AWS, Kubernetes, IAM, hardening, and incident retrospectives. +> {{ LLMS_DESCRIPTION }} -Curated index for language models. Full URL list: {{ SITEURL }}/sitemap.xml - -## Posts -{% for a in articles|sort(attribute='date', reverse=true) %} -- [{{ a.title|striptags }}]({{ SITEURL }}/{{ a.url }}): {{ (a.description or a.summary)|striptags|replace('\n', ' ')|truncate(160, true) }} -{% endfor %} +Curated index for language models. Generated: {{ LLMS_GENERATED }} | {{ articles|length }} posts. Full URL list: {{ SITEURL }}/sitemap.xml +Every post is also available as raw Markdown at the linked .md URL. The expanded index with full post content is at {{ SITEURL }}/llms-full.txt +{% for cat, cat_articles in categories %} +## {{ cat }} +{% for a in cat_articles|sort(attribute='date', reverse=true) %} +- [{{ a.title|striptags }}]({{ SITEURL }}/{{ a.url }}) ({{ a.date.strftime('%Y-%m') }}): {{ (a.description or a.summary)|striptags|replace('\n', ' ')|truncate(160, true) }} [Markdown]({{ SITEURL }}/{{ a.slug }}.md) +{% endfor %}{% endfor %} ## Key pages -- [About RivasSec]({{ SITEURL }}/pages/about.html) -- [DevSecOps Guide]({{ SITEURL }}/devsecops-guide.html) -{% endautoescape %} +- [About RivasSec]({{ SITEURL }}/pages/about.html): who writes this site and why. +- [DevSecOps Guide]({{ SITEURL }}/devsecops-guide.html): hub page linking the core DevSecOps writing. +- [Categories]({{ SITEURL }}/categories.html): all posts grouped by topic. +- [Accessibility]({{ SITEURL }}/accessibility/): accessibility statement. + +## Tools and code +- [IAM Blast Radius]({{ SITEURL }}/tools/iam-blast-radius/): in-browser AWS IAM policy analyzer; computes blast radius and privilege-escalation paths client-side, no policy leaves the page. +- [secure-iam-lint](https://github.com/rivassec/secure-iam-lint): the analyzer engine behind the tool page. +- [iam-safe-defaults](https://github.com/rivassec/iam-safe-defaults): Pulumi library for IAM roles with safety as a precondition. +- [elasticsearch-tools](https://github.com/rivassec/elasticsearch-tools): minimal-privilege Elasticsearch snapshot verification. +- [pwnagotchi plugins](https://github.com/rivassec/pwnagotchi): hardened Pwnagotchi plugins, including multi-phone Bluetooth tethering. +- [GitHub: rivassec](https://github.com/rivassec): all public code. + +## Meta +- Author: {{ AUTHOR }} +- Security contact: {{ SITEURL }}/.well-known/security.txt +- Feeds: {{ SITEURL }}/feeds/all.atom.xml and {{ SITEURL }}/feeds/all.rss.xml +- Cite as: rivassec.com ({{ AUTHOR }}), with the post URL and date. +{% endautoescape %} \ No newline at end of file diff --git a/themes/Flex/templates/llms_txt_wellknown.html b/themes/Flex/templates/llms_txt_wellknown.html new file mode 100644 index 00000000..eb13eb16 --- /dev/null +++ b/themes/Flex/templates/llms_txt_wellknown.html @@ -0,0 +1 @@ +{% include 'llms_txt.html' %} \ No newline at end of file