Skip to content

Instantly share code, notes, and snippets.

@AnythingLinux
Last active September 9, 2026 19:08
Show Gist options
  • Select an option

  • Save AnythingLinux/fdfef0524326ea4fe486114033146f23 to your computer and use it in GitHub Desktop.

Select an option

Save AnythingLinux/fdfef0524326ea4fe486114033146f23 to your computer and use it in GitHub Desktop.
AI.TXT Generator for Linux
#!/usr/bin/env python3
"""
generate_ai_txt.py
School Of Freelancing
AI.TXT generator + relative HTML prefetch discovery
Website:
https://schooloffreelancing.com
Document root:
/var/www/html
WHAT THIS SCRIPT DOES
=====================
1. Recursively discovers every index.html under /var/www/html.
2. Creates an ai.txt file in:
/var/www/html/ai.txt
and in every directory containing an index.html:
/var/www/html/about-us/ai.txt
/var/www/html/client-support/ai.txt
/var/www/html/client-support/centos-linux-support/ai.txt
etc.
3. Reads the existing:
/var/www/html/sitemap.xml
4. Adds a ROOT ai.txt prefetch link to every index.html.
5. The prefetch href is RELATIVE to the depth of that particular
index.html.
ROOT:
/index.html
<link rel="prefetch" href="ai.txt">
ONE LEVEL:
/about-us/index.html
<link rel="prefetch" href="../ai.txt">
TWO LEVELS:
/client-support/centos-linux-support/index.html
<link rel="prefetch" href="../../ai.txt">
THREE LEVELS:
/a/b/c/index.html
<link rel="prefetch" href="../../../ai.txt">
6. The prefetch link is inserted immediately BEFORE </head>.
7. Existing AI.TXT prefetch links are removed first so duplicate
links are not created.
8. Existing unrelated HTML is preserved.
9. DRY RUN is the default.
10. --apply actually changes files.
11. Existing files are backed up as .bak during modification.
12. Every changed file is verified after modification.
13. If verification fails, changes are rolled back from .bak.
14. After successful verification, .bak files are deleted.
15. Apache configuration is tested using:
apache2ctl -t
16. Apache is restarted after a successful apply unless:
--no-apache-restart
17. A detailed report is displayed in the terminal.
IMPORTANT
=========
The HTML prefetch always points to the ROOT ai.txt.
For example:
/about-us/index.html
../ai.txt
-> /ai.txt
/client-support/centos-linux-support/index.html
../../ai.txt
-> /ai.txt
The local subsection ai.txt files are generated separately for
AI discovery, but the HTML prefetch is deliberately relative to
the website ROOT ai.txt.
USAGE
=====
Dry run:
python3 /root/script/generate_ai_txt.py
Apply:
sudo python3 /root/script/generate_ai_txt.py --apply
Apply without Apache restart:
sudo python3 /root/script/generate_ai_txt.py \
--apply \
--no-apache-restart
Explicit configuration:
sudo python3 /root/script/generate_ai_txt.py \
--root /var/www/html \
--domain https://schooloffreelancing.com \
--apply
"""
from __future__ import annotations
import argparse
import datetime
import os
import re
import shutil
import subprocess
import sys
import tempfile
import xml.etree.ElementTree as ET
from pathlib import Path
from urllib.parse import urlparse, unquote
# ============================================================================
# CONFIGURATION
# ============================================================================
DEFAULT_ROOT = "/var/www/html"
DEFAULT_DOMAIN = "https://schooloffreelancing.com"
INDEX_FILENAME = "index.html"
AI_FILENAME = "ai.txt"
SITEMAP_FILENAME = "sitemap.xml"
START_MARKER = "<!-- AI.TXT PREFETCH DISCOVERY -->"
END_MARKER = "<!-- /AI.TXT PREFETCH DISCOVERY -->"
EXCLUDED_DIRECTORIES = {
".git",
".svn",
".hg",
"__pycache__",
"node_modules",
"vendor",
"cgi-bin",
"assets",
"api",
"mcp",
"gsc-repair-reports",
}
# ============================================================================
# REGULAR EXPRESSIONS
# ============================================================================
HEAD_CLOSE_RE = re.compile(
r"</head\s*>",
re.IGNORECASE,
)
MARKER_BLOCK_RE = re.compile(
re.escape(START_MARKER)
+ r".*?"
+ re.escape(END_MARKER),
re.IGNORECASE | re.DOTALL,
)
# Matches a prefetch link whose href ends in ai.txt.
#
# This intentionally catches:
#
# ai.txt
# ../ai.txt
# ../../ai.txt
# /ai.txt
# ./ai.txt
#
# so an old version of the script can safely be corrected.
AI_PREFETCH_RE = re.compile(
r"<link\b[^>]*"
r"\brel\s*=\s*[\"']prefetch[\"'][^>]*"
r"\bhref\s*=\s*[\"'][^\"']*ai\.txt[\"'][^>]*"
r"\s*/?>",
re.IGNORECASE,
)
# ============================================================================
# TERMINAL OUTPUT
# ============================================================================
def header(title: str) -> None:
print()
print("=" * 80)
print(title)
print("=" * 80)
def info(message: str) -> None:
print(f"[INFO] {message}")
def ok(message: str) -> None:
print(f"[ OK ] {message}")
def warn(message: str) -> None:
print(f"[WARN] {message}")
def error(message: str) -> None:
print(f"[ERROR] {message}")
# ============================================================================
# GENERAL UTILITIES
# ============================================================================
def normalize_domain(domain: str) -> str:
"""
Normalize domain to:
https://example.com
without a trailing slash.
"""
domain = domain.strip()
if not domain:
raise ValueError("Domain cannot be empty.")
if not domain.startswith(("http://", "https://")):
domain = "https://" + domain
return domain.rstrip("/")
def read_text(path: Path) -> str:
"""
Read UTF-8 HTML/text safely.
"""
return path.read_text(
encoding="utf-8",
errors="replace",
)
def write_atomic(
path: Path,
content: str,
) -> None:
"""
Atomically write a text file.
"""
path.parent.mkdir(
parents=True,
exist_ok=True,
)
fd, temporary_name = tempfile.mkstemp(
prefix=f".{path.name}.",
suffix=".tmp",
dir=str(path.parent),
text=True,
)
try:
with os.fdopen(
fd,
"w",
encoding="utf-8",
newline="",
) as handle:
handle.write(content)
handle.flush()
os.fsync(handle.fileno())
os.replace(
temporary_name,
path,
)
finally:
try:
if os.path.exists(temporary_name):
os.unlink(temporary_name)
except OSError:
pass
def display_path(
root: Path,
path: Path,
) -> str:
"""
Display filesystem path relative to the website root.
"""
try:
relative = path.relative_to(root)
if str(relative) in ("", "."):
return "/"
return "/" + relative.as_posix()
except ValueError:
return str(path)
# ============================================================================
# RELATIVE PREFETCH CALCULATION
# ============================================================================
def calculate_relative_ai_href(
root: Path,
index_path: Path,
) -> str:
"""
Calculate the relative path from THIS index.html's directory
to the ROOT ai.txt.
This function is intentionally based on the directory depth
of the individual index.html.
Examples:
/var/www/html/index.html
-> ai.txt
/var/www/html/about-us/index.html
-> ../ai.txt
/var/www/html/client-support/index.html
-> ../ai.txt
/var/www/html/client-support/centos-linux-support/index.html
-> ../../ai.txt
/var/www/html/a/b/c/index.html
-> ../../../ai.txt
"""
index_path = index_path.resolve()
root = root.resolve()
try:
relative_directory = index_path.parent.relative_to(root)
except ValueError as exc:
raise ValueError(
f"index.html is outside document root: {index_path}"
) from exc
depth = len(relative_directory.parts)
if depth == 0:
return "ai.txt"
return "../" * depth + "ai.txt"
# ============================================================================
# DISCOVER index.html
# ============================================================================
def discover_index_files(
root: Path,
) -> list[Path]:
"""
Discover every index.html recursively.
Excluded directories are pruned before traversal.
"""
discovered: list[Path] = []
for current_root, directories, files in os.walk(root):
current_path = Path(
current_root
)
directories[:] = sorted(
[
directory
for directory in directories
if directory not in EXCLUDED_DIRECTORIES
and not directory.startswith(".")
],
key=str.lower,
)
for filename in files:
if filename.lower() == INDEX_FILENAME:
discovered.append(
current_path / filename
)
return sorted(
discovered,
key=lambda p: p.as_posix().lower(),
)
# ============================================================================
# PUBLIC URL GENERATION
# ============================================================================
def directory_to_url(
root: Path,
directory: Path,
domain: str,
) -> str:
"""
Convert local directory to public trailing-slash URL.
"""
relative = directory.relative_to(root)
if str(relative) in ("", "."):
return domain + "/"
return (
domain
+ "/"
+ relative.as_posix().strip("/")
+ "/"
)
def index_to_url(
root: Path,
index_path: Path,
domain: str,
) -> str:
"""
Convert an index.html path to its public directory URL.
"""
return directory_to_url(
root=root,
directory=index_path.parent,
domain=domain,
)
# ============================================================================
# SITEMAP READER
# ============================================================================
def read_sitemap(
root: Path,
) -> list[str]:
"""
Read /var/www/html/sitemap.xml.
"""
sitemap_path = root / SITEMAP_FILENAME
if not sitemap_path.exists():
warn(
f"Sitemap not found: {sitemap_path}"
)
return []
try:
tree = ET.parse(
sitemap_path
)
sitemap_root = tree.getroot()
urls: list[str] = []
for element in sitemap_root.iter():
tag = element.tag
if not isinstance(tag, str):
continue
if tag.lower().endswith("loc"):
if element.text:
value = element.text.strip()
if value:
urls.append(value)
return sorted(
set(urls),
key=str.lower,
)
except ET.ParseError as exc:
error(
f"Invalid sitemap XML: {exc}"
)
return []
except OSError as exc:
error(
f"Could not read sitemap: {exc}"
)
return []
# ============================================================================
# HTML METADATA
# ============================================================================
def extract_title(
html: str,
) -> str:
"""
Extract <title>.
"""
match = re.search(
r"<title\b[^>]*>(.*?)</title\s*>",
html,
flags=re.IGNORECASE | re.DOTALL,
)
if not match:
return ""
return re.sub(
r"\s+",
" ",
match.group(1),
).strip()
def extract_description(
html: str,
) -> str:
"""
Extract meta description.
"""
# name before content
match = re.search(
r"<meta\b[^>]*"
r"\bname\s*=\s*[\"']description[\"'][^>]*"
r"\bcontent\s*=\s*[\"']([^\"']*)[\"']"
r"[^>]*>",
html,
flags=re.IGNORECASE,
)
if not match:
# content before name
match = re.search(
r"<meta\b[^>]*"
r"\bcontent\s*=\s*[\"']([^\"']*)[\"'][^>]*"
r"\bname\s*=\s*[\"']description[\"']"
r"[^>]*>",
html,
flags=re.IGNORECASE,
)
if not match:
return ""
return re.sub(
r"\s+",
" ",
match.group(1),
).strip()
def extract_canonical(
html: str,
domain: str,
page_url: str,
) -> str:
"""
Extract canonical URL.
If no canonical exists, use the page's public URL.
"""
# rel before href
match = re.search(
r"<link\b[^>]*"
r"\brel\s*=\s*[\"']canonical[\"'][^>]*"
r"\bhref\s*=\s*[\"']([^\"']+)[\"']"
r"[^>]*>",
html,
flags=re.IGNORECASE,
)
if not match:
# href before rel
match = re.search(
r"<link\b[^>]*"
r"\bhref\s*=\s*[\"']([^\"']+)[\"'][^>]*"
r"\brel\s*=\s*[\"']canonical[\"']"
r"[^>]*>",
html,
flags=re.IGNORECASE,
)
if not match:
return page_url
canonical = match.group(1).strip()
if canonical.startswith("/"):
return domain + canonical
if canonical.startswith(
("http://", "https://")
):
return canonical
return canonical
# ============================================================================
# PAGE RECORDS
# ============================================================================
def build_page_records(
root: Path,
domain: str,
index_files: list[Path],
) -> list[dict]:
"""
Build metadata for all discovered index.html pages.
"""
records: list[dict] = []
for index_path in index_files:
html = read_text(
index_path
)
page_url = index_to_url(
root=root,
index_path=index_path,
domain=domain,
)
record = {
"path": index_path,
"directory": index_path.parent,
"url": page_url,
"title": extract_title(html),
"description": extract_description(html),
"canonical": extract_canonical(
html=html,
domain=domain,
page_url=page_url,
),
}
records.append(
record
)
return sorted(
records,
key=lambda r: r["url"].lower(),
)
# ============================================================================
# SECTION RECORDS
# ============================================================================
def records_under_directory(
directory: Path,
records: list[dict],
) -> list[dict]:
"""
Return all page records under directory.
"""
selected: list[dict] = []
for record in records:
try:
record["directory"].relative_to(
directory
)
selected.append(
record
)
except ValueError:
pass
return sorted(
selected,
key=lambda r: r["url"].lower(),
)
def immediate_child_sections(
root: Path,
directory: Path,
records: list[dict],
domain: str,
) -> list[str]:
"""
Find immediate child sections.
"""
sections = set()
for record in records:
try:
relative = record["directory"].relative_to(
directory
)
except ValueError:
continue
if len(relative.parts) >= 1:
child_name = relative.parts[0]
if child_name:
child_directory = (
directory / child_name
)
sections.add(
directory_to_url(
root=root,
directory=child_directory,
domain=domain,
)
)
return sorted(
sections,
key=str.lower,
)
# ============================================================================
# AI.TXT GENERATION
# ============================================================================
def build_ai_txt(
root: Path,
directory: Path,
domain: str,
records: list[dict],
sitemap_urls: list[str],
) -> str:
"""
Generate directory-aware ai.txt.
Root ai.txt contains all discovered pages.
Subsection ai.txt contains that subsection and its descendants.
"""
section_url = directory_to_url(
root=root,
directory=directory,
domain=domain,
)
selected_records = records_under_directory(
directory=directory,
records=records,
)
selected_urls = {
record["url"].rstrip("/")
for record in selected_records
}
site_host = urlparse(
domain
).netloc.lower()
additional_sitemap_urls = []
for sitemap_url in sitemap_urls:
parsed = urlparse(
sitemap_url
)
if parsed.netloc.lower() != site_host:
continue
sitemap_path = unquote(
parsed.path or "/"
)
if not sitemap_path.startswith("/"):
sitemap_path = "/" + sitemap_path
normalized = (
domain
+ sitemap_path
).rstrip("/")
if directory == root:
belongs = True
else:
relative_directory = (
directory
.relative_to(root)
.as_posix()
)
prefix = (
"/"
+ relative_directory.strip("/")
+ "/"
)
belongs = (
sitemap_path == prefix.rstrip("/")
or sitemap_path.startswith(prefix)
)
if (
belongs
and normalized not in selected_urls
):
if sitemap_path.endswith("/"):
normalized += "/"
additional_sitemap_urls.append(
normalized
)
additional_sitemap_urls = sorted(
set(additional_sitemap_urls),
key=str.lower,
)
generated_at = datetime.datetime.now(
datetime.timezone.utc
).replace(
microsecond=0
).isoformat()
lines: list[str] = []
lines.append("# AI.TXT")
lines.append("# School Of Freelancing")
lines.append("# AI-readable website discovery document")
lines.append("#")
lines.append(
"# Generated from the website directory structure"
)
lines.append(
"# and the existing sitemap.xml."
)
lines.append(
f"# Generated UTC: {generated_at}"
)
lines.append("")
lines.append(
"Site: https://schooloffreelancing.com/"
)
lines.append(
"Sitemap: https://schooloffreelancing.com/sitemap.xml"
)
lines.append(
"Root-AI: https://schooloffreelancing.com/ai.txt"
)
lines.append(
f"Section: {section_url}"
)
lines.append("")
lines.append("# PURPOSE")
lines.append(
"# This document provides lightweight page discovery"
)
lines.append(
"# information for AI agents, automated clients,"
)
lines.append(
"# browsers and other crawlers."
)
lines.append("")
lines.append("# PAGES")
if selected_records:
for record in selected_records:
lines.append(
f"Page: {record['url']}"
)
if record["title"]:
lines.append(
f"Title: {record['title']}"
)
if record["canonical"]:
lines.append(
f"Canonical: {record['canonical']}"
)
if record["description"]:
lines.append(
f"Description: {record['description']}"
)
lines.append("")
else:
lines.append(
f"Page: {section_url}"
)
lines.append("")
if additional_sitemap_urls:
lines.append(
"# ADDITIONAL SITEMAP DISCOVERY"
)
for url in additional_sitemap_urls:
lines.append(
f"Sitemap-Page: {url}"
)
lines.append("")
# Parent section.
if directory != root:
parent = directory.parent
if parent >= root:
lines.append(
"# RELATIONSHIP"
)
lines.append(
"Parent: "
+ directory_to_url(
root=root,
directory=parent,
domain=domain,
)
)
lines.append(
f"Current: {section_url}"
)
lines.append("")
# Child sections.
children = immediate_child_sections(
root=root,
directory=directory,
records=records,
domain=domain,
)
if children:
lines.append(
"# CHILD SECTIONS"
)
for child in children:
lines.append(
f"Section: {child}"
)
lines.append("")
lines.append(
"# HTML DISCOVERY"
)
# This line is informational. The actual HTML href is calculated
# separately for each index.html according to its depth.
lines.append(
'HTML-Discovery: <link rel="prefetch" href="RELATIVE_TO_PAGE">'
)
lines.append("")
return "\n".join(lines)
# ============================================================================
# HTML PREFETCH
# ============================================================================
def remove_existing_ai_prefetch(
html: str,
) -> str:
"""
Remove any previously generated AI.TXT marker block and any
prefetch link whose href points to ai.txt.
"""
html = MARKER_BLOCK_RE.sub(
"",
html,
)
html = AI_PREFETCH_RE.sub(
"",
html,
)
return html
def add_relative_prefetch(
html: str,
relative_href: str,
) -> tuple[str, str]:
"""
Add the relative AI.TXT prefetch immediately BEFORE </head>.
Example:
<link rel="prefetch" href="../../ai.txt">
</head>
Returns:
updated_html, status
status:
UPDATED
NO_HEAD
UNCHANGED
"""
cleaned = remove_existing_ai_prefetch(
html
)
match = HEAD_CLOSE_RE.search(
cleaned
)
if not match:
return html, "NO_HEAD"
block = (
"\n"
f" {START_MARKER}\n"
f' <link rel="prefetch" href="{relative_href}">\n'
f" {END_MARKER}\n"
)
updated = (
cleaned[:match.start()]
+ block
+ cleaned[match.start():]
)
if updated == html:
return html, "UNCHANGED"
return updated, "UPDATED"
# ============================================================================
# HTML PREFETCH VALIDATION
# ============================================================================
def validate_relative_prefetch(
root: Path,
index_path: Path,
) -> tuple[bool, str]:
"""
Verify that exactly one correct relative AI.TXT prefetch
exists before </head>.
"""
html = read_text(
index_path
)
expected_href = calculate_relative_ai_href(
root=root,
index_path=index_path,
)
expected_pattern = re.compile(
r"<link\b[^>]*"
r"\brel\s*=\s*[\"']prefetch[\"'][^>]*"
r"\bhref\s*=\s*[\"']"
+ re.escape(expected_href)
+ r"[\"'][^>]*"
r"\s*/?>",
re.IGNORECASE,
)
matches = expected_pattern.findall(
html
)
if len(matches) != 1:
return False, (
f'Expected exactly one '
f'href="{expected_href}", '
f"found {len(matches)}."
)
head_match = HEAD_CLOSE_RE.search(
html
)
if not head_match:
return False, (
"</head> was not found."
)
before_head = html[
:head_match.start()
]
if not expected_pattern.search(
before_head
):
return False, (
f'href="{expected_href}" '
f"is not before </head>."
)
return True, "Valid."
# ============================================================================
# AI.TXT VALIDATION
# ============================================================================
def validate_ai_txt(
path: Path,
) -> tuple[bool, str]:
"""
Validate generated ai.txt.
"""
if not path.exists():
return False, (
"File does not exist."
)
if not path.is_file():
return False, (
"Path is not a regular file."
)
try:
content = read_text(
path
)
except Exception as exc:
return False, str(exc)
required = [
"# AI.TXT",
"Site: https://schooloffreelancing.com/",
"Sitemap: https://schooloffreelancing.com/sitemap.xml",
"Root-AI: https://schooloffreelancing.com/ai.txt",
"# HTML DISCOVERY",
]
for item in required:
if item not in content:
return False, (
f"Missing: {item}"
)
if not content.strip():
return False, (
"File is empty."
)
return True, "Valid."
# ============================================================================
# BACKUPS
# ============================================================================
def create_backup(
path: Path,
) -> Path:
"""
Create path.bak.
"""
backup = Path(
str(path) + ".bak"
)
shutil.copy2(
path,
backup,
)
return backup
def restore_backup(
original: Path,
backup: Path,
) -> bool:
try:
if backup.exists():
shutil.copy2(
backup,
original,
)
return True
except Exception as exc:
error(
f"Could not restore {original}: {exc}"
)
return False
def delete_backup(
backup: Path,
) -> bool:
try:
if backup.exists():
backup.unlink()
return not backup.exists()
except Exception as exc:
warn(
f"Could not remove {backup}: {exc}"
)
return False
# ============================================================================
# APACHE
# ============================================================================
def apache_config_test() -> tuple[bool, str]:
"""
Test Apache configuration.
"""
commands = [
["apache2ctl", "-t"],
["apachectl", "-t"],
]
last_output = ""
for command in commands:
try:
result = subprocess.run(
command,
stdout=subprocess.PIPE,
stderr=subprocess.STDOUT,
text=True,
timeout=30,
)
last_output = result.stdout.strip()
if result.returncode == 0:
return True, last_output
except FileNotFoundError:
continue
except Exception as exc:
last_output = str(exc)
return False, last_output
def restart_apache() -> tuple[bool, str]:
"""
Restart Apache.
"""
commands = [
["systemctl", "restart", "apache2"],
["service", "apache2", "restart"],
]
last_output = ""
for command in commands:
try:
result = subprocess.run(
command,
stdout=subprocess.PIPE,
stderr=subprocess.STDOUT,
text=True,
timeout=60,
)
last_output = result.stdout.strip()
if result.returncode == 0:
return True, last_output
except FileNotFoundError:
continue
except Exception as exc:
last_output = str(exc)
return False, last_output
# ============================================================================
# REPORT
# ============================================================================
class Report:
def __init__(self):
self.index_count = 0
self.ai_created = 0
self.ai_updated = 0
self.ai_unchanged = 0
self.html_updated = 0
self.html_unchanged = 0
self.html_failed = 0
self.backups_created = 0
self.backups_removed = 0
self.errors: list[str] = []
def add_error(
self,
message: str,
) -> None:
self.errors.append(
message
)
def print(
self,
) -> None:
header(
"FINAL REPORT"
)
print(
f"index.html discovered : {self.index_count}"
)
print(
f"ai.txt created : {self.ai_created}"
)
print(
f"ai.txt updated : {self.ai_updated}"
)
print(
f"ai.txt unchanged : {self.ai_unchanged}"
)
print(
f"HTML prefetch updated : {self.html_updated}"
)
print(
f"HTML prefetch unchanged : {self.html_unchanged}"
)
print(
f"HTML prefetch failed : {self.html_failed}"
)
print(
f".bak files created : {self.backups_created}"
)
print(
f".bak files removed : {self.backups_removed}"
)
print(
f"Errors : {len(self.errors)}"
)
if self.errors:
print()
print(
"ERROR DETAILS"
)
print(
"-" * 80
)
for item in self.errors:
print(
f" - {item}"
)
# ============================================================================
# MAIN
# ============================================================================
def main() -> int:
parser = argparse.ArgumentParser(
description=(
"Generate ai.txt and add depth-aware relative "
"AI.TXT prefetch links to every index.html."
)
)
parser.add_argument(
"--root",
default=DEFAULT_ROOT,
help=(
f"Website document root. "
f"Default: {DEFAULT_ROOT}"
),
)
parser.add_argument(
"--domain",
default=DEFAULT_DOMAIN,
help=(
f"Website domain. "
f"Default: {DEFAULT_DOMAIN}"
),
)
parser.add_argument(
"--apply",
action="store_true",
help=(
"Apply changes. Without --apply the script "
"performs a dry run."
),
)
parser.add_argument(
"--no-apache-restart",
action="store_true",
help=(
"Do not restart Apache after applying changes."
),
)
args = parser.parse_args()
# ------------------------------------------------------------------------
# Configuration.
# ------------------------------------------------------------------------
try:
root = Path(
args.root
).resolve()
domain = normalize_domain(
args.domain
)
except Exception as exc:
error(
f"Invalid configuration: {exc}"
)
return 1
report = Report()
header(
"SCHOOL OF FREELANCING - AI.TXT GENERATOR"
)
print(
"Mode : "
+ (
"APPLY"
if args.apply
else "DRY RUN"
)
)
print(
f"Document root : {root}"
)
print(
f"Domain : {domain}"
)
print(
f"Root ai.txt : {root / AI_FILENAME}"
)
print(
"Relative href : ENABLED"
)
print(
"Insertion position : immediately before </head>"
)
print(
"Apache restart : "
+ (
"NO"
if args.no_apache_restart
else "YES"
)
)
# ------------------------------------------------------------------------
# Root validation.
# ------------------------------------------------------------------------
header(
"1. DOCUMENT ROOT"
)
if not root.exists():
error(
f"Document root does not exist: {root}"
)
return 1
if not root.is_dir():
error(
f"Document root is not a directory: {root}"
)
return 1
ok(
f"Document root exists: {root}"
)
# ------------------------------------------------------------------------
# Discover index.html.
# ------------------------------------------------------------------------
header(
"2. DISCOVERING index.html FILES"
)
index_files = discover_index_files(
root
)
report.index_count = len(
index_files
)
if not index_files:
error(
"No index.html files were found."
)
return 1
ok(
f"Found {len(index_files)} index.html file(s)."
)
print()
for index_path in index_files:
href = calculate_relative_ai_href(
root=root,
index_path=index_path,
)
print(
f" {display_path(root, index_path):<65}"
f' -> href="{href}"'
)
# ------------------------------------------------------------------------
# Sitemap.
# ------------------------------------------------------------------------
header(
"3. READING sitemap.xml"
)
sitemap_urls = read_sitemap(
root
)
if sitemap_urls:
ok(
f"Loaded {len(sitemap_urls)} sitemap URL(s)."
)
else:
warn(
"No sitemap URLs were loaded."
)
# ------------------------------------------------------------------------
# Analyze pages.
# ------------------------------------------------------------------------
header(
"4. ANALYZING HTML PAGES"
)
records = build_page_records(
root=root,
domain=domain,
index_files=index_files,
)
ok(
f"Analyzed {len(records)} page(s)."
)
# ------------------------------------------------------------------------
# Determine directories.
# ------------------------------------------------------------------------
directories = sorted(
{
record["directory"]
for record in records
},
key=lambda path: (
len(
path.relative_to(root).parts
),
path.as_posix().lower(),
),
)
# Root MUST always have ai.txt.
if root not in directories:
directories.insert(
0,
root,
)
# ------------------------------------------------------------------------
# AI.TXT plan.
# ------------------------------------------------------------------------
header(
"5. AI.TXT PLAN"
)
ai_operations: list[dict] = []
for directory in directories:
ai_path = (
directory
/ AI_FILENAME
)
content = build_ai_txt(
root=root,
directory=directory,
domain=domain,
records=records,
sitemap_urls=sitemap_urls,
)
if ai_path.exists():
existing = read_text(
ai_path
)
if existing == content:
operation = "UNCHANGED"
report.ai_unchanged += 1
else:
operation = "UPDATE"
report.ai_updated += 1
else:
operation = "CREATE"
report.ai_created += 1
ai_operations.append(
{
"path": ai_path,
"content": content,
"operation": operation,
}
)
print(
f"[{operation:<9}] "
f"{display_path(root, ai_path)}"
)
# ------------------------------------------------------------------------
# HTML prefetch plan.
# ------------------------------------------------------------------------
header(
"6. HTML RELATIVE PREFETCH PLAN"
)
html_operations: list[dict] = []
for index_path in index_files:
original = read_text(
index_path
)
relative_href = calculate_relative_ai_href(
root=root,
index_path=index_path,
)
updated, status = add_relative_prefetch(
html=original,
relative_href=relative_href,
)
page_display = display_path(
root,
index_path,
)
if status == "NO_HEAD":
report.html_failed += 1
message = (
f"{page_display}: </head> not found."
)
report.add_error(
message
)
error(
message
)
continue
if updated == original:
operation = "UNCHANGED"
report.html_unchanged += 1
else:
operation = "UPDATE"
report.html_updated += 1
html_operations.append(
{
"path": index_path,
"original": original,
"updated": updated,
"href": relative_href,
"operation": operation,
}
)
print(
f"[{operation:<9}] "
f"{page_display}"
)
print(
f" "
f'<link rel="prefetch" href="{relative_href}">'
)
# ------------------------------------------------------------------------
# Dry run.
# ------------------------------------------------------------------------
if not args.apply:
header(
"7. DRY RUN COMPLETE"
)
ok(
"No website files were modified."
)
ok(
"No .bak files were created."
)
ok(
"Apache was NOT restarted."
)
print()
print(
"To apply the planned changes:"
)
print()
print(
"sudo python3 /root/script/generate_ai_txt.py --apply"
)
report.print()
return (
0
if not report.errors
else 2
)
# ------------------------------------------------------------------------
# Apply.
# ------------------------------------------------------------------------
header(
"7. APPLYING CHANGES"
)
backups: list[
tuple[Path, Path]
] = []
try:
# ====================================================================
# AI.TXT
# ====================================================================
for operation in ai_operations:
path = operation["path"]
if operation["operation"] == "UNCHANGED":
continue
if path.exists():
backup = create_backup(
path
)
backups.append(
(
path,
backup,
)
)
report.backups_created += 1
write_atomic(
path,
operation["content"],
)
valid, reason = validate_ai_txt(
path
)
if not valid:
raise RuntimeError(
f"AI.TXT validation failed for "
f"{path}: {reason}"
)
ok(
"Applied AI.TXT: "
+ display_path(
root,
path,
)
)
# ====================================================================
# HTML
# ====================================================================
for operation in html_operations:
path = operation["path"]
if operation["operation"] == "UNCHANGED":
continue
backup = create_backup(
path
)
backups.append(
(
path,
backup,
)
)
report.backups_created += 1
write_atomic(
path,
operation["updated"],
)
valid, reason = validate_relative_prefetch(
root=root,
index_path=path,
)
if not valid:
raise RuntimeError(
f"HTML prefetch validation failed for "
f"{path}: {reason}"
)
ok(
"Applied HTML: "
+ display_path(
root,
path,
)
+ f' -> href="{operation["href"]}"'
)
except Exception as exc:
# ====================================================================
# ROLLBACK
# ====================================================================
header(
"ROLLBACK"
)
error(
str(exc)
)
warn(
"Restoring files from .bak..."
)
rollback_failed = False
for original, backup in reversed(
backups
):
if restore_backup(
original,
backup,
):
ok(
"Restored: "
+ display_path(
root,
original,
)
)
else:
rollback_failed = True
if rollback_failed:
error(
"One or more files could not be restored."
)
warn(
"Apache was NOT restarted."
)
return 1
# ------------------------------------------------------------------------
# Post-apply verification.
# ------------------------------------------------------------------------
header(
"8. POST-APPLY VERIFICATION"
)
verification_failed = False
# Verify every ai.txt.
for operation in ai_operations:
path = operation["path"]
valid, reason = validate_ai_txt(
path
)
if valid:
ok(
"Verified AI.TXT: "
+ display_path(
root,
path,
)
)
else:
verification_failed = True
message = (
f"AI.TXT verification failed: "
f"{path}: {reason}"
)
report.add_error(
message
)
error(
message
)
# Verify every HTML prefetch.
for operation in html_operations:
path = operation["path"]
valid, reason = validate_relative_prefetch(
root=root,
index_path=path,
)
if valid:
ok(
"Verified HTML: "
+ display_path(
root,
path,
)
+ f' -> href="{operation["href"]}"'
)
else:
verification_failed = True
message = (
f"HTML verification failed: "
f"{path}: {reason}"
)
report.add_error(
message
)
error(
message
)
# ------------------------------------------------------------------------
# Rollback if verification failed.
# ------------------------------------------------------------------------
if verification_failed:
header(
"VERIFICATION FAILED - ROLLBACK"
)
warn(
"Restoring all backed-up files..."
)
for original, backup in reversed(
backups
):
if restore_backup(
original,
backup,
):
ok(
"Restored: "
+ display_path(
root,
original,
)
)
error(
"Changes were rolled back."
)
warn(
"Apache was NOT restarted."
)
return 1
# ------------------------------------------------------------------------
# Remove .bak files.
# ------------------------------------------------------------------------
header(
"9. REMOVING TEMPORARY .BAK FILES"
)
for original, backup in backups:
if delete_backup(
backup
):
report.backups_removed += 1
ok(
f"Removed: {backup}"
)
else:
report.add_error(
f"Backup remains: {backup}"
)
# ------------------------------------------------------------------------
# Apache.
# ------------------------------------------------------------------------
if args.no_apache_restart:
header(
"10. APACHE"
)
warn(
"Apache restart skipped because "
"--no-apache-restart was supplied."
)
else:
header(
"10. APACHE CONFIGURATION TEST"
)
apache_ok, apache_output = (
apache_config_test()
)
if apache_output:
print(
apache_output
)
if not apache_ok:
error(
"Apache configuration test FAILED."
)
report.add_error(
"Apache configuration test failed."
)
warn(
"Apache was NOT restarted."
)
report.print()
return 1
ok(
"Apache configuration test passed."
)
# -------------------------------------------------------------
# Restart.
# -------------------------------------------------------------
header(
"11. RESTARTING APACHE"
)
restart_ok, restart_output = (
restart_apache()
)
if restart_output:
print(
restart_output
)
if not restart_ok:
error(
"Apache restart FAILED."
)
report.add_error(
"Apache restart failed."
)
report.print()
return 1
ok(
"Apache restarted successfully."
)
# ------------------------------------------------------------------------
# Final report.
# ------------------------------------------------------------------------
report.print()
header(
"SUCCESS"
)
ok(
"AI.TXT generation completed."
)
ok(
"Relative HTML prefetch generation completed."
)
ok(
"Prefetch links were inserted immediately before </head>."
)
print()
print(
"ROOT AI.TXT:"
)
print(
f" {domain}/ai.txt"
)
print()
print(
"RELATIVE PREFETCH EXAMPLES:"
)
examples = [
"/index.html",
"/about-us/index.html",
"/client-support/index.html",
"/client-support/centos-linux-support/index.html",
"/freelancing-training/linux-freelancing-training/index.html",
]
for example in examples:
path = (
root
/ example.lstrip("/")
)
if path.exists():
href = calculate_relative_ai_href(
root=root,
index_path=path,
)
print(
f" {example:<65}"
f'href="{href}"'
)
print()
print(
"All relative prefetch links resolve to:"
)
print(
f" {domain}/ai.txt"
)
print()
print(
"DONE."
)
return 0
# ============================================================================
# ENTRY POINT
# ============================================================================
if __name__ == "__main__":
try:
sys.exit(
main()
)
except KeyboardInterrupt:
print()
warn(
"Interrupted by user."
)
sys.exit(130)
except PermissionError as exc:
print()
error(
f"Permission denied: {exc}"
)
print()
print(
"For --apply, run:"
)
print(
"sudo python3 /root/script/generate_ai_txt.py --apply"
)
sys.exit(1)
except Exception as exc:
print()
error(
f"Unexpected error: {exc}"
)
sys.exit(1)
@AnythingLinux

Copy link
Copy Markdown
Author

How to use:

touch generate_ai_txt.py
nano generate_ai_txt.py
chmod +x generate_ai_txt.py
python3 generate_ai_txt.py
sudo python3 generate_ai_txt.py --apply

Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment