Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
96 changes: 62 additions & 34 deletions scripts/llms_txt2ctx.py
100755 → 100644
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,7 @@ class AttrDict(dict):
"""A dictionary subclass that allows attribute-style access to its keys."""

def __init__(self, *args: Any, **kwargs: Any) -> None:
"""Initialize the mapping and convert nested dictionaries to attribute-accessible mappings."""
super().__init__(*args, **kwargs)
for k, v in self.items():
if isinstance(v, dict):
Expand All @@ -24,23 +25,41 @@ def __init__(self, *args: Any, **kwargs: Any) -> None:
self[k] = [AttrDict(i) if isinstance(i, dict) else i for i in v]

def __getattr__(self, name: str) -> Any:
"""
Retrieve a dictionary value through attribute-style access.

Parameters:
name (str): The dictionary key to retrieve.

Returns:
Any: The value associated with the key.
"""
try:
return self[name]
except KeyError:
raise AttributeError(name)

def __setattr__(self, name: str, value: Any) -> None:
"""
Set an attribute by storing its value under the corresponding dictionary key.

Parameters:
name (str): The attribute and dictionary key to assign.
value (Any): The value to store.
"""
self[name] = value


def slugify(title: str) -> str:
"""Convert a title string into a valid, clean XML tag name.

Args:
title: The input title string.

"""
Convert a title into a lowercase, hyphen-separated XML tag name.

Parameters:
title (str): The title to normalize.

Returns:
A lowercased, hyphenated alphanumeric string safe for XML tag names.
str: The normalized tag name, using ``page`` for empty results and
prefixing names that begin with a digit with ``p``.
"""
s: str = title.strip().lower()
s = re.sub(r'[^a-z0-9\-]', '-', s)
Expand All @@ -54,13 +73,14 @@ def slugify(title: str) -> str:


def escape_attr(val: str) -> str:
"""Escape special characters for use in XML attribute values.

"""
Escape a string for use in an XML attribute value.

Args:
val: The raw attribute string.

val: Raw attribute text.
Returns:
A serialized XML-safe attribute string.
The escaped attribute string, or an empty string for falsy input.
"""
if not val:
return ""
Expand All @@ -73,13 +93,13 @@ def escape_attr(val: str) -> str:


def parse_link(line: str) -> Optional[Dict[str, Optional[str]]]:
"""Parse a single markdown list line containing a hyperlink and optional description.

"""Parse a markdown list item containing a hyperlink and optional description.
Args:
line: The raw markdown list line.

line: The raw markdown list item.
Returns:
A dictionary with keys 'title', 'url', 'desc' if matched, otherwise None.
A dictionary with `title`, `url`, and `desc` keys if the line matches, otherwise `None`.
"""
# Regex matching optional list bullets, then a markdown link [title](url) followed optionally by : description
match = re.match(r'^\s*[-\*]\s*\[([^\]]+)\]\(([^)]+)\)(?:\s*:\s*(.*))?$', line.strip())
Expand All @@ -96,13 +116,14 @@ def parse_link(line: str) -> Optional[Dict[str, Optional[str]]]:


def parse_llms_file(txt: str) -> AttrDict:
"""Parse the raw content of an llms.txt file into a structured AttrDict object.

Args:
txt: The raw string content of the llms.txt file.

"""
Parse an llms.txt document into structured project metadata.

Parameters:
txt: The raw llms.txt content.

Returns:
An AttrDict containing 'title', 'summary', 'info', and 'sections'.
An AttrDict with title, summary, info, and sections.
"""
# Split text into introduction and sections using H2 headers
parts: List[str] = re.split(r'^##\s*(.*?)$', txt, flags=re.MULTILINE)
Expand Down Expand Up @@ -179,13 +200,14 @@ def parse_llms_file(txt: str) -> AttrDict:


def get_doc_content(url_or_path: str) -> str:
"""Retrieve document content. For local relative paths, reads from repository root.

"""
Read a local document relative to the repository root or provide a placeholder for unavailable content.

Args:
url_or_path: The URL or path to retrieve.

url_or_path: A local document path or an HTTP(S) URL.
Returns:
The content string or a placeholder message on failure or network block.
The document content, or a placeholder comment when remote content is skipped, the file is missing, or reading fails.
"""
if url_or_path.startswith(('http://', 'https://')):
return f"<!-- Remote content skipped: {url_or_path} -->"
Expand All @@ -203,14 +225,15 @@ def get_doc_content(url_or_path: str) -> str:


def create_ctx(txt: str, optional: bool = False) -> str:
"""Create an LLM context XML compilation from the raw text content of an llms.txt file.

"""
Compile llms.txt content into an XML context document.

Args:
txt: Raw text content of the input llms.txt file.
optional: If True, includes optional H2 sections. Otherwise, skips them.

txt: Raw llms.txt text to parse.
optional: Whether to include the section named Optional.
Returns:
An XML-formatted string compiling the project and documentation sections.
The assembled XML document.
"""
parsed = parse_llms_file(txt)

Expand Down Expand Up @@ -261,7 +284,12 @@ def create_ctx(txt: str, optional: bool = False) -> str:


def main() -> None:
"""CLI execution entrypoint."""
"""
Run the command-line converter and print the generated XML.

Reads the input `llms.txt` file specified on the command line. The optional
section can be included with `--optional` or `--optional=true|1|yes`.
"""
if len(sys.argv) < 2 or sys.argv[1] in ('-h', '--help'):
print("Usage: llms_txt2ctx <input_llms.txt> [--optional <True|False>]", file=sys.stderr)
sys.exit(1)
Expand Down
2 changes: 2 additions & 0 deletions tests/test_test_asimp_mock_data_yaml_validity.py
Original file line number Diff line number Diff line change
Expand Up @@ -99,13 +99,15 @@ def test_float_coerced_condition_evaluates_true_for_the_real_quoted_string_value
self.assertTrue(compiled(report_frontmatter=report_frontmatter))

def test_float_coerced_condition_evaluates_false_for_a_mismatched_version(self) -> None:
"""Verify that the expected version condition evaluates to false for a mismatched version."""
env = Environment()
compiled = env.compile_expression(EXPECTED_OKF_VERSION_CONDITION)
report_frontmatter = {"okf_version": "0.2"}
self.assertFalse(compiled(report_frontmatter=report_frontmatter))

def test_previous_unconverted_condition_would_regress_into_always_false(self) -> None:
# Demonstrates *why* comparing string to float literal directly is incorrect
"""Verify that the previous direct string-to-float comparison evaluates to `false` for the quoted version value."""
env = Environment()
compiled = env.compile_expression(PREVIOUS_UNCONVERTED_CONDITION)
report_frontmatter = {"okf_version": "0.1"}
Expand Down