diff --git a/docs/source-materials/specs/auto-uid.py b/docs/source-materials/specs/auto-uid.py old mode 100644 new mode 100755 index 1f8114f..a0072f6 --- a/docs/source-materials/specs/auto-uid.py +++ b/docs/source-materials/specs/auto-uid.py @@ -1,678 +1,349 @@ -python - #!/usr/bin/env python3 +from __future__ import annotations """ - Enhanced auto_uid.py - Production-ready CLI for managing markdown with UIDs - - Features: - - autoUID: Preserve and validate subject-based TR IDs - - hideUIDlocal: Move leading [ID] to HTML comments for clean reading - - autoNumbering: Ensure proper section numbering hierarchy - - autoTitle: Ensure YAML front-matter has title - - Dry-run mode: Preview changes safely - - Backup creation: Automatic .bak files for safety - - ID validation: Check for duplicates, format issues - - Statistics: Report processing details - - Usage: - python tools/auto_uid.py --file [options] - - Options: - --hide-local-uids Move leading [ID] to HTML comments - --auto-numbering Ensure numbered section headings - --auto-title Ensure YAML title exists - --validate-ids Check ID format and uniqueness - --dry-run Preview changes without modifying - --backup Create .bak file before changes - --inplace Modify file in place (default: create new file) - --output, -o Specify output file path - --stats Show processing statistics - """ import argparse - import re - import sys - -from collections import Counter - +from datetime import datetime, timezone from pathlib import Path -from typing import Dict, List, Set, Tuple - - - # Enhanced regex patterns - ID_LEADING_RE = re.compile(r'^\s*\[(?P[A-Z0-9\-]+)\]\s*(?P.*)') - YAML_BOUNDARY = '---' - SECTION_HEADING_RE = re.compile(r'^(\d+(?:\.\d+)*)\s+(.+)$') - TR_ID_FORMAT_RE = re.compile(r'^TR-[A-Z]+(?:-[A-Z]+)*-\d+$') - class ProcessingStats: - def __init__(self): - self.ids_processed = 0 - self.ids_hidden = 0 - self.sections_numbered = 0 - self.title_added = False - self.errors = [] - self.warnings = [] - self.duplicate_ids = [] - - def report(self) -> str: - lines = [ - - f"Processing Statistics:", - + "Processing Statistics:", f" IDs processed: {self.ids_processed}", - f" IDs hidden: {self.ids_hidden}", - f" Sections numbered: {self.sections_numbered}", - f" Title added: {'Yes' if self.title_added else 'No'}", - f" Errors: {len(self.errors)}", - - f" Warnings: {len(self.warnings)}" - + f" Warnings: {len(self.warnings)}", ] - if self.duplicate_ids: - lines.append(f" Duplicate IDs found: {', '.join(self.duplicate_ids)}") - return '\n'.join(lines) - -def validate_ids(text: str, stats: ProcessingStats) -> Dict[str, List[int]]: - +def validate_ids(text: str, stats: ProcessingStats) -> dict[str, list[int]]: """Validate ID format and detect duplicates. Returns dict of id -> line_numbers.""" - id_locations = {} - lines = text.splitlines() - - for line_num, line in enumerate(lines, 1): - match = ID_LEADING_RE.match(line) - if match: - id_val = match.group('id') - stats.ids_processed += 1 - - # Track locations - if id_val not in id_locations: - id_locations[id_val] = [] - id_locations[id_val].append(line_num) - - # Validate format - if not TR_ID_FORMAT_RE.match(id_val): - stats.warnings.append(f"Line {line_num}: ID '{id_val}' doesn't match TR format") - - # Find duplicates - for id_val, locations in id_locations.items(): - if len(locations) > 1: - stats.duplicate_ids.append(id_val) - stats.errors.append(f"Duplicate ID '{id_val}' found on lines: {', '.join(map(str, locations))}") - - return id_locations - def hide_local_uids(text: str, stats: ProcessingStats) -> str: - """Enhanced version with better formatting preservation.""" - out_lines = [] - for line in text.splitlines(): - match = ID_LEADING_RE.match(line) - if match: - id_val = match.group('id') - content = match.group('content') - # Preserve original spacing in content - out_lines.append(f'{content}') - stats.ids_hidden += 1 - else: - out_lines.append(line) - return '\n'.join(out_lines) + ('\n' if text.endswith('\n') else '') - def ensure_yaml_title(text: str, stats: ProcessingStats) -> str: - """Enhanced YAML parsing with better error handling.""" - lines = text.splitlines() - if not lines: - return text - - # Check for existing YAML front matter - if lines[0].strip() == YAML_BOUNDARY: - try: - end_idx = lines[1:].index(YAML_BOUNDARY) + 1 - except ValueError: - stats.errors.append("YAML front matter started but never closed") - return text - - yaml_block = lines[1:end_idx] - - # Check if title already exists - has_title = any(l.strip().lower().startswith('title:') for l in yaml_block) - if has_title: - return text - - # Extract potential title from content - content_lines = lines[end_idx + 1:] - title = extract_title_from_content(content_lines) - - # Insert title at the beginning of YAML block - new_yaml = [f'title: "{title}"'] + yaml_block - result_lines = [YAML_BOUNDARY] + new_yaml + [YAML_BOUNDARY] + content_lines - stats.title_added = True - return '\n'.join(result_lines) + ('\n' if text.endswith('\n') else '') - - else: - # No YAML front matter - create one - title = extract_title_from_content(lines) - yaml_block = [ - YAML_BOUNDARY, - f'title: "{title}"', - YAML_BOUNDARY, - - '' - + '', ] - result_lines = yaml_block + lines - stats.title_added = True - return '\n'.join(result_lines) + ('\n' if text.endswith('\n') else '') - -def extract_title_from_content(lines: List[str]) -> str: - +def extract_title_from_content(lines: list[str]) -> str: """Extract a reasonable title from content lines.""" - for line in lines: - line = line.strip() - if not line: - continue - # Skip HTML comments - if line.startswith('