Files
madomeda/core/processor.py
T
2025-10-23 20:14:21 +02:00

321 lines
12 KiB
Python

"""
Main frontmatter processor
"""
import sys
from pathlib import Path
from datetime import datetime
from core.database import Database
from core.repository import Repository
from core.frontmatter import FrontmatterParser
from core.rules_processor import RulesProcessor
from core.template import TemplateManager
from core.llm import LLMClient
from core.changelog import ChangeLog
from core.config import ConfigManager
class FrontmatterProcessor:
def __init__(self, repo: Repository, template_name: str,
whatif: bool = False, no_confirm: bool = False, force: bool = False, add_only: bool = False):
self.repo = repo
self.template_name = template_name
self.whatif = whatif
self.no_confirm = no_confirm
self.force = force
self.add_only = add_only
# Initialize configuration
self.config = ConfigManager()
self.config.ensure_config_exists()
# Get API configuration
api_config = self.config.get_api_config()
self.db = Database(repo.path / 'madomeda.db')
self.rules = RulesProcessor(self.config.rules_dir)
self.templates = TemplateManager(self.config.templates_dir)
self.llm = LLMClient(self.config.prompts_dir, api_config)
self.changelog = ChangeLog(repo.path)
self.parser = FrontmatterParser()
def process(self):
"""Main processing loop"""
# Print configuration info if running in interactive shell
if sys.stdin.isatty():
self.config.print_config_info()
print(f"Processing repository: {self.repo.path}")
print(f"Template: {self.template_name}")
print(f"Mode: {'DRY RUN' if self.whatif else 'LIVE'}")
if self.add_only:
print(f"Add-only mode: Preserving existing keys")
print()
# Get template
template = self.templates.load_template(self.template_name)
schema = self.templates.get_structured_output_schema(template)
# Get all markdown files
files = self.repo.get_tracked_files()
print(f"Found {len(files)} markdown files")
print()
# Start changelog session
self.changelog.start_session()
# Process each file
for file_path in files:
self._process_file(file_path, template, schema)
# Collect all tags and store in database
self._collect_all_tags()
# Write changelog
if not self.whatif:
self.changelog.write()
self.db.close()
def _process_file(self, file_path: Path, template: dict, schema: dict):
"""Process a single markdown file"""
relative_path = file_path.relative_to(self.repo.path)
print(f"Processing: {relative_path}")
# Add file to database
file_id = self.db.add_file(str(relative_path))
# Read file content
with open(file_path, 'r', encoding='utf-8') as f:
content = f.read()
# Parse frontmatter
original_fm, body = self.parser.parse(content)
if original_fm:
print(f" Found existing frontmatter")
else:
print(f" No frontmatter found")
original_fm = {}
# Store original frontmatter in database
if original_fm:
self.db.add_frontmatter(file_id, original_fm, conformant=False)
# Get git history
git_info = self.repo.get_file_history(file_path)
self.db.add_commit_info(
file_id,
git_info['author_name'],
git_info['author_email'],
git_info['commit_hash'],
git_info['commit_tag'],
git_info['latest_hash']
)
# Apply rules
normalized_fm, is_conformant, violations = self.rules.apply_rules(dict(original_fm))
# Check if we need to update
needs_update = self.force or not is_conformant or not self._matches_template(normalized_fm, template)
if not needs_update:
print(f" OK Already conformant")
if original_fm:
# Mark as conformant in database
fm_id = self.db.add_frontmatter(file_id, original_fm, conformant=True)
print()
return
# Build new frontmatter from template
new_fm = self._build_frontmatter(template, schema, original_fm, normalized_fm, body, file_path, git_info)
# In add-only mode, merge with normalized original
if self.add_only:
# Start with normalized original
merged_fm = dict(normalized_fm)
# Add missing template fields
for key, value in new_fm.items():
if key not in merged_fm:
merged_fm[key] = value
new_fm = merged_fm
# Store tags in database
if 'tags' in new_fm and isinstance(new_fm['tags'], list):
for tag in new_fm['tags']:
self.db.add_tag(tag)
# Check for metadata loss
discarded = self._check_discarded_metadata(original_fm, new_fm)
if discarded and not self.add_only: # Only warn in non-add-only mode
print(f" WARNING Metadata will be discarded: {', '.join(discarded)}")
if not self.no_confirm and not self.whatif:
response = input(" Continue? (y/n): ")
if response.lower() != 'y':
print(f" Skipped")
print()
return
# Show what would change
if self.whatif:
print(f" Would update frontmatter:")
print(f" Changes: {self._describe_changes(original_fm, new_fm)}")
else:
# Write new frontmatter
new_content = self.parser.serialize(new_fm, body)
with open(file_path, 'w', encoding='utf-8') as f:
f.write(new_content)
# Update database
self.db.add_frontmatter(file_id, new_fm, conformant=True)
# Add to changelog
self.changelog.add_entry(str(relative_path))
print(f" OK Updated frontmatter")
print()
def _matches_template(self, frontmatter: dict, template: dict) -> bool:
"""Check if frontmatter has all required template fields"""
for field in template.keys():
if field not in frontmatter:
return False
# Allow empty lists/strings as valid values
return True
def _build_frontmatter(self, template: dict, schema: dict, original_fm: dict,
normalized_fm: dict, body: str, file_path: Path, git_info: dict) -> dict:
"""Build new frontmatter from template"""
new_fm = {}
all_tags = self.db.get_all_tags()
for field, field_value in template.items():
# Parse field to get strategy and type
strategy_str, field_type = self.templates.parse_field(str(field_value))
options = strategy_str.split('|')
value = None
for option in options:
option = option.strip()
if option.startswith('lit:'):
# Literal/static value
value = option[4:].strip()
# Convert based on type
value = self._convert_value_type(value, field_type)
elif option == 'heur':
# Use heuristics
value = self._get_heuristic_value(field, normalized_fm, body, file_path, git_info)
elif option == 'orig':
# Use original value
value = normalized_fm.get(field)
elif option.startswith('ai'):
# Use LLM
parts = option.split(None, 1)
prompt_name = parts[1] if len(parts) > 1 else 'default'
value = self.llm.infer_metadata(field, prompt_name, original_fm, new_fm, body, all_tags, schema)
# Accept value if not None (empty lists/strings are valid)
if value is not None:
break
if value is not None:
new_fm[field] = value
return new_fm
def _convert_value_type(self, value: str, field_type: str):
"""Convert literal value to appropriate type"""
if field_type == 'checkbox':
return value.lower() in ('true', '1', 'yes', 'on')
elif field_type == 'number':
try:
if '.' in value:
return float(value)
return int(value)
except ValueError:
return value
elif field_type == 'list':
# Parse as JSON array or return empty list
if value == '[]':
return []
try:
import json
return json.loads(value)
except:
return [value]
else:
# text, date, datetime - keep as string
return value
def _get_heuristic_value(self, field: str, normalized_fm: dict, body: str,
file_path: Path, git_info: dict):
"""Get field value using heuristics"""
if field == 'title':
# Try to extract from document
title = self.parser.extract_title(body)
if not title:
# Use filename without extension
title = file_path.stem
return title
elif field == 'created':
return git_info.get('created')
elif field == 'changed':
return git_info.get('changed')
elif field == 'authors':
return git_info.get('authors', [])
elif field == 'version':
# Prefer latest tag, fallback to short hash
latest_tag = git_info.get('latest_tag', '')
if latest_tag:
return latest_tag
latest_hash = git_info.get('latest_hash', '')
return latest_hash[:7] if latest_hash else ''
elif field == 'tags':
tags = normalized_fm.get('tags', [])
# Return None if empty so AI inference is tried
return tags if tags else None
return None
def _check_discarded_metadata(self, original: dict, new: dict) -> list:
"""Check for metadata that will be discarded"""
discarded = []
for key in original.keys():
if key not in new:
discarded.append(key)
return discarded
def _describe_changes(self, original: dict, new: dict) -> str:
"""Describe changes between frontmatter versions"""
changes = []
# New fields
for key in new.keys():
if key not in original:
changes.append(f"+{key}")
# Modified fields
for key in new.keys():
if key in original and original[key] != new[key]:
changes.append(f"~{key}")
# Removed fields
for key in original.keys():
if key not in new:
changes.append(f"-{key}")
return ', '.join(changes) if changes else 'none'
def _collect_all_tags(self):
"""Collect all unique tags from processed files and store in database"""
# Tags are collected during file processing
all_tags = self.db.get_all_tags()
print(f"Total unique tags in repository: {len(all_tags)}")