"""Stage 3: Reader — Incremental context loading.""" import subprocess import tempfile import os import json # Loading order: README → docs/ → examples/ → package.json → requirements.txt → source code LOAD_ORDER = [ "README.md", "README", "readme.md", "docs/README.md", "docs/workflows.md", "docs/guide.md", "docs/architecture.md", "examples/", "example/", "demo/", "package.json", "requirements.txt", "setup.py", "pyproject.toml", "Cargo.toml", ] def extract_text_from_file(filepath): """Read file content, cap at max tokens.""" try: with open(filepath, 'r', errors='ignore') as f: content = f.read() if len(content) > 40000: content = content[:40000] + "\n\n... [truncated] ..." return content except: return None def read_repo(repo_url, config=None): """ Clone repo, load context incrementally, return structured context. Returns only what's needed to understand the workflow. """ result = { "repository": repo_url, "context_loaded": [], "source_code_loaded": False, "content": {}, "decision_reason": "", } repo_name = repo_url.rstrip("/").split("/")[-1] with tempfile.TemporaryDirectory() as tmpdir: clone_path = os.path.join(tmpdir, repo_name) # Clone — shallow clone but ensure top-level files are fetched try: clone_cmd = ["git", "clone", "--depth=1", "--no-single-branch", repo_url, clone_path] subprocess.run(clone_cmd, capture_output=True, timeout=60) except: result["error"] = "Clone failed" return result # Load in order for pattern in LOAD_ORDER: if pattern.endswith("/"): # Directory — scan for relevant files dirpath = os.path.join(clone_path, pattern) if os.path.isdir(dirpath): for fname in sorted(os.listdir(dirpath))[:5]: fpath = os.path.join(dirpath, fname) if os.path.isfile(fpath) and fname.endswith(('.md', '.py', '.js', '.ts', '.yaml', '.yml')): content = extract_text_from_file(fpath) if content and len(content.strip()) > 50: result["content"][f"{pattern}{fname}"] = content result["context_loaded"].append(f"{pattern}{fname}") else: # File path — check for it directly filepath = os.path.join(clone_path, pattern) if os.path.exists(filepath) and os.path.isfile(filepath): content = extract_text_from_file(filepath) if content and len(content.strip()) > 50: result["content"][pattern] = content result["context_loaded"].append(pattern) # Check if we have enough to proceed total_chars = sum(len(v) for v in result["content"].values()) if len(result["context_loaded"]) == 0: result["decision_reason"] = "No readable documentation found" result["status"] = "INSUFFICIENT" elif total_chars < 200: result["decision_reason"] = "Too little content to extract workflow" result["status"] = "INSUFFICIENT" else: result["decision_reason"] = f"Workflow identified from {len(result['context_loaded'])} files ({total_chars} chars)" result["status"] = "READY" return result