import os
import json
import re

log_path = r"C:\Users\Abhiram P M\.gemini\antigravity-ide\brain\cddac411-36f2-4599-a02d-70756e350cf6\.system_generated\logs\transcript_full.jsonl"
os.makedirs("recovered_html2", exist_ok=True)

last_target_file = "unknown.html"
with open(log_path, "r", encoding="utf-8") as f:
    for i, line in enumerate(f):
        try:
            step = json.loads(line)
            
            # If it's a planner response with a view_file call, save the file name
            if "tool_calls" in step:
                for call in step["tool_calls"]:
                    if call.get("name") == "default_api:view_file":
                        args = call.get("arguments", {})
                        if isinstance(args, str):
                            try:
                                args = json.loads(args)
                            except:
                                pass
                        
                        target = args.get("AbsolutePath", "")
                        if target:
                            last_target_file = os.path.basename(target)
            
            # If it's a tool response
            if step.get("type") == "TOOL_RESPONSE":
                content = step.get("content", "")
                
                # Check if it contains DOCTYPE
                if "<!DOCTYPE html>" in content:
                    # view_file output usually starts with:
                    # Created At: ...
                    # Completed At: ...
                    # File Path: ...
                    # Total Lines: ...
                    # Total Bytes: ...
                    # Showing lines ...
                    # The following code has been modified...
                    # 1: <!DOCTYPE html>
                    
                    if "Showing lines" in content:
                        # try to extract file path if present
                        m = re.search(r"File Path: `file:///(.*?)`", content)
                        if m:
                            filepath = m.group(1).replace("%20", " ")
                            filename = os.path.basename(filepath)
                        else:
                            filename = f"recovered_{i}_{last_target_file}"
                        
                        # Extract the actual code (remove line numbers)
                        # The lines look like "1: <!DOCTYPE html>"
                        lines = content.split("\n")
                        clean_lines = []
                        in_code = False
                        for l in lines:
                            if l.startswith("The following code has been modified to include a line number"):
                                in_code = True
                                continue
                            
                            if in_code:
                                # Regex to match "123: " at start
                                match = re.match(r"^\d+:\s?(.*)", l)
                                if match:
                                    clean_lines.append(match.group(1))
                                else:
                                    clean_lines.append(l) # just in case
                                    
                        full_html = "\n".join(clean_lines)
                        
                        if full_html.strip():
                            with open(os.path.join("recovered_html2", filename), "w", encoding="utf-8") as out:
                                out.write(full_html)
                            print(f"Recovered {filename} from step {i} ({len(full_html)} bytes)")
        except Exception as e:
            pass
