import os

# Let's inspect each file and clean any duplicate HTML blocks
for filename in ['ministries-pastoral.html', 'ministries-formation.html', 'ministries-education.html', 'community-history.html']:
    with open(filename, 'r', encoding='utf-8') as f:
        text = f.read()

    # Check for multiple DOCTYPE or <html> tags
    first_doctype = text.find('<!DOCTYPE html>')
    second_doctype = text.find('<!DOCTYPE html>', first_doctype + 15)
    second_doctype_partial = text.find('TYPE html>', first_doctype + 15)

    print(f"{filename} -> first: {first_doctype}, second: {second_doctype}, partial: {second_doctype_partial}")

    # Let's find the footer
    footer_idx = text.find('<footer')
    if footer_idx != -1:
        # Find closing </html> after footer
        end_html_idx = text.find('</html>', footer_idx)
        if end_html_idx != -1:
            clean_text = text[:end_html_idx + len('</html>')]
            with open(filename, 'w', encoding='utf-8') as f:
                f.write(clean_text)
            print(f"Cleaned trailing content in {filename}")
