#!/usr/bin/env python3
"""
Corporate Home Ownership Analysis Script
Prepared for: Welcome Home Coalition
Author: Nikki Ricks
Date: November 20, 2025

This script analyzes property ownership data to identify corporate and private equity
ownership of single-family homes in Portland, Oregon.

Usage:
    python analyze_corporate_ownership.py input_data.csv

Input CSV should contain at least these columns:
- NAME (owner name)
- SITUSADDR (property address)
- PROPCLASS (property classification)
- SITUSCITY (city)
"""

import pandas as pd
import re
import sys
from collections import Counter
from typing import Dict, List, Tuple

# Corporate identifier patterns
CORPORATE_PATTERNS = {
    'LLC': r'\b(LLC|L\.L\.C\.|Limited Liability Company)\b',
    'Corporation': r'\b(Inc\.|Incorporated|Corp\.|Corporation|Co\.)\b',
    'Trust': r'\b(Trust|Trustee|Trustees)\b',
    'Investment Entity': r'\b(Properties|Investments?|Capital|Partners|Ventures|Real Estate|RE\s+LLC|Holdings?)\b',
    'LP/LLP': r'\b(LP|L\.P\.|LLP|L\.L\.P\.|Limited Partnership)\b'
}

def identify_corporate_owner(name: str) -> Tuple[bool, str]:
    """
    Identify if an owner name appears to be corporate/institutional.

    Returns: (is_corporate, entity_type)
    """
    if pd.isna(name) or name == '':
        return False, 'Unknown'

    name_upper = name.upper()

    # Check each pattern
    for entity_type, pattern in CORPORATE_PATTERNS.items():
        if re.search(pattern, name_upper):
            return True, entity_type

    return False, 'Individual'

def categorize_investor_size(property_count: int) -> str:
    """Categorize investors by number of properties owned."""
    if property_count == 1:
        return 'Single Property Owner'
    elif property_count <= 10:
        return 'Small Investor (2-10 properties)'
    elif property_count <= 50:
        return 'Mid-Size Investor (11-50 properties)'
    elif property_count <= 100:
        return 'Large Investor (51-100 properties)'
    else:
        return 'Institutional Investor (100+ properties)'

def is_single_family_residential(prop_class: str) -> bool:
    """
    Determine if property classification indicates single-family residential.

    Common classifications:
    - 'SINGLE FAMILY RESIDENTIAL'
    - 'RESIDENTIAL'
    - '111' (assessment code for single family)
    """
    if pd.isna(prop_class):
        return False

    prop_class_upper = str(prop_class).upper()

    # Common patterns for single-family residential
    sf_patterns = [
        'SINGLE FAMILY',
        'SINGLE-FAMILY',
        'SFR',
        'RES - SINGLE',
        r'^\s*111\s*$'  # Assessment code
    ]

    for pattern in sf_patterns:
        if re.search(pattern, prop_class_upper):
            return True

    return False

def analyze_corporate_ownership(df: pd.DataFrame) -> Dict:
    """
    Main analysis function to identify corporate ownership patterns.
    """
    print("=" * 80)
    print("CORPORATE HOME OWNERSHIP ANALYSIS")
    print("=" * 80)
    print()

    # Filter for Portland single-family homes
    print("Filtering data...")
    portland_mask = df['SITUSCITY'].str.upper() == 'PORTLAND'
    sf_mask = df['PROPCLASS'].apply(is_single_family_residential)

    portland_sf = df[portland_mask & sf_mask].copy()

    print(f"Total properties in dataset: {len(df):,}")
    print(f"Portland single-family homes: {len(portland_sf):,}")
    print()

    # Identify corporate owners
    print("Identifying corporate owners...")
    portland_sf['is_corporate'], portland_sf['entity_type'] = zip(
        *portland_sf['NAME'].apply(identify_corporate_owner)
    )

    # Count properties by owner
    owner_counts = portland_sf['NAME'].value_counts()
    portland_sf['properties_owned'] = portland_sf['NAME'].map(owner_counts)
    portland_sf['investor_category'] = portland_sf['properties_owned'].apply(
        categorize_investor_size
    )

    # Calculate key metrics
    total_sf_homes = len(portland_sf)
    corporate_owned = portland_sf['is_corporate'].sum()
    corporate_percentage = (corporate_owned / total_sf_homes * 100) if total_sf_homes > 0 else 0

    # Entity type breakdown
    entity_types = portland_sf[portland_sf['is_corporate']]['entity_type'].value_counts()

    # Top corporate owners
    corporate_owners = portland_sf[portland_sf['is_corporate']]
    top_owners = corporate_owners.groupby('NAME').agg({
        'SITUSADDR': 'count',
        'entity_type': 'first',
        'investor_category': 'first'
    }).sort_values('SITUSADDR', ascending=False).head(50)
    top_owners.columns = ['Properties Owned', 'Entity Type', 'Investor Category']

    # Institutional investors (100+ properties)
    institutional = portland_sf[portland_sf['properties_owned'] >= 100]
    institutional_unique = institutional['NAME'].nunique()
    institutional_properties = len(institutional)

    # Results summary
    results = {
        'total_sf_homes': total_sf_homes,
        'corporate_owned': corporate_owned,
        'corporate_percentage': corporate_percentage,
        'entity_types': entity_types,
        'top_owners': top_owners,
        'institutional_investors': institutional_unique,
        'institutional_properties': institutional_properties,
        'data': portland_sf
    }

    # Print summary
    print("=" * 80)
    print("KEY FINDINGS")
    print("=" * 80)
    print()
    print(f"Total Single-Family Homes in Portland: {total_sf_homes:,}")
    print(f"Corporate/Institutional Owned: {corporate_owned:,} ({corporate_percentage:.1f}%)")
    print(f"Institutional Investors (100+ properties): {institutional_unique}")
    print(f"Properties owned by institutional investors: {institutional_properties:,}")
    print()

    print("CORPORATE ENTITY TYPE BREAKDOWN:")
    print("-" * 80)
    for entity_type, count in entity_types.items():
        percentage = (count / corporate_owned * 100) if corporate_owned > 0 else 0
        print(f"  {entity_type}: {count:,} properties ({percentage:.1f}%)")
    print()

    print("TOP 20 CORPORATE OWNERS:")
    print("-" * 80)
    print(top_owners.head(20).to_string())
    print()

    return results

def export_results(results: Dict, output_prefix: str = 'analysis'):
    """Export analysis results to CSV files."""
    print("=" * 80)
    print("EXPORTING RESULTS")
    print("=" * 80)
    print()

    # Export full data with corporate flags
    full_data_file = f'{output_prefix}_full_data.csv'
    results['data'].to_csv(full_data_file, index=False)
    print(f"✓ Full dataset with corporate flags: {full_data_file}")

    # Export top owners
    top_owners_file = f'{output_prefix}_top_owners.csv'
    results['top_owners'].to_csv(top_owners_file)
    print(f"✓ Top corporate owners: {top_owners_file}")

    # Export summary statistics
    summary_file = f'{output_prefix}_summary.txt'
    with open(summary_file, 'w') as f:
        f.write("CORPORATE HOME OWNERSHIP ANALYSIS - SUMMARY\n")
        f.write("=" * 80 + "\n\n")
        f.write(f"Total Single-Family Homes: {results['total_sf_homes']:,}\n")
        f.write(f"Corporate Owned: {results['corporate_owned']:,}\n")
        f.write(f"Percentage: {results['corporate_percentage']:.1f}%\n")
        f.write(f"Institutional Investors: {results['institutional_investors']}\n")
        f.write(f"Institutional Properties: {results['institutional_properties']:,}\n\n")
        f.write("Entity Type Breakdown:\n")
        for entity_type, count in results['entity_types'].items():
            f.write(f"  {entity_type}: {count:,}\n")
    print(f"✓ Summary statistics: {summary_file}")
    print()

def main():
    """Main execution function."""
    if len(sys.argv) < 2:
        print("Usage: python analyze_corporate_ownership.py <input_csv_file>")
        print()
        print("Input CSV should contain columns:")
        print("  - NAME: Owner name")
        print("  - SITUSADDR: Property address")
        print("  - PROPCLASS: Property classification")
        print("  - SITUSCITY: City name")
        sys.exit(1)

    input_file = sys.argv[1]

    try:
        print(f"Loading data from: {input_file}")
        df = pd.read_csv(input_file)
        print(f"Loaded {len(df):,} records")
        print()

        # Run analysis
        results = analyze_corporate_ownership(df)

        # Export results
        export_results(results, output_prefix='portland_corporate_ownership')

        print("=" * 80)
        print("ANALYSIS COMPLETE")
        print("=" * 80)

    except FileNotFoundError:
        print(f"ERROR: File not found: {input_file}")
        sys.exit(1)
    except Exception as e:
        print(f"ERROR: {str(e)}")
        import traceback
        traceback.print_exc()
        sys.exit(1)

if __name__ == '__main__':
    main()
