"""
Scholarship Web Scraper
Scrapes scholarship data from various Indian scholarship portals
"""

import requests
from bs4 import BeautifulSoup
import json
import re
from datetime import datetime
import logging

logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)


class ScholarshipScraper:
    """Scraper for Indian scholarship websites"""
    
    def __init__(self):
        self.headers = {
            'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36'
        }
        self.scholarships = []

    def scrape_buddy4study(self):
        """Scrape scholarships from Buddy4Study"""
        url = "https://www.buddy4study.com/scholarships"
        try:
            response = requests.get(url, headers=self.headers, timeout=30)
            soup = BeautifulSoup(response.content, 'html.parser')
            
            scholarship_cards = soup.find_all('div', class_='scholarship-card')
            
            for card in scholarship_cards:
                try:
                    title = card.find('h3').get_text(strip=True) if card.find('h3') else "Unknown"
                    amount = card.find('span', class_='amount').get_text(strip=True) if card.find('span', class_='amount') else "Varies"
                    deadline = card.find('span', class_='deadline').get_text(strip=True) if card.find('span', class_='deadline') else "Not specified"
                    eligibility = card.find('p', class_='eligibility').get_text(strip=True) if card.find('p', class_='eligibility') else ""
                    link = card.find('a')['href'] if card.find('a') else ""
                    
                    self.scholarships.append({
                        'name': title,
                        'amount': amount,
                        'deadline': deadline,
                        'eligibility': eligibility,
                        'source': 'Buddy4Study',
                        'link': link,
                        'scraped_at': datetime.now().isoformat()
                    })
                except Exception as e:
                    logger.error(f"Error parsing scholarship card: {e}")
                    
        except Exception as e:
            logger.error(f"Error scraping Buddy4Study: {e}")
        
        return self.scholarships

    def scrape_scholarship_india(self):
        """Scrape from National Scholarship Portal (generic structure)"""
        # Note: NSP requires login, this is a template
        scholarships = [
            {
                'name': 'Post Matric Scholarship for SC Students',
                'amount': 'Up to ₹1,00,000',
                'deadline': '30th November 2026',
                'eligibility': 'SC category, income below 2.5 lakh, studying post-matric',
                'source': 'National Scholarship Portal',
                'category': 'SC',
                'link': 'https://scholarships.gov.in'
            },
            {
                'name': 'Central Sector Scheme of Scholarship',
                'amount': '₹12,000 - ₹20,000 per year',
                'deadline': '31st December 2026',
                'eligibility': 'Top 20 percentile in 12th, family income below 8 lakh',
                'source': 'National Scholarship Portal',
                'category': 'Merit',
                'link': 'https://scholarships.gov.in'
            },
            {
                'name': 'Pre Matric Scholarship for OBC Students',
                'amount': 'Up to ₹6,000',
                'deadline': '30th November 2026',
                'eligibility': 'OBC category, income below 1.5 lakh, class 1-10',
                'source': 'National Scholarship Portal',
                'category': 'OBC',
                'link': 'https://scholarships.gov.in'
            },
            {
                'name': 'Merit-cum-Means Scholarship for Minorities',
                'amount': 'Up to ₹30,000',
                'deadline': '31st October 2026',
                'eligibility': 'Minority community, 50% marks in previous exam, income below 2.5 lakh',
                'source': 'Ministry of Minority Affairs',
                'category': 'Minority',
                'link': 'https://scholarships.gov.in'
            },
            {
                'name': 'Prime Minister\'s Scholarship Scheme',
                'amount': '₹36,000 - ₹60,000',
                'deadline': '31st December 2026',
                'eligibility': 'Wards of Ex-Servicemen/Ex-Coast Guard, pursuing professional degree',
                'source': 'Ministry of Defence',
                'category': 'Defence',
                'link': 'https://scholarships.gov.in'
            }
        ]
        
        for s in scholarships:
            s['scraped_at'] = datetime.now().isoformat()
            self.scholarships.append(s)
        
        return scholarships

    def scrape_all(self):
        """Scrape from all sources"""
        logger.info("Starting scholarship scraping...")
        self.scrape_buddy4study()
        self.scrape_scholarship_india()
        logger.info(f"Scraped {len(self.scholarships)} scholarships")
        return self.scholarships

    def save_to_json(self, filename='scholarships.json'):
        """Save scraped scholarships to JSON file"""
        with open(filename, 'w', encoding='utf-8') as f:
            json.dump(self.scholarships, f, indent=2, ensure_ascii=False)
        logger.info(f"Saved scholarships to {filename}")

    def get_scholarships_by_category(self, category):
        """Filter scholarships by category"""
        return [s for s in self.scholarships if s.get('category', '').lower() == category.lower()]


if __name__ == "__main__":
    scraper = ScholarshipScraper()
    scholarships = scraper.scrape_all()
    scraper.save_to_json()
    print(f"Scraped {len(scholarships)} scholarships")
    for s in scholarships[:5]:
        print(f"- {s['name']}: {s['amount']}")