# -*- coding: utf-8 -*- """ Created on Thu Jul 30 22:36:34 2026 @author: muhuri """ # ============================================================================= # Program Name : Program2_YouTube_API_Search_and_Classify.py # Author : ChatGPT (based on Pradip Muhuri's Program1A) # Purpose : Search YouTube Data API v3 and classify videos related to # violence against religious minorities in Bangladesh. # # NOTE: # - Requires config.py containing API_KEY # - Requires Keywords_list.py containing: # CATEGORY_MAP and CATEGORY_WEIGHTS # - Requires Keywords_list_v3_SearchThemes.py containing SEARCH_QUERIES # ============================================================================= from datetime import datetime import logging import os import pandas as pd from googleapiclient.discovery import build from googleapiclient.errors import HttpError from config import API_KEY from Keywords_list import ( CATEGORY_MAP, CATEGORY_WEIGHTS, ) from Keywords_list_v3_SearchThemes import SEARCH_QUERIES OUTPUT_FOLDER = r"C:\Explore\PythonBDM_YouTube\Output" os.makedirs(OUTPUT_FOLDER, exist_ok=True) youtube = build( "youtube", "v3", developerKey=API_KEY, cache_discovery=False ) # ----------------------------------------------------------------------------- # Classify video text # ----------------------------------------------------------------------------- def classify(text): """ Classify a video's title and description using CATEGORY_MAP. Returns: category: semicolon-separated category names matched_keywords: semicolon-separated matched keywords score: weighted classification score """ text = text.lower() categories = [] matched_keywords = [] score = 0 for category, terms in CATEGORY_MAP.items(): hits = [ term for term in terms if term.lower() in text ] if hits: categories.append(category) matched_keywords.extend(hits) score += ( CATEGORY_WEIGHTS.get(category, 1) * len(hits) ) if not categories: categories = ["Other"] return ( "; ".join(categories), "; ".join(sorted(set(matched_keywords))), score, ) # ----------------------------------------------------------------------------- # Get user input for max results # ----------------------------------------------------------------------------- while True: x = input("Maximum results per query (1-50 or max): ").strip().lower() if x == "max": RETRIEVE_ALL = True MAX_RESULTS = 50 break try: MAX_RESULTS = int(x) if 1 <= MAX_RESULTS <= 50: RETRIEVE_ALL = False break except ValueError: pass print("Enter 1-50 or max.") # ----------------------------------------------------------------------------- # Publication year filter # ----------------------------------------------------------------------------- DATE_WINDOWS = { "1": ("Jan–Jun 2024", "2024-01-01T00:00:00Z", "2024-07-01T00:00:00Z"), "2": ("Jul–Dec 2024", "2024-07-01T00:00:00Z", "2025-01-01T00:00:00Z"), "3": ("Jan–Jun 2025", "2025-01-01T00:00:00Z", "2025-07-01T00:00:00Z"), "4": ("Jul–Dec 2025", "2025-07-01T00:00:00Z", "2026-01-01T00:00:00Z"), "5": ("Jan–Jun 2026", "2026-01-01T00:00:00Z", "2026-07-01T00:00:00Z"), "6": ("Jul–Dec 2026", "2026-07-01T00:00:00Z", "2027-01-01T00:00:00Z"), "7": ("All of 2024", "2024-01-01T00:00:00Z", "2025-01-01T00:00:00Z"), "8": ("All of 2025", "2025-01-01T00:00:00Z", "2026-01-01T00:00:00Z"), "9": ("All of 2026", "2026-01-01T00:00:00Z", "2027-01-01T00:00:00Z"), "10": ("2023 and Earlier", None, "2024-01-01T00:00:00Z"), "11": ("All Years", None, None), } while True: print("\nPublication period to retrieve") print(" 1. Jan–Jun 2024") print(" 2. Jul–Dec 2024") print(" 3. Jan–Jun 2025") print(" 4. Jul–Dec 2025") print(" 5. Jan–Jun 2026") print(" 6. Jul–Dec 2026") print(" 7. All of 2024") print(" 8. All of 2025") print(" 9. All of 2026") print("10. 2023 and Earlier") print("11. All Years") choice = input("Select option [11]: ").strip() or "11" if choice in DATE_WINDOWS: YEAR_LABEL, PUBLISHED_AFTER, PUBLISHED_BEFORE = DATE_WINDOWS[choice] break print("Invalid option.") #---------------------------------------------------------- # Display selected publication period #---------------------------------------------------------- print("\n" + "=" * 70) print(f"Publication period: {YEAR_LABEL}") print("=" * 70) #---------------------------------------------------------- # Create filenames based on selected publication period #---------------------------------------------------------- safe_label = ( YEAR_LABEL .replace("–", "-") .replace("—", "-") .replace(" ", "_") .replace("/", "-") ) timestamp = datetime.now().strftime("%Y%m%d_%H%M%S") csv_file = os.path.join( OUTPUT_FOLDER, f"YTV_{safe_label}.csv" ) html_file = os.path.join( OUTPUT_FOLDER, f"YTV_{safe_label}.html" ) log_file = os.path.join( OUTPUT_FOLDER, f"YTV_{safe_label}_{timestamp}.log" ) logging.basicConfig( level=logging.INFO, format="%(message)s", handlers=[ logging.FileHandler(log_file,encoding="utf-8"), logging.StreamHandler() ] ) logging.info("=" * 70) logging.info("YouTube Data API Video Search") logging.info("=" * 70) logging.info(f"Publication period : {YEAR_LABEL}") logging.info( f"Maximum results : " f"{'All available' if RETRIEVE_ALL else MAX_RESULTS}" ) logging.info(f"Search queries : {len(SEARCH_QUERIES)}") logging.info("=" * 70) query_summary = [] # ----------------------------------------------------------------------------- # Main search and classification loop # ----------------------------------------------------------------------------- records = [] quota_exceeded = False for query in SEARCH_QUERIES: logging.info(f"Searching: {query}") page_token = None page_no = 0 videos_found = 0 while True: try: response = youtube.search().list( q=query, part="snippet", type="video", maxResults=MAX_RESULTS, order="date", relevanceLanguage="en", pageToken=page_token, publishedAfter=PUBLISHED_AFTER, publishedBefore=PUBLISHED_BEFORE, ).execute() page_no += 1 page_count=len(response.get("items", [])) logging.info(f" Page {page_no}: {page_count} videos") for item in response.get("items", []): s = item["snippet"] vid = item["id"]["videoId"] title = s["title"] description = s["description"] published_date = pd.to_datetime( s["publishedAt"], utc=True, errors="coerce" ) publication_year = ( published_date.year if pd.notna(published_date) else None ) text = f"{title or ''} {description or ''}" cat, keys, score = classify(text) records.append( { "SearchQuery": query, "Category": cat, "Score": score, "MatchedKeywords": keys, "VideoID": vid, "Title": title, "Channel": s["channelTitle"], "PublishedDate": published_date, "PublicationYear": publication_year, "Description": description, "URL": f"https://www.youtube.com/watch?v={vid}", } ) if not RETRIEVE_ALL: break page_token = response.get("nextPageToken") if not page_token: break except HttpError as e: if e.resp.status == 429: logging.info("\n" + "=" * 70) logging.info("YouTube API daily quota has been exceeded.") logging.info(f"Last attempted query: {query}") if records: logging.info("Processing the videos already collected...") else: logging.info( "No videos were collected before the quota was exceeded." ) logging.info("=" * 70) quota_exceeded = True break # exit pagination loop for this query else: raise # re-raise other HTTP errors query_summary.append((query, page_no, videos_found)) if quota_exceeded: break # stop all further queries # ----------------------------------------------------------------------------- # Post-processing and output # ----------------------------------------------------------------------------- df = pd.DataFrame(records) if len(df) == 0: logging.info("No videos retrieved.") raise SystemExit df = df.drop_duplicates(subset="VideoID") df = df.sort_values( ["Score", "PublishedDate"], ascending=[False, False] ) df = df.drop(columns=["VideoID"], errors="ignore") df.to_csv(csv_file, index=False, encoding="utf-8-sig") html = df.copy() #---------------------------------------------------------- # Remove videos classified only as "Other" #---------------------------------------------------------- html = html[ html["Category"].str.strip().str.lower() != "other" ].copy() # Score is used for internal ranking but is not displayed in the HTML report. html = html.drop(columns=["Score"], errors="ignore") html["PublishedDate"] = html["PublishedDate"].dt.strftime("%B %d, %Y") html["URL"] = html["URL"].apply( lambda u: f'{u}' ) summary = f"""
Generated: {datetime.now():%B %d, %Y %I:%M %p}
""" page = f"""
Search Queries: {len(SEARCH_QUERIES)}
Publication Period: {YEAR_LABEL}
Total Videos: {len(html):,}