# -*- coding: utf-8 -*- """ Created on Sun Jul 19 22:23:05 2026 @author: muhuri """ import pandas as pd import re from datetime import datetime #=========================================================== # Input and output files #=========================================================== input_file = r"C:\Explore\BDMDataTables\HTMLOutput\BD_Minority_Victims_2025_dashboard.html" output_file = r"C:\Explore\BDMDataTables\CSVText\BD_Minority_Victims_2025_incident.csv" #----------------------------------------------------------- # Quality Control output files #----------------------------------------------------------- qc_text_file = ( r"C:\Explore\BDMDataTables\CSVText" r"\BDM_Quality_Control_Summary.txt" ) qc_freq_file = ( r"C:\Explore\BDMDataTables\CSVText" r"\Incident_Category_Frequencies.csv" ) #=========================================================== # Keyword lists #=========================================================== BLASPHEMY_TERMS = [ "blasphemy", "blasphemous", "alleged blasphemy", "prophet muhammad", "holy prophet", "alleged prophet", "pbuh", "offensive islam remark", "offensive remark", "facebook post", "facebook status", "facebook comment", "facebook profile", "facebook account", "social media post", "religious post", "religious sentiments", "religious feelings", "quran", "quran desecration", "desecration", "desecrated", "alleged 'offensive' islam remark", "religious insult", "religious incitement", "false allegations of anti-Islam comments", "posted inflammatory content", "allegedly insulting the Prophet", "insulted islam", "insulting islam", "kaaba sharif insult on facebook", "insult the prophet", "insulted the prophet" ] #=========================================================== # Killings #=========================================================== KILLING_TERMS = [ "shot dead", "shot to death", "shot and killed", "kill", "killed", "killing", "murder", "murdered", "slain", "gunned down", "decomposed", "body found", "dead body", "recovered body", "recovered the body", "body recovered", "bodies recovered", "found dead", "behead", "beheaded", "hacked", "hacked to death", "chopped to death", "beat to death", "beaten to death", "strangled", "strangled to death", "died en route to the hospital", "custodial deaths", "throat slit", "hanging body", "hung himself", "hung herself", "fatal", "lynched", "butchered", "burned alive", "burnt alive" ] #=========================================================== # Missing / Abduction #=========================================================== MISSING_ABDUCTED_TERMS = [ # Missing "missing", "went missing", "reported missing", "has been missing", "still missing", "missing since", "disappeared", "disappearance", "untraceable", "not found", "whereabouts unknown", "never returned", "never arrived", "last seen", "lost contact", # Abduction "abduction", "abducted", "abducting", "kidnapped", "kidnapping", "kidnap", "forcibly taken", "forcefully taken", "taken away", "taken from", "dragged away", "picked up", "held captive", "held hostage", "confined", "detained against", "unlawfully detained", # Recovery "rescued", "recovered safely", "returned home" ] #=========================================================== # Rape #=========================================================== RAPE_TERMS = [ "gang rape", "gang-rape", "attempted gang rape", "attempted rape", "attempt to rape", "rape attempt", "raped", "raping", "rape" ] #=========================================================== # General Violence #=========================================================== GENERAL_VIOLENCE_TERMS = [ # General violence "attack", "attacked", "assault", "assaulted", "beaten", "beating", "stabbed", "stabbing", "injured", "injury", "harass", "harassed", "harassment", "threat", "threatened", "intimidated", "intimidation", "mob attack", "communal attack", "communal violence", "communal", "riot", "rioting", "false case", "false cases", "extortion", "extortionists", "forced conversion", "forced to convert", "boycott", "outrage among minorities", "removed from government service", "hurt over renaming Hindu luminaries' names", "obstructed a century-old indigenous cemetery burial", "forced Hindu school headmaster", "cultural outrage", "threatening forced removal", "police obstruction", "fascist ally", "bullets hit", "sexual assault", "sexual abuse", "sexual harassment", "sexual violence", "sexual exploitation", "sexual torture", "sexual misconduct", "indecent assault", "indecent behavior", "indecent proposal", "molest", "molested", "molestation", "eve teasing", "harassed sexually", "forced marriage", "forced to marry", "attempted forced marriage", "abducted for marriage", "kidnapped for marriage", "forced conversion and marriage", "outraging modesty" ] #=========================================================== # Religious Violence #=========================================================== RELIGIOUS_VIOLENCE_TERMS = [ # Religious institutions "temple", "mandir", "church", "monastery", "ashram", "pagoda", "cremation ground", "graveyard", "priest", "monk", "pastor", "purohit", "idol", "idols", "deity", "deities", "iskcon", "attempting to seize", "Cocktail explosions targeted St. Mary's Cathedral", # Common phrases "religious discrimination", "without a show-cause notice", "religious persecution", "religious intolerance", "religious hatred", "demanding release", "leave the village", "forced to flee" ] #=========================================================== # Property Damage #=========================================================== PROPERTY_DAMAGE_TERMS = [ # Fire / Arson "arson", "set fire", "burn", "burned", "burnt", "burning", "torched", "torch", "house burned", "shop burned", "temple burned", "business burned", "burned alive", # Theft / Robbery "theft", "stolen", "steal", "stole", "robbery", "robbed", "loot", "looted", "looting", "dacoity", "burglary", "burglarized", "extorting large sums and harassing", "property dispute", "throttled grocer", "Opponents harvested", "Hindu land via fake deeds", # Vandalism / Damage "vandal", "vandalized", "vandalism", "defaced", "destroy", "destroyed", "damage", "damaged", "demolish", "demolished", "ransack", "ransacked", "raided", "broken", "broke", "armed attacks and home demolitions", # Property-related phrases "house damaged", "shop damaged", "business establishment", "business premises", "house looted", "shop looted", "temple vandalized", "idol vandalized", "idol broken", "crops destroyed", "fish enclosure", "orchard", "fruit trees", "cash stolen", "gold ornaments", "valuable documents", "motorcycle stolen", "vehicle damaged", "vehicle burned", "boundary wall", "furniture destroyed" ] #=========================================================== # Land Grabbing #=========================================================== LAND_GRABBING_TERMS = [ # General "land grabbing", "land grab", "land grabber", "Land grabbers", "grabbed land", "grab land", "grabbed", "grabbing", "grab", # Possession / Occupation "occupied", "occupation", "occupied land", "illegal occupation", "illegal possession", "forcibly occupied", "forcefully occupied", "encroach", "encroached", "encroachment", "seized land", # Eviction "evict", "evicted", "eviction", "eviction notice", # Land disputes "land dispute", "disputed land", "ancestral land", "family land", "homestead", "homestead land", "agricultural land", # Lease-related "leased land", "lease cancellation", "lease cancelled", "lease canceled", "land protection committee", # Legal / Administrative "land office", "land registry", "registry office", "land record", "mutation", "deed", "forged deed", "fake deed", "land document", "property document", # Government land "khas land", "hill district council", "rubber plantation", "rubber cultivation", # Miscellaneous "boundary wall", "fence removed", "fencing", "land owner", "land ownership", "property ownership" ] OTHER_TERMS = [ "committed suicide", "suicide", "suicidal" ] # Keyword lists CATEGORY_KEYWORDS = { "Blasphemy-Related Attacks": BLASPHEMY_TERMS, "Killings": KILLING_TERMS, "Rape": RAPE_TERMS, "Missing-Abducted": MISSING_ABDUCTED_TERMS, "Violence Against Persons": GENERAL_VIOLENCE_TERMS, "Attacks on Religious Sites": RELIGIOUS_VIOLENCE_TERMS, "Property Damage": PROPERTY_DAMAGE_TERMS, "Land Grabbing": LAND_GRABBING_TERMS, "Other": OTHER_TERMS } #=========================================================== # Initialize keyword match counters #=========================================================== keyword_match_counts = { category: {keyword: 0 for keyword in keywords} for category, keywords in CATEGORY_KEYWORDS.items() } #=========================================================== # Classifier #=========================================================== def classify_incident(text): if pd.isna(text) or str(text).strip() == "": return "Missing Description" text = str(text).lower() categories = [] for category, keywords in CATEGORY_KEYWORDS.items(): matched = False for keyword in keywords: if re.search( r"\b" + re.escape(keyword.lower()) + r"\b", text ): keyword_match_counts[category][keyword] += 1 matched = True if matched: categories.append(category) categories = list(dict.fromkeys(categories)) return "; ".join(categories) # Read HTML tables = pd.read_html(input_file) df = tables[0] # Create incident_category df["incident_category"] = ( df["Description of Incident"] .apply(classify_incident) ) #----------------------------------------------------------- # Print to console and save to text file #----------------------------------------------------------- def print_and_save(message, file): print(message) file.write(message + "\n") #=========================================================== # Quality Control Summary #=========================================================== missing_description = ( df["incident_category"] == "Missing Description" ).sum() unclassified = ( (df["incident_category"] == "") ).sum() classified = ( len(df) - missing_description - unclassified ) classification_rate = ( classified / (len(df) - missing_description) * 100 if len(df) > missing_description else 0 ) multiple = df["incident_category"].str.contains(";", regex=False).sum() single = classified - multiple frequency_table = ( df["incident_category"] .value_counts(dropna=False) .sort_index() ) #------------------------------------------------------- # Incident counts by category #------------------------------------------------------- category_incident_counts = {} for category in CATEGORY_KEYWORDS: category_incident_counts[category] = ( df["incident_category"] .str.contains(category, regex=False) .sum() ) with open(qc_text_file, "w", encoding="utf-8") as f: print_and_save("=" * 60, f) print_and_save("INCIDENT CATEGORY SUMMARY", f) print_and_save("=" * 60, f) #------------------------------------------------------- # Run information #------------------------------------------------------- print_and_save( f"Run date and time: {datetime.now():%Y-%m-%d %H:%M:%S}", f ) print_and_save("Input file:", f) print_and_save(input_file, f) print_and_save("", f) print_and_save("Output file:", f) print_and_save(output_file, f) print_and_save("", f) #------------------------------------------------------- # Data quality #------------------------------------------------------- print_and_save(f"Total incidents: {len(df):,}", f) print_and_save(f"Incidents with descriptions: {len(df)-missing_description:,}", f) print_and_save(f"Missing descriptions: {missing_description:,}", f) print_and_save("", f) #------------------------------------------------------- # Classification results #------------------------------------------------------- print_and_save(f"Classified incidents: {classified:,}", f) print_and_save(f"Unclassified incidents: {unclassified:,}", f) print_and_save(f"Classification rate: {classification_rate:.1f}%", f) print_and_save("", f) #------------------------------------------------------- # Category assignment #------------------------------------------------------- print_and_save(f"Single-category incidents: {single:,}", f) print_and_save(f"Multiple-category incidents: {multiple:,}", f) print_and_save("", f) print_and_save("Frequency of incident categories:", f) print(frequency_table) # Display in Python console print_and_save(frequency_table.to_string(), f) # Write to QC file #------------------------------------------------------- # Keyword lists used for classification #------------------------------------------------------- print_and_save("", f) print_and_save("=" * 60, f) print_and_save("KEYWORD LISTS BY INCIDENT CATEGORY", f) print_and_save("=" * 60, f) total_keywords = 0 for category, keywords in CATEGORY_KEYWORDS.items(): print_and_save("", f) print_and_save(category, f) print_and_save("-" * len(category), f) for i, keyword in enumerate(keywords, start=1): print_and_save(f"{i:3d}. {keyword}", f) print_and_save(f"Total keywords in category: {len(keywords)}", f) total_keywords += len(keywords) #------------------------------------------------------- # Overall totals (print once) #------------------------------------------------------- print_and_save("", f) print_and_save(f"Total incident categories: {len(CATEGORY_KEYWORDS)}", f) print_and_save(f"Total classification keywords: {total_keywords}", f) print_and_save("", f) print_and_save("=" * 60, f) #------------------------------------------------------- # Keyword match summary #------------------------------------------------------- print_and_save("", f) print_and_save("=" * 60, f) print_and_save("KEYWORD MATCH SUMMARY BY CATEGORY", f) print_and_save("=" * 60, f) for category, keyword_counts in keyword_match_counts.items(): print_and_save("", f) print_and_save(category, f) print_and_save("-" * len(category), f) print_and_save( f"Incidents classified: {category_incident_counts[category]}", f ) print_and_save("", f) matched = 0 for keyword, count in sorted( keyword_counts.items(), key=lambda x: (-x[1], x[0].lower()) ): if count > 0: print_and_save( f"{keyword:<40} {count:>5}", f ) matched += 1 total_matches = sum(keyword_counts.values()) print_and_save("", f) print_and_save( f"Distinct keywords matched: {matched}", f ) print_and_save( f"Distinct keywords never matched: {len(keyword_counts) - matched}", f ) print_and_save( f"Total keyword matches: {total_matches}", f ) print_and_save( f"Keywords in category: {len(keyword_counts)}", f ) print_and_save("", f) #----------------------------------------------------------- # Save category frequencies to CSV #----------------------------------------------------------- freq_df = ( frequency_table .rename("count") .rename_axis("incident_category") .reset_index() ) freq_df.to_csv( qc_freq_file, index=False, encoding="utf-8-sig" ) #----------------------------------------------------------- # Completion messages #----------------------------------------------------------- print("\nProcessing completed successfully.") print(f"\nQuality Control Summary written to:\n{qc_text_file}") print(f"\nCategory Frequency Table written to:\n{qc_freq_file}") print(f"\nCSV data file written to:\n{output_file}") print(f"{len(df):,} observations written.") #=========================================================== # Create indicator variables #=========================================================== indicator_variables = [] for category in CATEGORY_KEYWORDS: # Create a valid Pandas column name column = ( category.lower() .replace(" ", "_") .replace("-", "_") .replace("/", "_") ) indicator_variables.append(column) # Create 0/1 indicator variable df[column] = ( df["incident_category"] .str.contains(category, regex=False) .astype(int) ) #=========================================================== # Convert DataFrame column names to valid SAS variable names #=========================================================== def make_sas_name(name): """Convert a column name to a valid SAS variable name.""" # Replace non-alphanumeric characters with underscores name = re.sub(r'[^A-Za-z0-9_]', '_', str(name)) # Collapse multiple consecutive underscores name = re.sub(r'_+', '_', name) # Remove leading and trailing underscores name = name.strip('_') # SAS variable names cannot begin with a digit if name and name[0].isdigit(): name = "_" + name # SAS variable names are limited to 32 characters return name[:32] # Rename all DataFrame columns df.columns = [make_sas_name(col) for col in df.columns] #=========================================================== # Convert DataFrame column names to valid SAS variable names #=========================================================== def make_sas_name(name): """Convert a column name to a valid SAS variable name.""" # Replace non-alphanumeric characters with underscores name = re.sub(r'[^A-Za-z0-9_]', '_', str(name)) # Collapse multiple consecutive underscores name = re.sub(r'_+', '_', name) # Remove leading and trailing underscores name = name.strip('_') # SAS variable names cannot begin with a digit if name and name[0].isdigit(): name = "_" + name # SAS variable names are limited to 32 characters return name[:32] # Rename all DataFrame columns df.columns = [make_sas_name(col) for col in df.columns] #----------------------------------------------------------- # Save output CSV file #----------------------------------------------------------- print("\nColumns before writing CSV:") for c in df.columns: print(c) df.to_csv( output_file, index=False, encoding="utf-8-sig" ) #----------------------------------------------------------- # List indicator variables and their frequencies #----------------------------------------------------------- print("\nIndicator variables created:") print(f"Total indicator variables: {len(indicator_variables)}") for var in indicator_variables: print(f" {var:<35} {df[var].sum():>5}") #----------------------------------------------------------- # Completion messages #----------------------------------------------------------- print("\nProcessing completed successfully.") print(f"\nQuality Control Summary written to:\n{qc_text_file}") print(f"\nCategory Frequency Table written to:\n{qc_freq_file}") print(f"\nCSV data file written to:\n{output_file}") print(f"{len(df):,} observations written.")