commit f7410dcfe4a690c9bd32ffbd4db9a50b614c803f Author: Warrenww Date: Wed Sep 30 21:05:53 2026 +0800 init commit diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..4ca893a --- /dev/null +++ b/.gitignore @@ -0,0 +1,17 @@ +analysis/ +bot_analysis/ +**/*.log +cache/ +report/ +*.mmdb + +# Python +venv/ +.venv/ +__pycache__/ +*.pyc +.env + +# Editor / OS +.vscode/ +.DS_Store diff --git a/README.md b/README.md new file mode 100644 index 0000000..2694c61 --- /dev/null +++ b/README.md @@ -0,0 +1,322 @@ +# IP Geo Visitor Analysis Toolkit + +A set of three Python scripts to process large-scale web visitor logs from MySQL, enrich them with geographic data, detect bots, and generate monthly analytics reports. + +--- + +## Overview + +| Script | Purpose | +|---|---| +| `ip_geo_report.py` | Read visitor records from MySQL, resolve IPs to countries, export daily visit counts per country | +| `analyze.py` | Read the CSV output from `ip_geo_report.py` and generate monthly visit stats with MoM changes and charts | +| `bot_analyzer.py` | Select records from a date range and score each visit to classify as human, bot, or suspicious | + +--- + +## Requirements + +- Python 3.8+ +- MySQL database with visitor records +- ipinfo `bundle_location_lite.mmdb` (free, download from [ipinfo.io](https://ipinfo.io/account/data-downloads)) + +--- + +## Installation + +```bash +# Clone or copy scripts to your project folder +mkdir visitor-analysis && cd visitor-analysis + +# Create virtual environment +python3 -m venv venv +source venv/bin/activate # Linux / macOS +# venv\Scripts\activate # Windows + +# Install dependencies +pip install mysql-connector-python maxminddb matplotlib +``` + +--- + +## Project Structure + +``` +visitor-analysis/ +├── venv/ ← virtual environment (do not commit) +├── bundle_location_lite.mmdb ← ipinfo offline GeoIP database +├── ip_geo_report.py ← Script 1: IP → Country report +├── analyze.py ← Script 2: Monthly analytics + charts +├── bot_analyzer.py ← Script 3: Bot detection +├── requirements.txt +├── README.md +├── cache/ ← Auto-created: LRU disk cache + checkpoints +└── report/ ← Auto-created: CSV output from Script 1 +``` + +--- + +## Script 1 — `ip_geo_report.py` + +Reads all visitor records from MySQL in batches, resolves each IP address to a country using the ipinfo mmdb database, and produces a CSV report of daily visit counts per country. + +### Features + +- Batch processing with cursor-based pagination (handles 100M+ rows) +- Two-level cache: LRU memory (100k IPs) + disk shelve spillover +- Resume from checkpoint on crash or interruption +- ipinfo mmdb offline lookup → ipinfo API fallback +- Bot/attack filtering before GeoIP lookup + +### Configuration + +Edit the top section of `ip_geo_report.py`: + +```python +DB_CONFIG = { + "host": "localhost", + "user": "your_user", + "password": "your_password", + "database": "your_database", +} + +TABLE_NAME = "visitors" # your table name +IP_COLUMN = "ip" # IP address column +DATE_COLUMN = "visit_date" # DATE or DATETIME column +UA_COLUMN = "user_agent" # set None if not available +PATH_COLUMN = "path" # set None if not available +GEOIP_DB_PATH = "./bundle_location_lite.mmdb" +IPINFO_TOKEN = "" # optional API fallback token +BATCH_SIZE = 50_000 +LRU_MAX_SIZE = 100_000 +``` + +### Usage + +```bash +source venv/bin/activate +python3 ip_geo_report.py +``` + +If interrupted, simply re-run — it resumes from the last checkpoint automatically. + +### Output + +``` +report/ +├── report_YYYY-MM-DD.csv ← Country, Date, Visit Count +└── summary_YYYY-MM-DD.csv ← Country, Total Visits (sorted) +``` + +--- + +## Script 2 — `analyze.py` + +Reads the CSV output from `ip_geo_report.py` and generates monthly analytics including MoM (Month-over-Month) percentage changes and a 4-panel visualization chart. + +### Features + +- Auto-detects latest `report_*.csv` in `./report/` +- Handles multiple date formats (`YYYY-MM-DD`, `YYYYMMDD`, `YYYY/MM/DD`) +- Monthly visit totals with MoM % change +- Monthly unique country counts with MoM % change +- Top N countries per month +- 4-panel PNG chart (bar charts + MoM line + stacked country chart) + +### Usage + +```bash +# Auto-detect latest report +python3 analyze.py + +# Specify input file +python3 analyze.py --input ./report/report_2026-09-27.csv + +# Custom output dir and top N countries +python3 analyze.py \ + --input ./report/report_2026-09-27.csv \ + --output ./analysis \ + --top 10 +``` + +### Arguments + +| Argument | Default | Description | +|---|---|---| +| `--input`, `-i` | latest in `./report/` | Path to input CSV | +| `--output`, `-o` | `./analysis/` | Output directory | +| `--top`, `-n` | `10` | Top N countries per month | + +### Output + +``` +analysis/ +├── monthly_summary_YYYYMMDD.csv ← Month, Visits, MoM%, Countries, MoM% +├── top_countries_YYYYMMDD.csv ← Month, Rank, Country, Visits, % +└── analysis_chart_YYYYMMDD.png ← 4-panel visualization chart +``` + +### Chart Panels + +| Panel | Content | +|---|---| +| Top-left | Monthly total visits bar + MoM % line | +| Top-right | Monthly unique countries bar + MoM % line | +| Bottom-left | MoM % comparison line chart with +/- shading | +| Bottom-right | Stacked bar — top 5 countries (last 6 months) | + +--- + +## Script 3 — `bot_analyzer.py` + +Selects visitor records from a MySQL date range and scores each visit across multiple signals to classify it as human, bot, or suspicious. + +### Features + +- Two-pass processing: frequency counting then per-record scoring +- Windowed frequency detection (per day, per hour, traffic ratio) +- Pattern-based signals: user agent, request path, IP range +- Outputs three separate CSVs: human / bot / suspicious +- Detailed summary with top high-frequency IPs + +### Scoring Rules + +| Signal | Score | Trigger | +|---|---|---| +| Invalid / Private IP | +10 | Reserved IP ranges (RFC 1918 etc.) | +| Bot User Agent | +8 | googlebot, scrapers, crawlers | +| High Freq Per Hour | +8 | IP hits > threshold in single hour | +| Attack Path | +9 | SQLi, shells, scanner paths | +| Tool User Agent | +7 | curl, wget, python, selenium | +| No User Agent | +7 | Empty or missing UA | +| High Freq Per Day | +7 | IP hits > threshold in single day | +| Old IE Browser | +5 | IE 6/7/8 (commonly spoofed) | +| High Traffic Ratio | +5 | IP > 1% of all traffic in range | +| Suspicious Combo | +4 | Old UA + attack path together | +| Empty Path | +3 | Request path is blank or just `/` | + +**Classification thresholds:** + +| Score | Classification | +|---|---| +| 0 – 2 | ✅ Human | +| 3 – 4 | ⚠️ Suspicious | +| 5+ | 🤖 Bot | + +### Configuration + +```python +HIGH_FREQ_PER_DAY = 100 # flag if IP hits > N times in a single day +HIGH_FREQ_PER_HOUR = 30 # flag if IP hits > N times in a single hour +HIGH_FREQ_TOTAL_RATIO = 0.01 # flag if IP > 1% of total traffic in range +BOT_SCORE_THRESHOLD = 5 # score >= this = bot +SUSPICIOUS_THRESHOLD = 3 # score >= this = suspicious +``` + +### Usage + +```bash +# Basic usage +python3 bot_analyzer.py --start 2026-01-01 --end 2026-09-30 + +# Custom thresholds +python3 bot_analyzer.py \ + --start 2026-01-01 \ + --end 2026-09-30 \ + --bot-threshold 5 \ + --freq-per-day 100 \ + --freq-per-hour 30 \ + --freq-ratio 0.01 \ + --output ./bot_analysis +``` + +### Arguments + +| Argument | Default | Description | +|---|---|---| +| `--start`, `-s` | required | Start date `YYYY-MM-DD` | +| `--end`, `-e` | required | End date `YYYY-MM-DD` | +| `--output`, `-o` | `./bot_analysis/` | Output directory | +| `--bot-threshold` | `5` | Minimum score to classify as bot | +| `--freq-per-day` | `100` | Max hits per day before flagging | +| `--freq-per-hour` | `30` | Max hits per hour before flagging | +| `--freq-ratio` | `0.01` | Max fraction of total traffic per IP | + +### Output + +``` +bot_analysis/ +├── human_DATERANGE_TIMESTAMP.csv ← Confirmed human visits +├── bot_DATERANGE_TIMESTAMP.csv ← Confirmed bot visits + score + reasons +├── suspicious_DATERANGE_TIMESTAMP.csv ← Borderline visits for manual review +└── summary_DATERANGE_TIMESTAMP.csv ← Stats, reason counts, top IPs +``` + +--- + +## Recommended Workflow + +``` +1. Run ip_geo_report.py + → Processes all historical data + → Outputs report/report_YYYY-MM-DD.csv + +2. Run analyze.py + → Reads the report CSV + → Outputs monthly stats + chart + +3. Run bot_analyzer.py --start YYYY-MM-DD --end YYYY-MM-DD + → Focuses on a specific time window + → Outputs human / bot / suspicious CSVs +``` + +--- + +## GeoIP Database Setup + +1. Register for a free account at [https://ipinfo.io/signup](https://ipinfo.io/signup) +2. Go to [https://ipinfo.io/account/data-downloads](https://ipinfo.io/account/data-downloads) +3. Download **`bundle_location_lite.mmdb`** +4. Place it in the project root directory + +The database provides: `country`, `country_code`, `continent`, `asn`, `as_name`, `as_domain` + +--- + +## Database Table Requirements + +Your MySQL visitor table should have at minimum: + +| Column | Type | Required | Description | +|---|---|---|---| +| `id` | INT | ✅ Yes | Primary key for cursor pagination | +| `ip` | VARCHAR | ✅ Yes | Visitor IP address | +| `visit_date` | DATE / DATETIME | ✅ Yes | Visit timestamp | +| `user_agent` | VARCHAR | ⚠️ Optional | Browser user agent string | +| `path` | VARCHAR | ⚠️ Optional | Request path / URL | + +> Set `UA_COLUMN = None` and `PATH_COLUMN = None` in scripts if those columns do not exist. + +--- + +## requirements.txt + +``` +mysql-connector-python +maxminddb +matplotlib +``` + +Install with: +```bash +pip install -r requirements.txt +``` + +--- + +## Notes + +- All scripts support resuming interrupted runs (Script 1 via checkpoint file, Script 3 via re-running with same date range) +- Cache files in `./cache/` persist between runs to avoid redundant GeoIP lookups +- Partial report data is saved every 10 batches to survive crashes +- Scripts are designed for Python 3.8+ and tested on Ubuntu 20.04 \ No newline at end of file diff --git a/analyze.py b/analyze.py new file mode 100644 index 0000000..3f3fe48 --- /dev/null +++ b/analyze.py @@ -0,0 +1,549 @@ +""" +Report Analyzer +Reads the CSV output from ip_geo_report.py and generates: + 1. Monthly total visit count with MoM change + 2. Monthly unique country count with MoM change + 3. Top countries per month + 4. Export to CSV + print to console +""" + +import os +import csv +import argparse +from datetime import datetime +from collections import defaultdict +from typing import Dict, List, Tuple, Optional + +try: + import matplotlib.pyplot as plt + import matplotlib.ticker as mticker + from matplotlib.gridspec import GridSpec + HAS_MATPLOTLIB = True +except ImportError: + HAS_MATPLOTLIB = False + +# ─── Config ─────────────────────────────────────────────────────────────────── + +REPORT_DIR = "./report" +OUTPUT_DIR = "./analysis" +TOP_N_COUNTRIES = 10 # top N countries to show per month + +# ─── Data Loader ────────────────────────────────────────────────────────────── + +def load_report(path: str) -> List[dict]: + """ + Load report CSV generated by ip_geo_report.py + Expected columns: Country, Date, Visit Count + """ + rows = [] + with open(path, newline="", encoding="utf-8") as f: + reader = csv.DictReader(f) + for row in reader: + try: + rows.append({ + "country": row["Country"].strip(), + "date": row["Date"].strip(), + "count": int(row["Visit Count"]), + "year_month": parse_year_month(row["Date"]), + }) + except (KeyError, ValueError): + continue + print(f"Loaded {len(rows):,} rows from {path}") + return rows + + +def parse_year_month(date_str: str) -> str: + """ + Parse date string to YYYY-MM regardless of format: + - 2023-08-01 → 2023-08 + - 20230801 → 2023-08 + - 2023/08/01 → 2023-08 + - 2023-08-01 00:00:00 → 2023-08 + """ + date_str = date_str.strip().split(" ")[0] # strip time part + date_str = date_str.replace("/", "-") # normalize slashes + + if "-" in date_str: + # YYYY-MM-DD + return date_str[:7] + elif len(date_str) == 8: + # YYYYMMDD + return f"{date_str[:4]}-{date_str[4:6]}" + else: + # fallback — try common formats + for fmt in ("%Y%m%d", "%Y-%m-%d", "%d/%m/%Y", "%m/%d/%Y"): + try: + return datetime.strptime(date_str, fmt).strftime("%Y-%m") + except ValueError: + continue + return date_str[:7] # last resort + """Find the most recent report_*.csv in report dir.""" + files = [ + f for f in os.listdir(report_dir) + if f.startswith("report_") and f.endswith(".csv") + ] + if not files: + return None + files.sort(reverse=True) + return os.path.join(report_dir, files[0]) + + +# ─── Aggregation ────────────────────────────────────────────────────────────── + +def aggregate_monthly(rows: List[dict]) -> Dict[str, dict]: + """ + Aggregate by year_month: + { + "2025-01": { + "total_visits": 12345, + "countries": {"United States": 5000, "Taiwan": 3000, ...}, + "unique_countries": 42 + } + } + """ + monthly: Dict[str, dict] = defaultdict(lambda: { + "total_visits": 0, + "countries": defaultdict(int) + }) + + for row in rows: + ym = row["year_month"] + monthly[ym]["total_visits"] += row["count"] + monthly[ym]["countries"][row["country"]] += row["count"] + + # Compute unique country count per month + result = {} + for ym in sorted(monthly.keys()): + data = monthly[ym] + result[ym] = { + "total_visits": data["total_visits"], + "countries": dict(data["countries"]), + "unique_countries": len(data["countries"]), + } + + return result + + +# ─── MoM Calculation ────────────────────────────────────────────────────────── + +def calc_mom(current: int, previous: int) -> Tuple[float, str]: + """ + Returns (pct_change, formatted_string) + e.g. (12.5, '+12.5%') or (-3.2, '-3.2%') or (0.0, 'N/A') + """ + if previous == 0: + return 0.0, "N/A" + pct = (current - previous) / previous * 100 + sign = "+" if pct >= 0 else "" + return pct, f"{sign}{pct:.1f}%" + + +def build_monthly_stats(monthly: Dict[str, dict]) -> List[dict]: + """ + Build a flat list of monthly stats with MoM columns. + """ + months = sorted(monthly.keys()) + records = [] + + for i, ym in enumerate(months): + prev_ym = months[i - 1] if i > 0 else None + curr_data = monthly[ym] + prev_data = monthly[prev_ym] if prev_ym else None + + curr_visits = curr_data["total_visits"] + curr_countries = curr_data["unique_countries"] + prev_visits = prev_data["total_visits"] if prev_data else 0 + prev_countries = prev_data["unique_countries"] if prev_data else 0 + + visits_mom_pct, visits_mom_str = calc_mom(curr_visits, prev_visits) + countries_mom_pct, countries_mom_str = calc_mom(curr_countries, prev_countries) + + # Top N countries this month + top = sorted( + curr_data["countries"].items(), + key=lambda x: -x[1] + )[:TOP_N_COUNTRIES] + + records.append({ + "year_month": ym, + "total_visits": curr_visits, + "visits_prev_month": prev_visits, + "visits_mom": visits_mom_str, + "visits_mom_pct": visits_mom_pct, + "unique_countries": curr_countries, + "countries_prev_month": prev_countries, + "countries_mom": countries_mom_str, + "countries_mom_pct": countries_mom_pct, + "top_countries": top, + }) + + return records + + +# ─── Export ─────────────────────────────────────────────────────────────────── + +def export_monthly_summary(records: List[dict], output_dir: str) -> str: + """Export monthly summary CSV.""" + os.makedirs(output_dir, exist_ok=True) + path = os.path.join(output_dir, f"monthly_summary_{datetime.today().strftime('%Y%m%d')}.csv") + + with open(path, "w", newline="", encoding="utf-8") as f: + writer = csv.writer(f) + writer.writerow([ + "Year-Month", + "Total Visits", + "Prev Month Visits", + "Visits MoM", + "Unique Countries", + "Prev Month Countries", + "Countries MoM", + "Top 1 Country", + "Top 1 Visits", + "Top 2 Country", + "Top 2 Visits", + "Top 3 Country", + "Top 3 Visits", + ]) + for r in records: + top = r["top_countries"] + def tc(n, field): + return top[n][field] if len(top) > n else "" + writer.writerow([ + r["year_month"], + r["total_visits"], + r["visits_prev_month"], + r["visits_mom"], + r["unique_countries"], + r["countries_prev_month"], + r["countries_mom"], + tc(0, 0), tc(0, 1), + tc(1, 0), tc(1, 1), + tc(2, 0), tc(2, 1), + ]) + + print(f"Monthly summary saved: {path}") + return path + + +def export_top_countries(records: List[dict], output_dir: str) -> str: + """Export per-month top countries CSV.""" + path = os.path.join(output_dir, f"top_countries_{datetime.today().strftime('%Y%m%d')}.csv") + + with open(path, "w", newline="", encoding="utf-8") as f: + writer = csv.writer(f) + writer.writerow(["Year-Month", "Rank", "Country", "Visits", "% of Month"]) + for r in records: + total = r["total_visits"] + for rank, (country, visits) in enumerate(r["top_countries"], 1): + pct = f"{visits/total*100:.1f}%" if total else "0%" + writer.writerow([r["year_month"], rank, country, visits, pct]) + + print(f"Top countries saved: {path}") + return path + + +# ─── Console Report ─────────────────────────────────────────────────────────── + +def _mom_arrow(pct: float) -> str: + if pct > 0: return "↑" + if pct < 0: return "↓" + return "→" + +def _bar(value: int, max_value: int, width: int = 20) -> str: + if max_value == 0: + return "" + filled = int(value / max_value * width) + return "█" * filled + "░" * (width - filled) + +def print_console_report(records: List[dict]): + """Print a formatted report to the terminal.""" + + max_visits = max(r["total_visits"] for r in records) if records else 1 + max_countries = max(r["unique_countries"] for r in records) if records else 1 + + # ── Monthly visits ──────────────────────────────────────────────────────── + print() + print("═" * 72) + print(" 月份訪問量統計 Monthly Visit Count") + print("═" * 72) + print(f" {'月份':<10} {'訪問量':>12} {'MoM':>10} {'趨勢'}") + print(" " + "─" * 68) + + for r in records: + arrow = _mom_arrow(r["visits_mom_pct"]) + bar = _bar(r["total_visits"], max_visits) + print( + f" {r['year_month']:<10} " + f"{r['total_visits']:>12,} " + f"{r['visits_mom']:>10} " + f"{arrow} {bar}" + ) + + print() + + # ── Monthly unique countries ─────────────────────────────────────────────── + print("═" * 72) + print(" 月份不重複國家數 Monthly Unique Country Count") + print("═" * 72) + print(f" {'月份':<10} {'國家數':>10} {'MoM':>10} {'趨勢'}") + print(" " + "─" * 68) + + for r in records: + arrow = _mom_arrow(r["countries_mom_pct"]) + bar = _bar(r["unique_countries"], max_countries) + print( + f" {r['year_month']:<10} " + f"{r['unique_countries']:>10,} " + f"{r['countries_mom']:>10} " + f"{arrow} {bar}" + ) + + print() + + # ── Top countries per month ──────────────────────────────────────────────── + print("═" * 72) + print(f" 各月 Top {TOP_N_COUNTRIES} 國家 Top Countries per Month") + print("═" * 72) + + for r in records: + total = r["total_visits"] + print(f"\n {r['year_month']} (共 {total:,} 次訪問)") + print(f" {'排名':>4} {'國家':<28} {'訪問量':>10} {'佔比':>6}") + print(" " + "─" * 56) + for rank, (country, visits) in enumerate(r["top_countries"], 1): + pct = f"{visits/total*100:.1f}%" if total else "0%" + print(f" {rank:>4}. {country:<28} {visits:>10,} {pct:>6}") + + print() + + # ── MoM Summary ─────────────────────────────────────────────────────────── + if len(records) >= 2: + latest = records[-1] + prev = records[-2] + print("═" * 72) + print(" 最新月份 MoM 摘要 Latest Month MoM Summary") + print("═" * 72) + print( + f" {prev['year_month']} → {latest['year_month']}\n" + f" 訪問量:{prev['total_visits']:,} → {latest['total_visits']:,} " + f"({latest['visits_mom']})\n" + f" 國家數:{prev['unique_countries']} → {latest['unique_countries']} " + f"({latest['countries_mom']})" + ) + print() + + +# ─── Visualization ──────────────────────────────────────────────────────────── + +def generate_charts(records: List[dict], output_dir: str) -> str: + """ + Generate a 4-panel chart: + 1. Monthly total visits (bar + MoM line) + 2. Monthly unique countries (bar + MoM line) + 3. MoM % change comparison (line chart) + 4. Top 5 countries stacked bar (latest 6 months) + """ + if not HAS_MATPLOTLIB: + print("matplotlib not installed — skipping charts.") + print("Install with: pip install matplotlib") + return "" + + months = [r["year_month"] for r in records] + visits = [r["total_visits"] for r in records] + ctries = [r["unique_countries"] for r in records] + v_mom = [r["visits_mom_pct"] for r in records] + c_mom = [r["countries_mom_pct"] for r in records] + + # replace 0.0 (N/A first month) with nan for cleaner line + # fill_between requires float nan, not None + v_mom_plot = [v if i > 0 else float("nan") for i, v in enumerate(v_mom)] + c_mom_plot = [v if i > 0 else float("nan") for i, v in enumerate(c_mom)] + + fig = plt.figure(figsize=(16, 12)) + fig.suptitle("IP Geo Report — Monthly Analysis", fontsize=16, fontweight="bold", y=0.98) + gs = GridSpec(2, 2, figure=fig, hspace=0.45, wspace=0.35) + + colors = { + "visits": "#4C6EF5", + "countries": "#20C997", + "mom_pos": "#51CF66", + "mom_neg": "#FF6B6B", + "neutral": "#ADB5BD", + } + + x = range(len(months)) + + # ── Panel 1: Monthly visits bar ─────────────────────────────────────────── + ax1 = fig.add_subplot(gs[0, 0]) + bars = ax1.bar(x, visits, color=colors["visits"], alpha=0.85, zorder=2) + ax1.set_title("Monthly Total Visits", fontweight="bold") + ax1.set_xticks(list(x)) + ax1.set_xticklabels(months, rotation=45, ha="right", fontsize=8) + ax1.yaxis.set_major_formatter(mticker.FuncFormatter(lambda v, _: f"{v/1000:.0f}K" if v >= 1000 else str(int(v)))) + ax1.set_ylabel("Visits") + ax1.grid(axis="y", linestyle="--", alpha=0.4, zorder=1) + ax1.spines[["top","right"]].set_visible(False) + + # MoM % on secondary axis + ax1b = ax1.twinx() + ax1b.plot(list(x), v_mom_plot, color="#FF6B6B", marker="o", linewidth=1.8, + markersize=5, linestyle="--", label="MoM %", zorder=3) + ax1b.axhline(0, color="#FF6B6B", linewidth=0.6, linestyle=":") + ax1b.set_ylabel("MoM %", color="#FF6B6B", fontsize=9) + ax1b.tick_params(axis="y", labelcolor="#FF6B6B") + ax1b.yaxis.set_major_formatter(mticker.FuncFormatter(lambda v, _: f"{v:+.0f}%")) + + # value labels on bars + for bar, val in zip(bars, visits): + ax1.text(bar.get_x() + bar.get_width()/2, bar.get_height() + max(visits)*0.01, + f"{val:,}", ha="center", va="bottom", fontsize=7, color="#333") + + # ── Panel 2: Monthly unique countries bar ───────────────────────────────── + ax2 = fig.add_subplot(gs[0, 1]) + bars2 = ax2.bar(x, ctries, color=colors["countries"], alpha=0.85, zorder=2) + ax2.set_title("Monthly Unique Countries", fontweight="bold") + ax2.set_xticks(list(x)) + ax2.set_xticklabels(months, rotation=45, ha="right", fontsize=8) + ax2.set_ylabel("Countries") + ax2.grid(axis="y", linestyle="--", alpha=0.4, zorder=1) + ax2.spines[["top","right"]].set_visible(False) + + ax2b = ax2.twinx() + ax2b.plot(list(x), c_mom_plot, color="#FF922B", marker="s", linewidth=1.8, + markersize=5, linestyle="--", label="MoM %", zorder=3) + ax2b.axhline(0, color="#FF922B", linewidth=0.6, linestyle=":") + ax2b.set_ylabel("MoM %", color="#FF922B", fontsize=9) + ax2b.tick_params(axis="y", labelcolor="#FF922B") + ax2b.yaxis.set_major_formatter(mticker.FuncFormatter(lambda v, _: f"{v:+.0f}%")) + + for bar, val in zip(bars2, ctries): + ax2.text(bar.get_x() + bar.get_width()/2, bar.get_height() + max(ctries)*0.02, + str(val), ha="center", va="bottom", fontsize=7, color="#333") + + # ── Panel 3: MoM % comparison line chart ───────────────────────────────── + ax3 = fig.add_subplot(gs[1, 0]) + ax3.plot(list(x), v_mom_plot, color=colors["visits"], marker="o", linewidth=2, + markersize=6, label="Visits MoM %") + ax3.plot(list(x), c_mom_plot, color=colors["countries"], marker="s", linewidth=2, + markersize=6, label="Countries MoM %", linestyle="--") + ax3.axhline(0, color=colors["neutral"], linewidth=1, linestyle=":") + + # shade positive / negative regions + import math + x_num = list(x) + ax3.fill_between(x_num, v_mom_plot, 0, + where=[not math.isnan(v) and v > 0 for v in v_mom_plot], + alpha=0.12, color=colors["mom_pos"], interpolate=True) + ax3.fill_between(x_num, v_mom_plot, 0, + where=[not math.isnan(v) and v < 0 for v in v_mom_plot], + alpha=0.12, color=colors["mom_neg"], interpolate=True) + + ax3.set_title("MoM % Change Comparison", fontweight="bold") + ax3.set_xticks(list(x)) + ax3.set_xticklabels(months, rotation=45, ha="right", fontsize=8) + ax3.set_ylabel("MoM %") + ax3.yaxis.set_major_formatter(mticker.FuncFormatter(lambda v, _: f"{v:+.0f}%")) + ax3.legend(fontsize=8) + ax3.grid(linestyle="--", alpha=0.4) + ax3.spines[["top","right"]].set_visible(False) + + # ── Panel 4: Top 5 countries stacked bar (latest 6 months) ─────────────── + ax4 = fig.add_subplot(gs[1, 1]) + last = records[-6:] if len(records) >= 6 else records + top_months = [r["year_month"] for r in last] + + # Collect global top 5 countries across shown months + country_totals: Dict[str, int] = defaultdict(int) + for r in last: + for country, cnt in r["top_countries"]: + country_totals[country] += cnt + top5 = [c for c, _ in sorted(country_totals.items(), key=lambda x: -x[1])[:5]] + + palette = ["#4C6EF5","#20C997","#FF922B","#CC5DE8","#FF6B6B"] + bottoms = [0] * len(last) + + for i, country in enumerate(top5): + vals = [] + for r in last: + cmap = dict(r["top_countries"]) + vals.append(cmap.get(country, 0)) + ax4.bar(range(len(last)), vals, bottom=bottoms, + label=country, color=palette[i % len(palette)], alpha=0.88) + bottoms = [b + v for b, v in zip(bottoms, vals)] + + ax4.set_title(f"Top 5 Countries — Last {len(last)} Months", fontweight="bold") + ax4.set_xticks(range(len(last))) + ax4.set_xticklabels(top_months, rotation=45, ha="right", fontsize=8) + ax4.yaxis.set_major_formatter(mticker.FuncFormatter(lambda v, _: f"{v/1000:.0f}K" if v >= 1000 else str(int(v)))) + ax4.set_ylabel("Visits") + ax4.legend(fontsize=7, loc="upper left") + ax4.grid(axis="y", linestyle="--", alpha=0.4, zorder=1) + ax4.spines[["top","right"]].set_visible(False) + + # ── Save ───────────────────────────────────────────────────────────────── + os.makedirs(output_dir, exist_ok=True) + path = os.path.join(output_dir, f"analysis_chart_{datetime.today().strftime('%Y%m%d')}.png") + fig.savefig(path, dpi=150, bbox_inches="tight") + plt.close(fig) + print(f"Chart saved: {path}") + return path + + +# ─── Entry Point ────────────────────────────────────────────────────────────── + +def main(): + global TOP_N_COUNTRIES + parser = argparse.ArgumentParser(description="IP Geo Report Analyzer") + parser.add_argument( + "--input", "-i", + help="Path to report CSV (default: latest in ./report/)", + default=None + ) + parser.add_argument( + "--output", "-o", + help="Output directory (default: ./analysis/)", + default=OUTPUT_DIR + ) + parser.add_argument( + "--top", "-n", + help=f"Top N countries per month (default: {TOP_N_COUNTRIES})", + type=int, + default=TOP_N_COUNTRIES + ) + args = parser.parse_args() + + # Find input file + input_path = args.input + if not input_path: + input_path = find_latest_report(REPORT_DIR) + if not input_path: + print(f"No report_*.csv found in {REPORT_DIR}") + print("Run ip_geo_report.py first, or specify --input path") + return + + print(f"Analyzing: {input_path}") + + # Load → aggregate → build stats + rows = load_report(input_path) + monthly = aggregate_monthly(rows) + records = build_monthly_stats(monthly) + + if not records: + print("No data found in report.") + return + + # Print to console + print_console_report(records) + + # Export CSVs + export_monthly_summary(records, args.output) + export_top_countries(records, args.output) + + # Generate chart + generate_charts(records, args.output) + + print(f"\nAnalysis complete. Files saved to: {args.output}/") + + +if __name__ == "__main__": + main() diff --git a/bot.py b/bot.py new file mode 100644 index 0000000..dc8924d --- /dev/null +++ b/bot.py @@ -0,0 +1,601 @@ +""" +Bot Analyzer +- Select visitor records from a date range +- Analyze which visits are likely non-human +- Score each record with reasons +- Export human / bot separated reports +""" + +import os +import re +import csv +import argparse +import ipaddress +import logging +from datetime import datetime, date +from collections import defaultdict +from typing import Optional, List, Dict, Tuple + +# ─── Config ─────────────────────────────────────────────────────────────────── + +DB_CONFIG = { + "host": "localhost", + "user": "user", + "password": "password", + "database": "database", + "charset": "utf8mb4", +} + +TABLE_NAME = "wp_statpress" +IP_COLUMN = "ip" +DATE_COLUMN = "timestamp" +UA_COLUMN = "agent" # set None if not available +PATH_COLUMN = "urlrequested" # set None if not available +BATCH_SIZE = 50_000 +OUTPUT_DIR = "./bot_analysis" + +# Scoring thresholds +BOT_SCORE_THRESHOLD = 5 # score >= this = bot +SUSPICIOUS_THRESHOLD = 3 # score >= this = suspicious + +# IP frequency thresholds — scaled to time window +HIGH_FREQ_PER_DAY = 100 # flag if IP hits > N times in a single day +HIGH_FREQ_PER_HOUR = 30 # flag if IP hits > N times in a single hour (if datetime available) +HIGH_FREQ_TOTAL_RATIO = 0.01 # flag if IP accounts for > 1% of total traffic in date range + +# ─── Logging ────────────────────────────────────────────────────────────────── + +logging.basicConfig( + level=logging.INFO, + format="%(asctime)s [%(levelname)s] %(message)s", + handlers=[ + logging.FileHandler("bot_analyzer.log"), + logging.StreamHandler() + ] +) +log = logging.getLogger(__name__) + +# ─── Scoring Rules ──────────────────────────────────────────────────────────── + +# Each entry: (score, reason, description) +SCORING_RULES = { + "private_ip": (10, "Private/Reserved IP", "IP is in private or reserved range"), + "invalid_ip": (10, "Invalid IP", "IP address format is invalid"), + "no_ua": (7, "No User Agent", "Empty or missing user agent"), + "bot_ua": (8, "Bot User Agent", "User agent matches known bot/crawler pattern"), + "tool_ua": (7, "Tool User Agent", "User agent matches automation tool (curl, wget, python...)"), + "old_ie": (5, "Suspicious Old Browser", "IE6/IE7/IE8 + suspicious combination"), + "attack_path": (9, "Attack Path", "Request path matches attack/scanner pattern"), + "scanner_path": (8, "Scanner Path", "Request path matches vulnerability scanner"), + "high_freq_day": (7, "High Freq Per Day", f"IP appears > {HIGH_FREQ_PER_DAY} times in a single day"), + "high_freq_hour": (8, "High Freq Per Hour", f"IP appears > {HIGH_FREQ_PER_HOUR} times in a single hour"), + "high_freq_ratio": (5, "High Traffic Ratio", f"IP accounts for > {HIGH_FREQ_TOTAL_RATIO*100:.1f}% of total traffic"), + "empty_path": (3, "Empty Path", "Request path is empty or just /"), + "suspicious_combo": (4, "Suspicious Combination", "Old UA + unusual path combination"), +} + +# Known bot / crawler UA patterns +BOT_UA_RE = re.compile( + r"(?i)(" + r"googlebot|bingbot|slurp|duckduckbot|baiduspider|yandexbot|" + r"facebookexternalhit|twitterbot|linkedinbot|pinterest|" + r"mj12bot|dotbot|rogerbot|semrushbot|ahrefsbot|" + r"archive\.org_bot|ia_archiver|wayback|" + r"petalbot|bytespider|gptbot|claudebot|anthropic|" + r"crawler|spider|scraper|bot\b" + r")" +) + +# Automation tool UA patterns +TOOL_UA_RE = re.compile( + r"(?i)(" + r"curl|wget|python-requests|python-urllib|httpie|" + r"go-http-client|java/|okhttp|axios|node-fetch|" + r"libwww-perl|lwp-|ruby|perl|php/|" + r"zgrab|masscan|nmap|nikto|sqlmap|nuclei|" + r"dirbuster|gobuster|wfuzz|hydra|" + r"headlesschrome|phantomjs|selenium|puppeteer|playwright|" + r"postman|insomnia" + r")" +) + +# Attack / scanner path patterns +ATTACK_PATH_RE = re.compile( + r"(?i)(" + r"wp-login\.php|xmlrpc\.php|" + r"\.env|\.git|\.svn|\.htaccess|\.htpasswd|" + r"phpmyadmin|pma|adminer|" + r"manager/html|solr/admin|jenkins|" + r"actuator|/api/v1/pods|" + r"/etc/passwd|/proc/self|/windows/win\.ini|" + r"select\s+.+from|union\s+select|waitfor\s+delay|pg_sleep|" + r"exec\s*\(|eval\s*\(|base64_decode|" + r"\.\./|%2e%2e|%252e|\.\.%2f|" + r"= 13 and ":" in s else None + + self.daily[ip][date_str] += 1 + if hour_str: + self.hourly[ip][hour_str] += 1 + + def max_daily(self, ip: str) -> Tuple[int, str]: + """Return (max_count, date) for the busiest day.""" + if ip not in self.daily or not self.daily[ip]: + return 0, "" + day, cnt = max(self.daily[ip].items(), key=lambda x: x[1]) + return cnt, day + + def max_hourly(self, ip: str) -> Tuple[int, str]: + """Return (max_count, hour) for the busiest hour.""" + if ip not in self.hourly or not self.hourly[ip]: + return 0, "" + hour, cnt = max(self.hourly[ip].items(), key=lambda x: x[1]) + return cnt, hour + + def traffic_ratio(self, ip: str) -> float: + """Fraction of total traffic this IP accounts for.""" + if self.grand_total == 0: + return 0.0 + return self.total[ip] / self.grand_total + + def freq_signals(self, ip: str) -> List[Tuple[str, str]]: + """ + Returns list of (rule_key, detail_string) for triggered freq signals. + """ + signals = [] + + max_day_cnt, max_day = self.max_daily(ip) + max_hour_cnt, max_hour = self.max_hourly(ip) + ratio = self.traffic_ratio(ip) + + if max_hour_cnt > HIGH_FREQ_PER_HOUR: + signals.append(( + "high_freq_hour", + f"{max_hour_cnt} hits in {max_hour}" + )) + + if max_day_cnt > HIGH_FREQ_PER_DAY: + signals.append(( + "high_freq_day", + f"{max_day_cnt} hits on {max_day}" + )) + + if ratio > HIGH_FREQ_TOTAL_RATIO: + signals.append(( + "high_freq_ratio", + f"{ratio*100:.2f}% of total traffic" + )) + + return signals + + def top(self, n: int = 20) -> List[Tuple[str, int, int, float]]: + """Return top N IPs: (ip, total, max_daily, ratio%)""" + result = [] + for ip, total in sorted(self.total.items(), key=lambda x: -x[1])[:n]: + max_day, _ = self.max_daily(ip) + ratio = self.traffic_ratio(ip) * 100 + result.append((ip, total, max_day, ratio)) + return result + + +# ─── Scorer ─────────────────────────────────────────────────────────────────── + +class BotScorer: + """Score a single visit record and return reasons.""" + + def __init__(self, ip_freq: IPFrequencyTracker): + self.ip_freq = ip_freq + + def is_private_ip(self, ip: str) -> bool: + try: + addr = ipaddress.ip_address(ip) + return any(addr in net for net in PRIVATE_NETWORKS) + except ValueError: + return True + + def score( + self, + ip: str, + ua: Optional[str] = None, + path: Optional[str] = None + ) -> Tuple[int, List[str]]: + """ + Returns (total_score, [list of reasons]) + Higher score = more likely to be a bot. + """ + total = 0 + reasons = [] + + # ── IP checks ──────────────────────────────────────────────────────── + if not ip or ip.strip() in ("", "-", "unknown"): + total += SCORING_RULES["invalid_ip"][0] + reasons.append(SCORING_RULES["invalid_ip"][1]) + elif self.is_private_ip(ip): + total += SCORING_RULES["private_ip"][0] + reasons.append(SCORING_RULES["private_ip"][1]) + + # Windowed frequency signals + for rule_key, detail in self.ip_freq.freq_signals(ip): + score_val, label, _ = SCORING_RULES[rule_key] + total += score_val + reasons.append(f"{label} ({detail})") + + # ── User Agent checks ──────────────────────────────────────────────── + if ua is not None: + ua_clean = (ua or "").strip() + + if not ua_clean or ua_clean == "-": + total += SCORING_RULES["no_ua"][0] + reasons.append(SCORING_RULES["no_ua"][1]) + elif BOT_UA_RE.search(ua_clean): + total += SCORING_RULES["bot_ua"][0] + reasons.append(SCORING_RULES["bot_ua"][1]) + elif TOOL_UA_RE.search(ua_clean): + total += SCORING_RULES["tool_ua"][0] + reasons.append(SCORING_RULES["tool_ua"][1]) + elif re.search(r"(?i)MSIE [678]\.", ua_clean): + # IE 6/7/8 is ancient — very suspicious in 2024+ + total += SCORING_RULES["old_ie"][0] + reasons.append(SCORING_RULES["old_ie"][1]) + + # ── Path checks ────────────────────────────────────────────────────── + if path is not None: + path_clean = (path or "").strip() + + if ATTACK_PATH_RE.search(path_clean): + total += SCORING_RULES["attack_path"][0] + reasons.append(SCORING_RULES["attack_path"][1]) + + if path_clean in ("", "/", "-"): + total += SCORING_RULES["empty_path"][0] + reasons.append(SCORING_RULES["empty_path"][1]) + + # ── Combination checks ─────────────────────────────────────────────── + if ua is not None and path is not None: + ua_clean = (ua or "").strip() + path_clean = (path or "").strip() + if re.search(r"(?i)MSIE", ua_clean) and ATTACK_PATH_RE.search(path_clean): + total += SCORING_RULES["suspicious_combo"][0] + reasons.append(SCORING_RULES["suspicious_combo"][1]) + + return total, reasons + + +# ─── Database ───────────────────────────────────────────────────────────────── + +class Database: + + def __init__(self): + import mysql.connector + self.conn = mysql.connector.connect(**DB_CONFIG) + self.cursor = self.conn.cursor(buffered=False) + log.info("MySQL connected") + + def count_range(self, start: str, end: str) -> int: + self.cursor.execute( + f"SELECT COUNT(*) FROM {TABLE_NAME} " + f"WHERE {DATE_COLUMN} BETWEEN %s AND %s", + (start, end) + ) + return self.cursor.fetchone()[0] + + def fetch_batch(self, start: str, end: str, last_id: int) -> list: + cols = ["id", IP_COLUMN, DATE_COLUMN] + if UA_COLUMN: + cols.append(UA_COLUMN) + if PATH_COLUMN: + cols.append(PATH_COLUMN) + + self.cursor.execute( + f"SELECT {', '.join(cols)} FROM {TABLE_NAME} " + f"WHERE {DATE_COLUMN} BETWEEN %s AND %s " + f"AND id > %s " + f"ORDER BY id ASC " + f"LIMIT %s", + (start, end, last_id, BATCH_SIZE) + ) + return self.cursor.fetchall() + + def close(self): + self.cursor.close() + self.conn.close() + log.info("MySQL disconnected") + + +# ─── Reporter ───────────────────────────────────────────────────────────────── + +class Reporter: + + def __init__(self, output_dir: str, date_range: str): + os.makedirs(output_dir, exist_ok=True) + ts = datetime.today().strftime("%Y%m%d_%H%M%S") + self.human_path = os.path.join(output_dir, f"human_{date_range}_{ts}.csv") + self.bot_path = os.path.join(output_dir, f"bot_{date_range}_{ts}.csv") + self.suspicious_path = os.path.join(output_dir, f"suspicious_{date_range}_{ts}.csv") + self.summary_path = os.path.join(output_dir, f"summary_{date_range}_{ts}.csv") + + cols_base = ["id", "ip", "date"] + cols_ua = ["user_agent"] if UA_COLUMN else [] + cols_path = ["path"] if PATH_COLUMN else [] + cols_score = ["score", "reasons"] + + self.cols_human = cols_base + cols_ua + cols_path + self.cols_bot = cols_base + cols_ua + cols_path + cols_score + + self._human_f = open(self.human_path, "w", newline="", encoding="utf-8") + self._bot_f = open(self.bot_path, "w", newline="", encoding="utf-8") + self._suspicious_f = open(self.suspicious_path, "w", newline="", encoding="utf-8") + + self._human_w = csv.writer(self._human_f) + self._bot_w = csv.writer(self._bot_f) + self._suspicious_w = csv.writer(self._suspicious_f) + + self._human_w.writerow(self.cols_human) + self._bot_w.writerow(self.cols_bot) + self._suspicious_w.writerow(self.cols_bot) + + self.stats = { + "total": 0, + "human": 0, + "bot": 0, + "suspicious": 0, + } + self.reason_counts: Dict[str, int] = defaultdict(int) + + def write(self, row_id, ip, visit_date, ua, path, score, reasons): + self.stats["total"] += 1 + base = [row_id, ip, visit_date] + if UA_COLUMN: + base.append(ua or "") + if PATH_COLUMN: + base.append(path or "") + + if score >= BOT_SCORE_THRESHOLD: + self.stats["bot"] += 1 + self._bot_w.writerow(base + [score, " | ".join(reasons)]) + for r in reasons: + self.reason_counts[r.split("(")[0].strip()] += 1 + + elif score >= SUSPICIOUS_THRESHOLD: + self.stats["suspicious"] += 1 + self._suspicious_w.writerow(base + [score, " | ".join(reasons)]) + for r in reasons: + self.reason_counts[r.split("(")[0].strip()] += 1 + + else: + self.stats["human"] += 1 + self._human_w.writerow(base) + + def close(self): + self._human_f.close() + self._bot_f.close() + self._suspicious_f.close() + + def export_summary(self, ip_freq: IPFrequencyTracker): + total = self.stats["total"] + with open(self.summary_path, "w", newline="", encoding="utf-8") as f: + writer = csv.writer(f) + + writer.writerow(["=== Overall Stats ==="]) + writer.writerow(["Category", "Count", "Percentage"]) + for key in ["human", "bot", "suspicious"]: + pct = f"{self.stats[key]/total*100:.1f}%" if total else "0%" + writer.writerow([key.capitalize(), self.stats[key], pct]) + writer.writerow(["Total", total, "100%"]) + + writer.writerow([]) + writer.writerow(["=== Bot Detection Reasons ==="]) + writer.writerow(["Reason", "Count"]) + for reason, count in sorted( + self.reason_counts.items(), key=lambda x: -x[1] + ): + writer.writerow([reason, count]) + + writer.writerow([]) + writer.writerow(["=== Top 20 High Frequency IPs ==="]) + writer.writerow(["IP", "Total Hits", "Max Daily", "Traffic Ratio %", "Classification"]) + for ip, total_hits, max_day, ratio in ip_freq.top(20): + signals = ip_freq.freq_signals(ip) + classification = "Bot" if signals else "Normal" + writer.writerow([ip, total_hits, max_day, f"{ratio:.2f}%", classification]) + + log.info(f"Summary saved: {self.summary_path}") + + +# ─── Console Output ─────────────────────────────────────────────────────────── + +def print_summary(stats: dict, reason_counts: dict, ip_freq: IPFrequencyTracker): + total = stats["total"] + + print(f"\n{'═'*55}") + print(" Bot Analysis Summary") + print(f"{'═'*55}") + print(f" {'Total records':<25} {total:>12,}") + print(f" {'─'*52}") + + for key, label in [ + ("human", "Human"), + ("bot", "Bot"), + ("suspicious", "Suspicious"), + ]: + count = stats[key] + pct = f"{count/total*100:.1f}%" if total else "0%" + bar = "█" * int(count/total*30) if total else "" + print(f" {label:<25} {count:>12,} {pct:>6} {bar}") + + print(f"\n{'═'*55}") + print(" Top Bot Detection Reasons") + print(f"{'═'*55}") + for reason, count in sorted(reason_counts.items(), key=lambda x: -x[1])[:10]: + pct = f"{count/total*100:.1f}%" if total else "0%" + print(f" {reason:<35} {count:>8,} {pct:>6}") + + print(f"\n{'═'*55}") + print(f" Top 10 High Frequency IPs") + print(f"{'═'*55}") + print(f" {'IP':<20} {'Total':>8} {'Max/Day':>8} {'Ratio':>7} {'Status'}") + print(f" {'─'*60}") + for ip, total_hits, max_day, ratio in ip_freq.top(10): + signals = ip_freq.freq_signals(ip) + status = "Bot" if signals else "Normal" + print(f" {ip:<20} {total_hits:>8,} {max_day:>8,} {ratio:>6.2f}% {status}") + print() + + +# ─── Main ───────────────────────────────────────────────────────────────────── + +def main(): + global BOT_SCORE_THRESHOLD, HIGH_FREQ_PER_DAY, HIGH_FREQ_PER_HOUR, HIGH_FREQ_TOTAL_RATIO + parser = argparse.ArgumentParser(description="Bot Analyzer — Detect non-human visits") + parser.add_argument("--start", "-s", required=True, help="Start date YYYY-MM-DD") + parser.add_argument("--end", "-e", required=True, help="End date YYYY-MM-DD") + parser.add_argument("--output", "-o", default=OUTPUT_DIR, help="Output directory") + parser.add_argument("--bot-threshold", type=int, default=BOT_SCORE_THRESHOLD, + help=f"Bot score threshold (default: {BOT_SCORE_THRESHOLD})") + parser.add_argument("--freq-per-day", type=int, default=HIGH_FREQ_PER_DAY, + help=f"Max hits per day threshold (default: {HIGH_FREQ_PER_DAY})") + parser.add_argument("--freq-per-hour", type=int, default=HIGH_FREQ_PER_HOUR, + help=f"Max hits per hour threshold (default: {HIGH_FREQ_PER_HOUR})") + parser.add_argument("--freq-ratio", type=float, default=HIGH_FREQ_TOTAL_RATIO, + help=f"Max traffic ratio threshold (default: {HIGH_FREQ_TOTAL_RATIO})") + args = parser.parse_args() + + BOT_SCORE_THRESHOLD = args.bot_threshold + HIGH_FREQ_PER_DAY = args.freq_per_day + HIGH_FREQ_PER_HOUR = args.freq_per_hour + HIGH_FREQ_TOTAL_RATIO = args.freq_ratio + + start_date = args.start + end_date = args.end + date_range = f"{start_date}_to_{end_date}" + + log.info(f"Analyzing: {start_date} → {end_date}") + + db = Database() + ip_freq = IPFrequencyTracker() + reporter = Reporter(args.output, date_range) + + total = db.count_range(start_date, end_date) + log.info(f"Total records in range: {total:,}") + + # ── Pass 1: Count IP frequencies ────────────────────────────────────────── + log.info("Pass 1/2: Counting IP frequencies...") + last_id = 0 + processed = 0 + + while True: + rows = db.fetch_batch(start_date, end_date, last_id) + if not rows: + break + for row in rows: + ip = str(row[1]).strip() if row[1] else "" + visit_date = row[2] + ip_freq.add(ip, visit_date) + last_id = row[0] + processed += len(rows) + log.info(f"Pass 1: {processed:,}/{total:,} ({processed/total*100:.1f}%)") + + # ── Pass 2: Score each record ────────────────────────────────────────────── + log.info("Pass 2/2: Scoring records...") + scorer = BotScorer(ip_freq) + last_id = 0 + processed = 0 + + while True: + rows = db.fetch_batch(start_date, end_date, last_id) + if not rows: + break + + for row in rows: + idx = 0 + row_id = row[idx]; idx += 1 + ip = str(row[idx]).strip() if row[idx] else ""; idx += 1 + visit_date = row[idx]; idx += 1 + ua = str(row[idx]).strip() if UA_COLUMN and idx < len(row) else None + if ua is not None: idx += 1 + path = str(row[idx]).strip() if PATH_COLUMN and idx < len(row) else None + + score, reasons = scorer.score(ip, ua, path) + reporter.write(row_id, ip, visit_date, ua, path, score, reasons) + last_id = row_id + + processed += len(rows) + log.info(f"Pass 2: {processed:,}/{total:,} ({processed/total*100:.1f}%)") + + # ── Finalize ────────────────────────────────────────────────────────────── + reporter.close() + reporter.export_summary(ip_freq) + db.close() + + print_summary(reporter.stats, dict(reporter.reason_counts), ip_freq) + + print(f"{'═'*55}") + print(" Output Files") + print(f"{'═'*55}") + print(f" Human : {reporter.human_path}") + print(f" Bot : {reporter.bot_path}") + print(f" Suspicious: {reporter.suspicious_path}") + print(f" Summary : {reporter.summary_path}") + print() + + +if __name__ == "__main__": + main() diff --git a/main.py b/main.py new file mode 100644 index 0000000..63df047 --- /dev/null +++ b/main.py @@ -0,0 +1,727 @@ +""" +IP Geo Report Generator +- Batch processing for 480M rows +- LRU memory cache with disk spillover +- Resume strategy via checkpoint +- Daily visit count per country report +- Uses ipinfo bundle_location_lite.mmdb + API fallback +- Bot/attack detection and filtering +""" + +import os +import re +import csv +import json +import time +import pickle +import shelve +import logging +import ipaddress +from datetime import datetime, date +from collections import defaultdict, OrderedDict +from typing import Optional, Dict, Tuple, Set +from dataclasses import dataclass, field + +# ─── Config ─────────────────────────────────────────────────────────────────── + +DB_CONFIG = { + "host": "localhost", + "user": "user", + "password": "password", + "database": "database", + "charset": "utf8mb4", +} + +TABLE_NAME = "wp_statpress" # your table name +IP_COLUMN = "ip" # IP address column +DATE_COLUMN = "date" # date column (DATE or DATETIME) +UA_COLUMN = "agent" # user agent column (set None if not available) +PATH_COLUMN = "urlrequested" +BATCH_SIZE = 50_000 # rows per batch +LRU_MAX_SIZE = 100_000 # max IPs in memory cache +DISK_CACHE_PATH = "./cache/ip_geo" # shelve disk cache path +CHECKPOINT_FILE = "./cache/checkpoint.json" +REPORT_OUTPUT = "./report" # output directory +GEOIP_DB_PATH = "./ipinfo_lite.mmdb" # ipinfo free DB path +IPINFO_TOKEN = "6f2ffc9ac24285" # ipinfo.io API token (free tier) + +# Disk cache settings +DISK_CACHE_TTL = 86400 * 30 # 30 days in seconds +DISK_FLUSH_EVERY = 10_000 # flush disk cache every N lookups + +# Bot detection thresholds +BOT_IP_REQ_THRESHOLD = 5000 # flag IP if it appears more than this in a batch +BOT_RATE_WINDOW_SEC = 60 # time window for rate detection + +# ─── Logging ────────────────────────────────────────────────────────────────── + +logging.basicConfig( + level=logging.INFO, + format="%(asctime)s [%(levelname)s] %(message)s", + handlers=[ + logging.FileHandler("ip_geo_report.log"), + logging.StreamHandler() + ] +) +log = logging.getLogger(__name__) + +# ─── Bot Detector ───────────────────────────────────────────────────────────── + +class BotDetector: + """ + Detects and filters bots/attacks based on: + 1. IP address (private, reserved, known bot ranges) + 2. User agent string (known bot patterns) + 3. Request path (scanner/attack patterns) + 4. IP request frequency (too many hits = bot) + """ + + # Known bot / crawler user agent patterns + BOT_UA_PATTERNS = re.compile( + r"(?i)(" + r"bot|crawler|spider|scraper|scan|curl|wget|python|java|" + r"go-http|ruby|perl|php|axios|libwww|httpclient|okhttp|" + r"zgrab|masscan|nmap|nikto|sqlmap|dirbuster|nuclei|" + r"semrush|ahrefs|mj12bot|dotbot|rogerbot|bingbot|" + r"googlebot|yandexbot|baiduspider|duckduckbot|" + r"facebookexternalhit|twitterbot|linkedinbot|" + r"archive\.org|ia_archiver|wayback|" + r"headlesschrome|phantomjs|selenium|puppeteer|playwright" + r")" + ) + + # Suspicious request path patterns (scanners / attacks) + ATTACK_PATH_PATTERNS = re.compile( + r"(?i)(" + r"wp-login|xmlrpc|\.env|\.git|\.svn|\.htaccess|" + r"phpmyadmin|adminer|manager/html|solr/admin|" + r"actuator|/etc/passwd|/proc/self|" + r"select\s+.+from|union\s+select|waitfor\s+delay|" + r"pg_sleep|exec\(|eval\(|base64_decode|" + r"\.php\?.*=http|\.php\?.*=//|" + r"\.\./|%2e%2e|%252e|" # path traversal + r" bool: + """Check if IP is private/reserved.""" + try: + addr = ipaddress.ip_address(ip) + return any(addr in net for net in self.PRIVATE_RANGES) + except ValueError: + return True # invalid IP = skip + + def is_bot_ua(self, ua: str) -> bool: + """Check if user agent matches known bot patterns.""" + if not ua or ua.strip() == "-": + return True # missing UA = likely bot + return bool(self.BOT_UA_PATTERNS.search(ua)) + + def is_attack_path(self, path: str) -> bool: + """Check if request path matches attack/scanner patterns.""" + if not path: + return False + return bool(self.ATTACK_PATH_PATTERNS.search(path)) + + def track_ip_frequency(self, ip: str) -> bool: + """Track IP hit count. Returns True if IP exceeds threshold.""" + self.ip_freq[ip] += 1 + if self.ip_freq[ip] > BOT_IP_REQ_THRESHOLD: + self.skipped_ips.add(ip) + return True + return False + + def reset_frequency(self): + """Reset per-batch frequency counter.""" + self.ip_freq.clear() + + def is_bot(self, ip: str, ua: str = None, path: str = None) -> Tuple[bool, str]: + """ + Main bot detection method. + Returns (is_bot: bool, reason: str) + """ + self.stats["total"] += 1 + + # Already flagged as bot IP + if ip in self.skipped_ips: + self.stats["skipped_ip"] += 1 + return True, "known_bot_ip" + + # Private / invalid IP + if self.is_private_ip(ip): + self.stats["skipped_private"] += 1 + return True, "private_ip" + + # IP frequency check + if self.track_ip_frequency(ip): + self.stats["skipped_freq"] += 1 + log.info(f"Bot detected by frequency: {ip} ({self.ip_freq[ip]} hits)") + return True, "high_frequency" + + # User agent check + if ua is not None and self.is_bot_ua(ua): + self.stats["skipped_ua"] += 1 + return True, "bot_ua" + + # Attack path check + if path is not None and self.is_attack_path(path): + self.stats["skipped_path"] += 1 + return True, "attack_path" + + return False, "" + + def report(self) -> dict: + total = self.stats["total"] + skipped = sum(v for k, v in self.stats.items() if k != "total") + return { + **self.stats, + "skipped_total": skipped, + "skip_rate": f"{skipped/total*100:.1f}%" if total else "0%", + "bot_ips_found": len(self.skipped_ips), + } + + +# ─── LRU Memory Cache ───────────────────────────────────────────────────────── + +class LRUCache: + """LRU cache. Returns evicted key so caller can spill to disk.""" + + def __init__(self, max_size: int = 100_000): + self.max_size = max_size + self.cache = OrderedDict() + self.hits = 0 + self.misses = 0 + + def get(self, key: str) -> Optional[str]: + if key in self.cache: + self.cache.move_to_end(key) + self.hits += 1 + return self.cache[key] + self.misses += 1 + return None + + def put(self, key: str, value: str) -> Optional[str]: + if key in self.cache: + self.cache.move_to_end(key) + self.cache[key] = value + if len(self.cache) > self.max_size: + evicted_key, _ = self.cache.popitem(last=False) + return evicted_key + return None + + def stats(self) -> dict: + total = self.hits + self.misses + return { + "size": len(self.cache), + "hits": self.hits, + "misses": self.misses, + "hit_rate": f"{self.hits/total*100:.1f}%" if total else "0%" + } + + +# ─── Disk Cache ─────────────────────────────────────────────────────────────── + +class DiskCache: + """Persistent shelve cache with TTL expiry.""" + + def __init__(self, path: str, ttl: int = DISK_CACHE_TTL): + os.makedirs(os.path.dirname(path) or ".", exist_ok=True) + self.ttl = ttl + self.db = shelve.open(path, writeback=False) + self.hits = 0 + self.misses = 0 + log.info(f"Disk cache opened: {path} ({len(self.db)} entries)") + + def get(self, key: str) -> Optional[str]: + entry = self.db.get(key) + if entry: + value, ts = entry + if time.time() - ts < self.ttl: + self.hits += 1 + return value + del self.db[key] + self.misses += 1 + return None + + def put(self, key: str, value: str): + self.db[key] = (value, time.time()) + + def flush(self): + self.db.sync() + + def close(self): + self.db.sync() + self.db.close() + + def stats(self) -> dict: + total = self.hits + self.misses + return { + "size": len(self.db), + "hits": self.hits, + "misses": self.misses, + "hit_rate": f"{self.hits/total*100:.1f}%" if total else "0%" + } + + +# ─── Two-Level Cache ────────────────────────────────────────────────────────── + +class TwoLevelCache: + """L1: LRU memory | L2: Disk shelve (handles LRU evictions)""" + + def __init__(self, lru_size: int, disk_path: str): + self.lru = LRUCache(max_size=lru_size) + self.disk = DiskCache(disk_path) + self.lookup_cnt = 0 + + def get(self, ip: str) -> Optional[str]: + val = self.lru.get(ip) + if val is not None: + return val + val = self.disk.get(ip) + if val is not None: + self.lru.put(ip, val) + return val + return None + + def put(self, ip: str, country: str): + evicted_key = self.lru.put(ip, country) + if evicted_key: + self.disk.put(evicted_key, country) + self.lookup_cnt += 1 + if self.lookup_cnt % DISK_FLUSH_EVERY == 0: + self.disk.flush() + log.info( + f"Cache → Memory: {self.lru.stats()} | " + f"Disk: {self.disk.stats()}" + ) + + def close(self): + self.disk.flush() + self.disk.close() + + def stats(self) -> dict: + return {"memory": self.lru.stats(), "disk": self.disk.stats()} + + +# ─── GeoIP Lookup ───────────────────────────────────────────────────────────── + +class GeoIPLookup: + """ + 1. ipinfo bundle_location_lite.mmdb (offline, no limit) + Fields: country, country_code, continent, asn, as_name, as_domain + 2. ipinfo API fallback (50k/month free tier) + """ + + def __init__(self, db_path: str = None, api_token: str = None): + self.reader = None + self.client = None + self.api_calls = 0 + sources = [] + + if db_path and os.path.exists(db_path): + try: + import maxminddb + self.reader = maxminddb.open_database(db_path) + db_type = self.reader.metadata().database_type + sources.append("mmdb") + log.info(f"mmdb loaded: {db_path} (type: {db_type})") + test = self.reader.get("8.8.8.8") + if test: + log.info(f"mmdb fields: {list(test.keys())}") + except ImportError: + log.error("pip install maxminddb") + except Exception as e: + log.error(f"mmdb open failed: {e}") + else: + log.warning(f"mmdb not found: {db_path}") + + if api_token: + try: + import ipinfo + self.client = ipinfo.getHandler(api_token) + sources.append("api") + log.info("ipinfo API loaded as fallback") + log.warning("API active — watch 50k/month free limit!") + except ImportError: + log.error("pip install ipinfo") + + if not sources: + log.warning("No GeoIP source — all IPs = Unknown") + else: + log.info(f"Lookup order: {' → '.join(sources)}") + + def _extract_country(self, result: dict) -> Optional[str]: + """ + ipinfo bundle_location_lite structure: + {'country': 'United States', 'country_code': 'US', ...} + """ + if not result: + return None + return result.get("country") or None + + def _lookup_mmdb(self, ip: str) -> Optional[str]: + try: + return self._extract_country(self.reader.get(ip)) + except Exception: + return None + + def _lookup_api(self, ip: str) -> Optional[str]: + try: + details = self.client.getDetails(ip) + self.api_calls += 1 + if self.api_calls % 100 == 0: + log.info(f"API calls: {self.api_calls:,}") + return ( + getattr(details, "country_name", None) or + getattr(details, "country", None) + ) + except Exception as e: + log.warning(f"API failed for {ip}: {e}") + return None + + def lookup(self, ip: str) -> str: + if self.reader: + country = self._lookup_mmdb(ip) + if country: + return country + if self.client: + country = self._lookup_api(ip) + if country: + return country + return "Unknown" + + def close(self): + if self.reader: + self.reader.close() + if self.api_calls > 0: + log.info(f"Total API fallback calls: {self.api_calls:,}") + + +# ─── Checkpoint ─────────────────────────────────────────────────────────────── + +@dataclass +class Checkpoint: + last_id: int = 0 + rows_processed: int = 0 + batches_done: int = 0 + started_at: str = field(default_factory=lambda: datetime.now().isoformat()) + updated_at: str = field(default_factory=lambda: datetime.now().isoformat()) + + def save(self, path: str): + self.updated_at = datetime.now().isoformat() + with open(path, "w") as f: + json.dump(self.__dict__, f, indent=2) + + @classmethod + def load(cls, path: str) -> "Checkpoint": + if os.path.exists(path): + with open(path) as f: + data = json.load(f) + log.info(f"Resuming from checkpoint: {data}") + return cls(**data) + log.info("No checkpoint — starting fresh") + return cls() + + +# ─── Report Aggregator ──────────────────────────────────────────────────────── + +class ReportAggregator: + """Aggregates visit counts by country and date. Survives crashes via partial saves.""" + + def __init__(self, output_dir: str): + os.makedirs(output_dir, exist_ok=True) + self.output_dir = output_dir + self.partial_file = os.path.join(output_dir, "partial_data.pkl") + self.data: Dict[str, Dict[str, int]] = defaultdict(lambda: defaultdict(int)) + self._load_partial() + + def _load_partial(self): + if os.path.exists(self.partial_file): + with open(self.partial_file, "rb") as f: + saved = pickle.load(f) + for country, dates in saved.items(): + for d, cnt in dates.items(): + self.data[country][d] += cnt + log.info(f"Loaded partial data: {len(self.data)} countries") + + def add(self, country: str, visit_date): + d = ( + visit_date.strftime("%Y-%m-%d") + if hasattr(visit_date, "strftime") + else str(visit_date)[:10] + ) + self.data[country][d] += 1 + + def save_partial(self): + with open(self.partial_file, "wb") as f: + pickle.dump(dict(self.data), f) + + def export_csv(self) -> str: + path = os.path.join(self.output_dir, f"report_{date.today()}.csv") + with open(path, "w", newline="", encoding="utf-8") as f: + writer = csv.writer(f) + writer.writerow(["Country", "Date", "Visit Count"]) + for country in sorted(self.data): + for d in sorted(self.data[country]): + writer.writerow([country, d, self.data[country][d]]) + log.info(f"CSV saved: {path}") + return path + + def export_summary(self) -> str: + path = os.path.join(self.output_dir, f"summary_{date.today()}.csv") + totals = {c: sum(d.values()) for c, d in self.data.items()} + with open(path, "w", newline="", encoding="utf-8") as f: + writer = csv.writer(f) + writer.writerow(["Country", "Total Visits"]) + for country, total in sorted(totals.items(), key=lambda x: -x[1]): + writer.writerow([country, total]) + log.info(f"Summary saved: {path}") + return path + + def print_top(self, n: int = 20): + totals = {c: sum(d.values()) for c, d in self.data.items()} + top = sorted(totals.items(), key=lambda x: -x[1])[:n] + print(f"\n{'─'*45}") + print(f"{'Country':<30} {'Total Visits':>12}") + print(f"{'─'*45}") + for country, count in top: + print(f"{country:<30} {count:>12,}") + print(f"{'─'*45}\n") + + +# ─── Main Processor ─────────────────────────────────────────────────────────── + +class IPGeoProcessor: + + def __init__(self): + os.makedirs("./cache", exist_ok=True) + os.makedirs(REPORT_OUTPUT, exist_ok=True) + + self.cache = TwoLevelCache(LRU_MAX_SIZE, DISK_CACHE_PATH) + self.geoip = GeoIPLookup( + db_path = GEOIP_DB_PATH, + api_token = IPINFO_TOKEN or None + ) + self.bot = BotDetector() + self.checkpoint = Checkpoint.load(CHECKPOINT_FILE) + self.report = ReportAggregator(REPORT_OUTPUT) + self.conn = None + self.cursor = None + + def connect(self): + import mysql.connector + self.conn = mysql.connector.connect(**DB_CONFIG) + self.cursor = self.conn.cursor(buffered=False) + log.info("MySQL connected") + + def disconnect(self): + if self.cursor: + self.cursor.close() + if self.conn: + self.conn.close() + log.info("MySQL disconnected") + + def get_total_rows(self) -> int: + self.cursor.execute(f"SELECT COUNT(*) FROM {TABLE_NAME}") + return self.cursor.fetchone()[0] + + def build_select(self) -> str: + """Build SELECT query based on available columns.""" + cols = ["id", IP_COLUMN, DATE_COLUMN] + if UA_COLUMN: + cols.append(UA_COLUMN) + if PATH_COLUMN: + cols.append(PATH_COLUMN) + return ", ".join(cols) + + def resolve_ip(self, ip: str) -> str: + if not ip: + return "Unknown" + cached = self.cache.get(ip) + if cached is not None: + return cached + country = self.geoip.lookup(ip) + self.cache.put(ip, country) + return country + + def process_batch(self, last_id: int) -> Tuple[int, int, int]: + """ + Returns (last_id_processed, rows_in_batch, skipped_count) + """ + cols = self.build_select() + query = f""" + SELECT {cols} + FROM {TABLE_NAME} + WHERE id > {last_id} + ORDER BY id ASC + LIMIT {BATCH_SIZE} + """ + self.cursor.execute(query) + rows = self.cursor.fetchall() + skipped = 0 + + if not rows: + return last_id, 0, 0 + + # Reset per-batch frequency tracking + self.bot.reset_frequency() + + for row in rows: + # Unpack columns dynamically + idx = 0 + row_id = row[idx]; idx += 1 + ip = str(row[idx]).strip() if row[idx] else ""; idx += 1 + visit_date = row[idx]; idx += 1 + ua = str(row[idx]).strip() if UA_COLUMN and idx < len(row) else None + if ua: idx += 1 + path = str(row[idx]).strip() if PATH_COLUMN and idx < len(row) else None + + last_id = row_id + + # Bot / attack detection + is_bot, reason = self.bot.is_bot(ip, ua, path) + if is_bot: + skipped += 1 + continue + + # Resolve IP → country + country = self.resolve_ip(ip) + self.report.add(country, visit_date) + + return last_id, len(rows), skipped + + def run(self): + log.info("=" * 60) + log.info("IP Geo Report Generator starting") + log.info("=" * 60) + log.info(f"Bot detection: UA={'on' if UA_COLUMN else 'off'} | " + f"Path={'on' if PATH_COLUMN else 'off'} | " + f"Frequency threshold={BOT_IP_REQ_THRESHOLD}") + + self.connect() + + total = self.get_total_rows() + start_id = self.checkpoint.last_id + processed = self.checkpoint.rows_processed + batches = self.checkpoint.batches_done + start_time = time.time() + + log.info(f"Total rows : {total:,}") + log.info(f"Resuming at : ID {start_id:,} ({processed:,} already done)") + + try: + while True: + last_id, count, skipped = self.process_batch(start_id) + + if count == 0: + log.info("All rows processed!") + break + + processed += count + batches += 1 + start_id = last_id + elapsed = time.time() - start_time + rate = processed / elapsed if elapsed > 0 else 0 + remaining = (total - processed) / rate if rate > 0 else 0 + + log.info( + f"Batch {batches:>6,} | " + f"{processed:>12,}/{total:,} " + f"({processed/total*100:.1f}%) | " + f"Skipped(bot): {skipped:,} | " + f"{rate:>8,.0f} rows/s | " + f"ETA {remaining/3600:.1f}h" + ) + + # Save checkpoint every 10 batches + if batches % 10 == 0: + self.checkpoint.last_id = start_id + self.checkpoint.rows_processed = processed + self.checkpoint.batches_done = batches + self.checkpoint.save(CHECKPOINT_FILE) + self.report.save_partial() + log.info("Checkpoint saved") + + except KeyboardInterrupt: + log.warning("Interrupted! Saving checkpoint...") + self.checkpoint.last_id = start_id + self.checkpoint.rows_processed = processed + self.checkpoint.batches_done = batches + self.checkpoint.save(CHECKPOINT_FILE) + self.report.save_partial() + log.info("Re-run to resume.") + + except Exception as e: + log.error(f"Error: {e}", exc_info=True) + self.checkpoint.save(CHECKPOINT_FILE) + self.report.save_partial() + raise + + finally: + self.disconnect() + self.cache.close() + self.geoip.close() + + # Print bot detection summary + bot_report = self.bot.report() + log.info(f"Bot detection summary: {bot_report}") + + # Export reports + csv_path = self.report.export_csv() + summary_path = self.report.export_summary() + self.report.print_top(20) + + # Print bot stats + bot_report = self.bot.report() + print(f"\n{'─'*45}") + print("Bot Detection Summary") + print(f"{'─'*45}") + print(f"Total rows checked : {bot_report['total']:>12,}") + print(f"Skipped (total) : {bot_report['skipped_total']:>12,} ({bot_report['skip_rate']})") + print(f" Private IP : {bot_report['skipped_private']:>12,}") + print(f" High frequency : {bot_report['skipped_freq']:>12,}") + print(f" Bot user agent : {bot_report['skipped_ua']:>12,}") + print(f" Attack path : {bot_report['skipped_path']:>12,}") + print(f"Unique bot IPs : {bot_report['bot_ips_found']:>12,}") + print(f"{'─'*45}\n") + + # Cleanup checkpoint + if os.path.exists(CHECKPOINT_FILE): + os.remove(CHECKPOINT_FILE) + + elapsed = time.time() - start_time + log.info(f"Completed in {elapsed/3600:.2f}h") + log.info(f"Reports: {csv_path}, {summary_path}") + log.info(f"Cache: {self.cache.stats()}") + + +# ─── Entry Point ────────────────────────────────────────────────────────────── + +if __name__ == "__main__": + processor = IPGeoProcessor() + processor.run() diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..0e925d2 --- /dev/null +++ b/requirements.txt @@ -0,0 +1,3 @@ +mysql-connector-python +maxminddb +matplotlib \ No newline at end of file