"""Laedt alle GDELT-2.0-Events eines Tages und filtert auf Europa. python load_europe.py # heute (UTC), bis zum letzten Paket python load_europe.py --date 20260901 # ein voller vergangener Tag python load_europe.py --no-transcontinental # ohne Russland und Tuerkei Schreibt events_europe_.csv.gz ins Projektverzeichnis. """ import argparse import datetime as dt import gzip import io import urllib.error import urllib.request import zipfile from concurrent.futures import ThreadPoolExecutor import pandas as pd from tracker import COLUMNS, NUMERIC BASE = "http://data.gdeltproject.org/gdeltv2/{}.export.CSV.zip" # FIPS-10-4-Codes, wie sie in ActionGeo_CountryCode stehen (NICHT ISO!) EUROPE = { 'AL': 'Albanien', 'AN': 'Andorra', 'AU': 'Oesterreich', 'BE': 'Belgien', 'BK': 'Bosnien-Herzegowina', 'BO': 'Belarus', 'BU': 'Bulgarien', 'CY': 'Zypern', 'DA': 'Daenemark', 'EI': 'Irland', 'EN': 'Estland', 'EZ': 'Tschechien', 'FI': 'Finnland', 'FO': 'Faeroeer', 'FR': 'Frankreich', 'GI': 'Gibraltar', 'GK': 'Guernsey', 'GM': 'Deutschland', 'GR': 'Griechenland', 'HR': 'Kroatien', 'HU': 'Ungarn', 'IC': 'Island', 'IM': 'Isle of Man', 'IT': 'Italien', 'JE': 'Jersey', 'KV': 'Kosovo', 'LG': 'Lettland', 'LH': 'Litauen', 'LO': 'Slowakei', 'LS': 'Liechtenstein', 'LU': 'Luxemburg', 'MD': 'Moldau', 'MJ': 'Montenegro', 'MK': 'Nordmazedonien', 'MN': 'Monaco', 'MT': 'Malta', 'NL': 'Niederlande', 'NO': 'Norwegen', 'PL': 'Polen', 'PO': 'Portugal', 'RI': 'Serbien', 'RO': 'Rumaenien', 'SI': 'Slowenien', 'SM': 'San Marino', 'SP': 'Spanien', 'SV': 'Svalbard', 'SW': 'Schweden', 'SZ': 'Schweiz', 'UK': 'Vereinigtes Koenigreich', 'UP': 'Ukraine', 'VT': 'Vatikanstadt', } # transkontinental - je nach Fragestellung mitzaehlen oder nicht TRANSCONTINENTAL = {'RS': 'Russland', 'TU': 'Tuerkei'} def slots(date, until=None): """Alle 15-Minuten-Zeitstempel eines Tages, optional bis 'until' (UTC).""" start = dt.datetime.combine(date, dt.time.min, tzinfo=dt.timezone.utc) out = [] for i in range(96): ts = start + dt.timedelta(minutes=15 * i) if until and ts > until: break out.append(ts.strftime("%Y%m%d%H%M%S")) return out def fetch_slot(stamp): """Ein Paket laden. Gibt (DataFrame|None, bytes) zurueck; None bei 404.""" try: with urllib.request.urlopen(BASE.format(stamp), timeout=90) as r: raw = r.read() except urllib.error.HTTPError as err: if err.code == 404: return None, 0 # Slot fehlt oder noch nicht hochgeladen raise z = zipfile.ZipFile(io.BytesIO(raw)) df = pd.read_csv(z.open(z.namelist()[0]), sep='\t', header=None, names=COLUMNS, dtype=str) return df, len(raw) def main(): p = argparse.ArgumentParser(description=__doc__) p.add_argument('--date', help='YYYYMMDD, default heute (UTC)') p.add_argument('--no-transcontinental', action='store_true', help='Russland und Tuerkei ausschliessen') p.add_argument('--workers', type=int, default=8) p.add_argument('--out', help='Zieldatei, default events_europe_.csv.gz') args = p.parse_args() now = dt.datetime.now(dt.timezone.utc) date = (dt.datetime.strptime(args.date, "%Y%m%d").date() if args.date else now.date()) stamps = slots(date, until=now if date == now.date() else None) codes = dict(EUROPE) if not args.no_transcontinental: codes.update(TRANSCONTINENTAL) print(f"{date}: {len(stamps)} Pakete a 15 Minuten") with ThreadPoolExecutor(max_workers=args.workers) as pool: results = list(pool.map(fetch_slot, stamps)) frames = [df for df, _ in results if df is not None] missing = sum(1 for df, _ in results if df is None) downloaded = sum(n for _, n in results) world = pd.concat(frames, ignore_index=True) europe = world[world['ActionGeo_CountryCode'].isin(codes)].copy() for col in NUMERIC: europe[col] = pd.to_numeric(europe[col], errors='coerce') europe['ActionGeo_Country'] = europe['ActionGeo_CountryCode'].map(codes) out = args.out or f"events_europe_{date:%Y%m%d}.csv.gz" europe.to_csv(out, index=False, compression='gzip') print(f"geladen: {len(world):>8,} Events weltweit " f"({downloaded / 1e6:.1f} MB gezippt, {missing} Pakete fehlten)") print(f"Europa: {len(europe):>8,} Events " f"({len(europe) / len(world):.1%})") print(f"eindeutige Event-IDs: {europe['GLOBALEVENTID'].nunique():,}") print(f"geschrieben: {out}") print() print("Top-Laender:") print(europe['ActionGeo_Country'].value_counts().head(12).to_string()) if __name__ == '__main__': main()