Files
gdelt-bot/load_europe.py
T
2026-09-02 14:19:09 +02:00

121 lines
4.7 KiB
Python

"""Laedt alle GDELT-2.0-Events eines Tages und filtert auf Europa.
python load_europe.py # heute (UTC), bis zum letzten Paket
python load_europe.py --date 20260901 # ein voller vergangener Tag
python load_europe.py --no-transcontinental # ohne Russland und Tuerkei
Schreibt events_europe_<datum>.csv.gz ins Projektverzeichnis.
"""
import argparse
import datetime as dt
import gzip
import io
import urllib.error
import urllib.request
import zipfile
from concurrent.futures import ThreadPoolExecutor
import pandas as pd
from tracker import COLUMNS, NUMERIC
BASE = "http://data.gdeltproject.org/gdeltv2/{}.export.CSV.zip"
# FIPS-10-4-Codes, wie sie in ActionGeo_CountryCode stehen (NICHT ISO!)
EUROPE = {
'AL': 'Albanien', 'AN': 'Andorra', 'AU': 'Oesterreich', 'BE': 'Belgien',
'BK': 'Bosnien-Herzegowina', 'BO': 'Belarus', 'BU': 'Bulgarien',
'CY': 'Zypern', 'DA': 'Daenemark', 'EI': 'Irland', 'EN': 'Estland',
'EZ': 'Tschechien', 'FI': 'Finnland', 'FO': 'Faeroeer', 'FR': 'Frankreich',
'GI': 'Gibraltar', 'GK': 'Guernsey', 'GM': 'Deutschland', 'GR': 'Griechenland',
'HR': 'Kroatien', 'HU': 'Ungarn', 'IC': 'Island', 'IM': 'Isle of Man',
'IT': 'Italien', 'JE': 'Jersey', 'KV': 'Kosovo', 'LG': 'Lettland',
'LH': 'Litauen', 'LO': 'Slowakei', 'LS': 'Liechtenstein', 'LU': 'Luxemburg',
'MD': 'Moldau', 'MJ': 'Montenegro', 'MK': 'Nordmazedonien', 'MN': 'Monaco',
'MT': 'Malta', 'NL': 'Niederlande', 'NO': 'Norwegen', 'PL': 'Polen',
'PO': 'Portugal', 'RI': 'Serbien', 'RO': 'Rumaenien', 'SI': 'Slowenien',
'SM': 'San Marino', 'SP': 'Spanien', 'SV': 'Svalbard', 'SW': 'Schweden',
'SZ': 'Schweiz', 'UK': 'Vereinigtes Koenigreich', 'UP': 'Ukraine',
'VT': 'Vatikanstadt',
}
# transkontinental - je nach Fragestellung mitzaehlen oder nicht
TRANSCONTINENTAL = {'RS': 'Russland', 'TU': 'Tuerkei'}
def slots(date, until=None):
"""Alle 15-Minuten-Zeitstempel eines Tages, optional bis 'until' (UTC)."""
start = dt.datetime.combine(date, dt.time.min, tzinfo=dt.timezone.utc)
out = []
for i in range(96):
ts = start + dt.timedelta(minutes=15 * i)
if until and ts > until:
break
out.append(ts.strftime("%Y%m%d%H%M%S"))
return out
def fetch_slot(stamp):
"""Ein Paket laden. Gibt (DataFrame|None, bytes) zurueck; None bei 404."""
try:
with urllib.request.urlopen(BASE.format(stamp), timeout=90) as r:
raw = r.read()
except urllib.error.HTTPError as err:
if err.code == 404:
return None, 0 # Slot fehlt oder noch nicht hochgeladen
raise
z = zipfile.ZipFile(io.BytesIO(raw))
df = pd.read_csv(z.open(z.namelist()[0]), sep='\t', header=None,
names=COLUMNS, dtype=str)
return df, len(raw)
def main():
p = argparse.ArgumentParser(description=__doc__)
p.add_argument('--date', help='YYYYMMDD, default heute (UTC)')
p.add_argument('--no-transcontinental', action='store_true',
help='Russland und Tuerkei ausschliessen')
p.add_argument('--workers', type=int, default=8)
p.add_argument('--out', help='Zieldatei, default events_europe_<datum>.csv.gz')
args = p.parse_args()
now = dt.datetime.now(dt.timezone.utc)
date = (dt.datetime.strptime(args.date, "%Y%m%d").date() if args.date
else now.date())
stamps = slots(date, until=now if date == now.date() else None)
codes = dict(EUROPE)
if not args.no_transcontinental:
codes.update(TRANSCONTINENTAL)
print(f"{date}: {len(stamps)} Pakete a 15 Minuten")
with ThreadPoolExecutor(max_workers=args.workers) as pool:
results = list(pool.map(fetch_slot, stamps))
frames = [df for df, _ in results if df is not None]
missing = sum(1 for df, _ in results if df is None)
downloaded = sum(n for _, n in results)
world = pd.concat(frames, ignore_index=True)
europe = world[world['ActionGeo_CountryCode'].isin(codes)].copy()
for col in NUMERIC:
europe[col] = pd.to_numeric(europe[col], errors='coerce')
europe['ActionGeo_Country'] = europe['ActionGeo_CountryCode'].map(codes)
out = args.out or f"events_europe_{date:%Y%m%d}.csv.gz"
europe.to_csv(out, index=False, compression='gzip')
print(f"geladen: {len(world):>8,} Events weltweit "
f"({downloaded / 1e6:.1f} MB gezippt, {missing} Pakete fehlten)")
print(f"Europa: {len(europe):>8,} Events "
f"({len(europe) / len(world):.1%})")
print(f"eindeutige Event-IDs: {europe['GLOBALEVENTID'].nunique():,}")
print(f"geschrieben: {out}")
print()
print("Top-Laender:")
print(europe['ActionGeo_Country'].value_counts().head(12).to_string())
if __name__ == '__main__':
main()