"""Minimaler GDELT-2.0-Tracker: pollt den 15-Minuten-Feed und meldet neue Events. python tracker.py --country DE --root-codes 14 18 19 20 --once python tracker.py --country DE --min-mentions 5 # Endlosschleife Kein API-Key noetig. GDELT veroeffentlicht alle 15 Minuten eine neue export.CSV.zip; lastupdate.txt zeigt auf die juengste. """ import argparse import io import json import time import urllib.error import urllib.request import zipfile from pathlib import Path import pandas as pd LASTUPDATE = "http://data.gdeltproject.org/gdeltv2/lastupdate.txt" STATE = Path(__file__).with_name(".tracker_state.json") # Spaltennamen der GDELT 2.0 Event-Tabelle (61 Spalten, Datei hat keinen Header) COLUMNS = [ 'GLOBALEVENTID', 'SQLDATE', 'MonthYear', 'Year', 'FractionDate', 'Actor1Code', 'Actor1Name', 'Actor1CountryCode', 'Actor1KnownGroupCode', 'Actor1EthnicCode', 'Actor1Religion1Code', 'Actor1Religion2Code', 'Actor1Type1Code', 'Actor1Type2Code', 'Actor1Type3Code', 'Actor2Code', 'Actor2Name', 'Actor2CountryCode', 'Actor2KnownGroupCode', 'Actor2EthnicCode', 'Actor2Religion1Code', 'Actor2Religion2Code', 'Actor2Type1Code', 'Actor2Type2Code', 'Actor2Type3Code', 'IsRootEvent', 'EventCode', 'EventBaseCode', 'EventRootCode', 'QuadClass', 'GoldsteinScale', 'NumMentions', 'NumSources', 'NumArticles', 'AvgTone', 'Actor1Geo_Type', 'Actor1Geo_FullName', 'Actor1Geo_CountryCode', 'Actor1Geo_ADM1Code', 'Actor1Geo_ADM2Code', 'Actor1Geo_Lat', 'Actor1Geo_Long', 'Actor1Geo_FeatureID', 'Actor2Geo_Type', 'Actor2Geo_FullName', 'Actor2Geo_CountryCode', 'Actor2Geo_ADM1Code', 'Actor2Geo_ADM2Code', 'Actor2Geo_Lat', 'Actor2Geo_Long', 'Actor2Geo_FeatureID', 'ActionGeo_Type', 'ActionGeo_FullName', 'ActionGeo_CountryCode', 'ActionGeo_ADM1Code', 'ActionGeo_ADM2Code', 'ActionGeo_Lat', 'ActionGeo_Long', 'ActionGeo_FeatureID', 'DATEADDED', 'SOURCEURL', ] NUMERIC = ['GoldsteinScale', 'NumMentions', 'NumSources', 'NumArticles', 'AvgTone', 'ActionGeo_Lat', 'ActionGeo_Long'] def _get(url, timeout=60): with urllib.request.urlopen(url, timeout=timeout) as r: return r.read() def latest_export_url(): """URL der juengsten export.CSV.zip laut lastupdate.txt.""" lines = _get(LASTUPDATE, timeout=30).decode().split() return next(u for u in lines if u.endswith("export.CSV.zip")) def fetch_events(url): """Laedt ein 15-Minuten-Paket und gibt es als DataFrame zurueck.""" z = zipfile.ZipFile(io.BytesIO(_get(url))) df = pd.read_csv(z.open(z.namelist()[0]), sep='\t', header=None, names=COLUMNS, dtype=str) for col in NUMERIC: df[col] = pd.to_numeric(df[col], errors='coerce') return df def filter_events(df, country=None, root_codes=(), min_mentions=1, max_tone=None): if country: df = df[df['ActionGeo_CountryCode'] == country] if root_codes: df = df[df['EventRootCode'].isin(root_codes)] if min_mentions: df = df[df['NumMentions'] >= min_mentions] if max_tone is not None: df = df[df['AvgTone'] <= max_tone] return df.sort_values('NumMentions', ascending=False) def report(df): df = df.fillna({'Actor1Name': '?', 'Actor2Name': '?'}) for _, e in df.iterrows(): print(f"[{e.DATEADDED}] {e.EventCode} goldstein={e.GoldsteinScale} " f"tone={e.AvgTone:.1f} mentions={int(e.NumMentions)}") print(f" {e.Actor1Name} -> {e.Actor2Name} " f"@ {e.ActionGeo_FullName}") print(f" {e.SOURCEURL}") def load_seen(): if STATE.exists(): return set(json.loads(STATE.read_text())) return set() def save_seen(seen): # nur die letzten paar tausend IDs behalten, sonst waechst die Datei ewig STATE.write_text(json.dumps(sorted(seen)[-20000:])) def main(): p = argparse.ArgumentParser(description=__doc__) p.add_argument('--country', help='FIPS-Laendercode des Ereignisorts, z.B. GM fuer Deutschland') p.add_argument('--root-codes', nargs='*', default=[], help='CAMEO EventRootCodes, z.B. 14 (Protest) 18 (Assault) 19 (Fight)') p.add_argument('--min-mentions', type=int, default=1) p.add_argument('--max-tone', type=float, default=None, help='nur Events mit AvgTone <= X (negativ = schlechte Nachrichten)') p.add_argument('--interval', type=int, default=900, help='Sekunden zwischen Polls') p.add_argument('--once', action='store_true') args = p.parse_args() seen = load_seen() while True: try: url = latest_export_url() df = fetch_events(url) except urllib.error.HTTPError as err: # lastupdate.txt zeigt gelegentlich auf ein Paket, das noch nicht # hochgeladen ist -> beim naechsten Durchlauf erneut versuchen print(f"noch nicht verfuegbar ({err.code}), warte...") except urllib.error.URLError as err: print(f"Netzwerkfehler: {err.reason}") else: hits = filter_events(df, args.country, args.root_codes, args.min_mentions, args.max_tone) new = hits[~hits['GLOBALEVENTID'].isin(seen)] print(f"{url.rsplit('/', 1)[-1]}: {len(df)} Events, " f"{len(hits)} passend, {len(new)} neu") report(new) seen.update(new['GLOBALEVENTID']) save_seen(seen) if args.once: break time.sleep(args.interval) if __name__ == '__main__': main()