Files
2026-09-02 14:19:09 +02:00

146 lines
5.4 KiB
Python

"""Minimaler GDELT-2.0-Tracker: pollt den 15-Minuten-Feed und meldet neue Events.
python tracker.py --country DE --root-codes 14 18 19 20 --once
python tracker.py --country DE --min-mentions 5 # Endlosschleife
Kein API-Key noetig. GDELT veroeffentlicht alle 15 Minuten eine neue
export.CSV.zip; lastupdate.txt zeigt auf die juengste.
"""
import argparse
import io
import json
import time
import urllib.error
import urllib.request
import zipfile
from pathlib import Path
import pandas as pd
LASTUPDATE = "http://data.gdeltproject.org/gdeltv2/lastupdate.txt"
STATE = Path(__file__).with_name(".tracker_state.json")
# Spaltennamen der GDELT 2.0 Event-Tabelle (61 Spalten, Datei hat keinen Header)
COLUMNS = [
'GLOBALEVENTID', 'SQLDATE', 'MonthYear', 'Year', 'FractionDate',
'Actor1Code', 'Actor1Name', 'Actor1CountryCode', 'Actor1KnownGroupCode',
'Actor1EthnicCode', 'Actor1Religion1Code', 'Actor1Religion2Code',
'Actor1Type1Code', 'Actor1Type2Code', 'Actor1Type3Code',
'Actor2Code', 'Actor2Name', 'Actor2CountryCode', 'Actor2KnownGroupCode',
'Actor2EthnicCode', 'Actor2Religion1Code', 'Actor2Religion2Code',
'Actor2Type1Code', 'Actor2Type2Code', 'Actor2Type3Code',
'IsRootEvent', 'EventCode', 'EventBaseCode', 'EventRootCode', 'QuadClass',
'GoldsteinScale', 'NumMentions', 'NumSources', 'NumArticles', 'AvgTone',
'Actor1Geo_Type', 'Actor1Geo_FullName', 'Actor1Geo_CountryCode',
'Actor1Geo_ADM1Code', 'Actor1Geo_ADM2Code', 'Actor1Geo_Lat',
'Actor1Geo_Long', 'Actor1Geo_FeatureID',
'Actor2Geo_Type', 'Actor2Geo_FullName', 'Actor2Geo_CountryCode',
'Actor2Geo_ADM1Code', 'Actor2Geo_ADM2Code', 'Actor2Geo_Lat',
'Actor2Geo_Long', 'Actor2Geo_FeatureID',
'ActionGeo_Type', 'ActionGeo_FullName', 'ActionGeo_CountryCode',
'ActionGeo_ADM1Code', 'ActionGeo_ADM2Code', 'ActionGeo_Lat',
'ActionGeo_Long', 'ActionGeo_FeatureID',
'DATEADDED', 'SOURCEURL',
]
NUMERIC = ['GoldsteinScale', 'NumMentions', 'NumSources', 'NumArticles',
'AvgTone', 'ActionGeo_Lat', 'ActionGeo_Long']
def _get(url, timeout=60):
with urllib.request.urlopen(url, timeout=timeout) as r:
return r.read()
def latest_export_url():
"""URL der juengsten export.CSV.zip laut lastupdate.txt."""
lines = _get(LASTUPDATE, timeout=30).decode().split()
return next(u for u in lines if u.endswith("export.CSV.zip"))
def fetch_events(url):
"""Laedt ein 15-Minuten-Paket und gibt es als DataFrame zurueck."""
z = zipfile.ZipFile(io.BytesIO(_get(url)))
df = pd.read_csv(z.open(z.namelist()[0]), sep='\t', header=None,
names=COLUMNS, dtype=str)
for col in NUMERIC:
df[col] = pd.to_numeric(df[col], errors='coerce')
return df
def filter_events(df, country=None, root_codes=(), min_mentions=1,
max_tone=None):
if country:
df = df[df['ActionGeo_CountryCode'] == country]
if root_codes:
df = df[df['EventRootCode'].isin(root_codes)]
if min_mentions:
df = df[df['NumMentions'] >= min_mentions]
if max_tone is not None:
df = df[df['AvgTone'] <= max_tone]
return df.sort_values('NumMentions', ascending=False)
def report(df):
df = df.fillna({'Actor1Name': '?', 'Actor2Name': '?'})
for _, e in df.iterrows():
print(f"[{e.DATEADDED}] {e.EventCode} goldstein={e.GoldsteinScale} "
f"tone={e.AvgTone:.1f} mentions={int(e.NumMentions)}")
print(f" {e.Actor1Name} -> {e.Actor2Name} "
f"@ {e.ActionGeo_FullName}")
print(f" {e.SOURCEURL}")
def load_seen():
if STATE.exists():
return set(json.loads(STATE.read_text()))
return set()
def save_seen(seen):
# nur die letzten paar tausend IDs behalten, sonst waechst die Datei ewig
STATE.write_text(json.dumps(sorted(seen)[-20000:]))
def main():
p = argparse.ArgumentParser(description=__doc__)
p.add_argument('--country', help='FIPS-Laendercode des Ereignisorts, z.B. GM fuer Deutschland')
p.add_argument('--root-codes', nargs='*', default=[],
help='CAMEO EventRootCodes, z.B. 14 (Protest) 18 (Assault) 19 (Fight)')
p.add_argument('--min-mentions', type=int, default=1)
p.add_argument('--max-tone', type=float, default=None,
help='nur Events mit AvgTone <= X (negativ = schlechte Nachrichten)')
p.add_argument('--interval', type=int, default=900, help='Sekunden zwischen Polls')
p.add_argument('--once', action='store_true')
args = p.parse_args()
seen = load_seen()
while True:
try:
url = latest_export_url()
df = fetch_events(url)
except urllib.error.HTTPError as err:
# lastupdate.txt zeigt gelegentlich auf ein Paket, das noch nicht
# hochgeladen ist -> beim naechsten Durchlauf erneut versuchen
print(f"noch nicht verfuegbar ({err.code}), warte...")
except urllib.error.URLError as err:
print(f"Netzwerkfehler: {err.reason}")
else:
hits = filter_events(df, args.country, args.root_codes,
args.min_mentions, args.max_tone)
new = hits[~hits['GLOBALEVENTID'].isin(seen)]
print(f"{url.rsplit('/', 1)[-1]}: {len(df)} Events, "
f"{len(hits)} passend, {len(new)} neu")
report(new)
seen.update(new['GLOBALEVENTID'])
save_seen(seen)
if args.once:
break
time.sleep(args.interval)
if __name__ == '__main__':
main()