146 lines
5.4 KiB
Python
146 lines
5.4 KiB
Python
"""Minimaler GDELT-2.0-Tracker: pollt den 15-Minuten-Feed und meldet neue Events.
|
|
|
|
python tracker.py --country DE --root-codes 14 18 19 20 --once
|
|
python tracker.py --country DE --min-mentions 5 # Endlosschleife
|
|
|
|
Kein API-Key noetig. GDELT veroeffentlicht alle 15 Minuten eine neue
|
|
export.CSV.zip; lastupdate.txt zeigt auf die juengste.
|
|
"""
|
|
|
|
import argparse
|
|
import io
|
|
import json
|
|
import time
|
|
import urllib.error
|
|
import urllib.request
|
|
import zipfile
|
|
from pathlib import Path
|
|
|
|
import pandas as pd
|
|
|
|
LASTUPDATE = "http://data.gdeltproject.org/gdeltv2/lastupdate.txt"
|
|
STATE = Path(__file__).with_name(".tracker_state.json")
|
|
|
|
# Spaltennamen der GDELT 2.0 Event-Tabelle (61 Spalten, Datei hat keinen Header)
|
|
COLUMNS = [
|
|
'GLOBALEVENTID', 'SQLDATE', 'MonthYear', 'Year', 'FractionDate',
|
|
'Actor1Code', 'Actor1Name', 'Actor1CountryCode', 'Actor1KnownGroupCode',
|
|
'Actor1EthnicCode', 'Actor1Religion1Code', 'Actor1Religion2Code',
|
|
'Actor1Type1Code', 'Actor1Type2Code', 'Actor1Type3Code',
|
|
'Actor2Code', 'Actor2Name', 'Actor2CountryCode', 'Actor2KnownGroupCode',
|
|
'Actor2EthnicCode', 'Actor2Religion1Code', 'Actor2Religion2Code',
|
|
'Actor2Type1Code', 'Actor2Type2Code', 'Actor2Type3Code',
|
|
'IsRootEvent', 'EventCode', 'EventBaseCode', 'EventRootCode', 'QuadClass',
|
|
'GoldsteinScale', 'NumMentions', 'NumSources', 'NumArticles', 'AvgTone',
|
|
'Actor1Geo_Type', 'Actor1Geo_FullName', 'Actor1Geo_CountryCode',
|
|
'Actor1Geo_ADM1Code', 'Actor1Geo_ADM2Code', 'Actor1Geo_Lat',
|
|
'Actor1Geo_Long', 'Actor1Geo_FeatureID',
|
|
'Actor2Geo_Type', 'Actor2Geo_FullName', 'Actor2Geo_CountryCode',
|
|
'Actor2Geo_ADM1Code', 'Actor2Geo_ADM2Code', 'Actor2Geo_Lat',
|
|
'Actor2Geo_Long', 'Actor2Geo_FeatureID',
|
|
'ActionGeo_Type', 'ActionGeo_FullName', 'ActionGeo_CountryCode',
|
|
'ActionGeo_ADM1Code', 'ActionGeo_ADM2Code', 'ActionGeo_Lat',
|
|
'ActionGeo_Long', 'ActionGeo_FeatureID',
|
|
'DATEADDED', 'SOURCEURL',
|
|
]
|
|
|
|
NUMERIC = ['GoldsteinScale', 'NumMentions', 'NumSources', 'NumArticles',
|
|
'AvgTone', 'ActionGeo_Lat', 'ActionGeo_Long']
|
|
|
|
|
|
def _get(url, timeout=60):
|
|
with urllib.request.urlopen(url, timeout=timeout) as r:
|
|
return r.read()
|
|
|
|
|
|
def latest_export_url():
|
|
"""URL der juengsten export.CSV.zip laut lastupdate.txt."""
|
|
lines = _get(LASTUPDATE, timeout=30).decode().split()
|
|
return next(u for u in lines if u.endswith("export.CSV.zip"))
|
|
|
|
|
|
def fetch_events(url):
|
|
"""Laedt ein 15-Minuten-Paket und gibt es als DataFrame zurueck."""
|
|
z = zipfile.ZipFile(io.BytesIO(_get(url)))
|
|
df = pd.read_csv(z.open(z.namelist()[0]), sep='\t', header=None,
|
|
names=COLUMNS, dtype=str)
|
|
for col in NUMERIC:
|
|
df[col] = pd.to_numeric(df[col], errors='coerce')
|
|
return df
|
|
|
|
|
|
def filter_events(df, country=None, root_codes=(), min_mentions=1,
|
|
max_tone=None):
|
|
if country:
|
|
df = df[df['ActionGeo_CountryCode'] == country]
|
|
if root_codes:
|
|
df = df[df['EventRootCode'].isin(root_codes)]
|
|
if min_mentions:
|
|
df = df[df['NumMentions'] >= min_mentions]
|
|
if max_tone is not None:
|
|
df = df[df['AvgTone'] <= max_tone]
|
|
return df.sort_values('NumMentions', ascending=False)
|
|
|
|
|
|
def report(df):
|
|
df = df.fillna({'Actor1Name': '?', 'Actor2Name': '?'})
|
|
for _, e in df.iterrows():
|
|
print(f"[{e.DATEADDED}] {e.EventCode} goldstein={e.GoldsteinScale} "
|
|
f"tone={e.AvgTone:.1f} mentions={int(e.NumMentions)}")
|
|
print(f" {e.Actor1Name} -> {e.Actor2Name} "
|
|
f"@ {e.ActionGeo_FullName}")
|
|
print(f" {e.SOURCEURL}")
|
|
|
|
|
|
def load_seen():
|
|
if STATE.exists():
|
|
return set(json.loads(STATE.read_text()))
|
|
return set()
|
|
|
|
|
|
def save_seen(seen):
|
|
# nur die letzten paar tausend IDs behalten, sonst waechst die Datei ewig
|
|
STATE.write_text(json.dumps(sorted(seen)[-20000:]))
|
|
|
|
|
|
def main():
|
|
p = argparse.ArgumentParser(description=__doc__)
|
|
p.add_argument('--country', help='FIPS-Laendercode des Ereignisorts, z.B. GM fuer Deutschland')
|
|
p.add_argument('--root-codes', nargs='*', default=[],
|
|
help='CAMEO EventRootCodes, z.B. 14 (Protest) 18 (Assault) 19 (Fight)')
|
|
p.add_argument('--min-mentions', type=int, default=1)
|
|
p.add_argument('--max-tone', type=float, default=None,
|
|
help='nur Events mit AvgTone <= X (negativ = schlechte Nachrichten)')
|
|
p.add_argument('--interval', type=int, default=900, help='Sekunden zwischen Polls')
|
|
p.add_argument('--once', action='store_true')
|
|
args = p.parse_args()
|
|
|
|
seen = load_seen()
|
|
while True:
|
|
try:
|
|
url = latest_export_url()
|
|
df = fetch_events(url)
|
|
except urllib.error.HTTPError as err:
|
|
# lastupdate.txt zeigt gelegentlich auf ein Paket, das noch nicht
|
|
# hochgeladen ist -> beim naechsten Durchlauf erneut versuchen
|
|
print(f"noch nicht verfuegbar ({err.code}), warte...")
|
|
except urllib.error.URLError as err:
|
|
print(f"Netzwerkfehler: {err.reason}")
|
|
else:
|
|
hits = filter_events(df, args.country, args.root_codes,
|
|
args.min_mentions, args.max_tone)
|
|
new = hits[~hits['GLOBALEVENTID'].isin(seen)]
|
|
print(f"{url.rsplit('/', 1)[-1]}: {len(df)} Events, "
|
|
f"{len(hits)} passend, {len(new)} neu")
|
|
report(new)
|
|
seen.update(new['GLOBALEVENTID'])
|
|
save_seen(seen)
|
|
|
|
if args.once:
|
|
break
|
|
time.sleep(args.interval)
|
|
|
|
|
|
if __name__ == '__main__':
|
|
main()
|