init
This commit is contained in:
+145
@@ -0,0 +1,145 @@
|
||||
"""Minimaler GDELT-2.0-Tracker: pollt den 15-Minuten-Feed und meldet neue Events.
|
||||
|
||||
python tracker.py --country DE --root-codes 14 18 19 20 --once
|
||||
python tracker.py --country DE --min-mentions 5 # Endlosschleife
|
||||
|
||||
Kein API-Key noetig. GDELT veroeffentlicht alle 15 Minuten eine neue
|
||||
export.CSV.zip; lastupdate.txt zeigt auf die juengste.
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import io
|
||||
import json
|
||||
import time
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
import zipfile
|
||||
from pathlib import Path
|
||||
|
||||
import pandas as pd
|
||||
|
||||
LASTUPDATE = "http://data.gdeltproject.org/gdeltv2/lastupdate.txt"
|
||||
STATE = Path(__file__).with_name(".tracker_state.json")
|
||||
|
||||
# Spaltennamen der GDELT 2.0 Event-Tabelle (61 Spalten, Datei hat keinen Header)
|
||||
COLUMNS = [
|
||||
'GLOBALEVENTID', 'SQLDATE', 'MonthYear', 'Year', 'FractionDate',
|
||||
'Actor1Code', 'Actor1Name', 'Actor1CountryCode', 'Actor1KnownGroupCode',
|
||||
'Actor1EthnicCode', 'Actor1Religion1Code', 'Actor1Religion2Code',
|
||||
'Actor1Type1Code', 'Actor1Type2Code', 'Actor1Type3Code',
|
||||
'Actor2Code', 'Actor2Name', 'Actor2CountryCode', 'Actor2KnownGroupCode',
|
||||
'Actor2EthnicCode', 'Actor2Religion1Code', 'Actor2Religion2Code',
|
||||
'Actor2Type1Code', 'Actor2Type2Code', 'Actor2Type3Code',
|
||||
'IsRootEvent', 'EventCode', 'EventBaseCode', 'EventRootCode', 'QuadClass',
|
||||
'GoldsteinScale', 'NumMentions', 'NumSources', 'NumArticles', 'AvgTone',
|
||||
'Actor1Geo_Type', 'Actor1Geo_FullName', 'Actor1Geo_CountryCode',
|
||||
'Actor1Geo_ADM1Code', 'Actor1Geo_ADM2Code', 'Actor1Geo_Lat',
|
||||
'Actor1Geo_Long', 'Actor1Geo_FeatureID',
|
||||
'Actor2Geo_Type', 'Actor2Geo_FullName', 'Actor2Geo_CountryCode',
|
||||
'Actor2Geo_ADM1Code', 'Actor2Geo_ADM2Code', 'Actor2Geo_Lat',
|
||||
'Actor2Geo_Long', 'Actor2Geo_FeatureID',
|
||||
'ActionGeo_Type', 'ActionGeo_FullName', 'ActionGeo_CountryCode',
|
||||
'ActionGeo_ADM1Code', 'ActionGeo_ADM2Code', 'ActionGeo_Lat',
|
||||
'ActionGeo_Long', 'ActionGeo_FeatureID',
|
||||
'DATEADDED', 'SOURCEURL',
|
||||
]
|
||||
|
||||
NUMERIC = ['GoldsteinScale', 'NumMentions', 'NumSources', 'NumArticles',
|
||||
'AvgTone', 'ActionGeo_Lat', 'ActionGeo_Long']
|
||||
|
||||
|
||||
def _get(url, timeout=60):
|
||||
with urllib.request.urlopen(url, timeout=timeout) as r:
|
||||
return r.read()
|
||||
|
||||
|
||||
def latest_export_url():
|
||||
"""URL der juengsten export.CSV.zip laut lastupdate.txt."""
|
||||
lines = _get(LASTUPDATE, timeout=30).decode().split()
|
||||
return next(u for u in lines if u.endswith("export.CSV.zip"))
|
||||
|
||||
|
||||
def fetch_events(url):
|
||||
"""Laedt ein 15-Minuten-Paket und gibt es als DataFrame zurueck."""
|
||||
z = zipfile.ZipFile(io.BytesIO(_get(url)))
|
||||
df = pd.read_csv(z.open(z.namelist()[0]), sep='\t', header=None,
|
||||
names=COLUMNS, dtype=str)
|
||||
for col in NUMERIC:
|
||||
df[col] = pd.to_numeric(df[col], errors='coerce')
|
||||
return df
|
||||
|
||||
|
||||
def filter_events(df, country=None, root_codes=(), min_mentions=1,
|
||||
max_tone=None):
|
||||
if country:
|
||||
df = df[df['ActionGeo_CountryCode'] == country]
|
||||
if root_codes:
|
||||
df = df[df['EventRootCode'].isin(root_codes)]
|
||||
if min_mentions:
|
||||
df = df[df['NumMentions'] >= min_mentions]
|
||||
if max_tone is not None:
|
||||
df = df[df['AvgTone'] <= max_tone]
|
||||
return df.sort_values('NumMentions', ascending=False)
|
||||
|
||||
|
||||
def report(df):
|
||||
df = df.fillna({'Actor1Name': '?', 'Actor2Name': '?'})
|
||||
for _, e in df.iterrows():
|
||||
print(f"[{e.DATEADDED}] {e.EventCode} goldstein={e.GoldsteinScale} "
|
||||
f"tone={e.AvgTone:.1f} mentions={int(e.NumMentions)}")
|
||||
print(f" {e.Actor1Name} -> {e.Actor2Name} "
|
||||
f"@ {e.ActionGeo_FullName}")
|
||||
print(f" {e.SOURCEURL}")
|
||||
|
||||
|
||||
def load_seen():
|
||||
if STATE.exists():
|
||||
return set(json.loads(STATE.read_text()))
|
||||
return set()
|
||||
|
||||
|
||||
def save_seen(seen):
|
||||
# nur die letzten paar tausend IDs behalten, sonst waechst die Datei ewig
|
||||
STATE.write_text(json.dumps(sorted(seen)[-20000:]))
|
||||
|
||||
|
||||
def main():
|
||||
p = argparse.ArgumentParser(description=__doc__)
|
||||
p.add_argument('--country', help='FIPS-Laendercode des Ereignisorts, z.B. GM fuer Deutschland')
|
||||
p.add_argument('--root-codes', nargs='*', default=[],
|
||||
help='CAMEO EventRootCodes, z.B. 14 (Protest) 18 (Assault) 19 (Fight)')
|
||||
p.add_argument('--min-mentions', type=int, default=1)
|
||||
p.add_argument('--max-tone', type=float, default=None,
|
||||
help='nur Events mit AvgTone <= X (negativ = schlechte Nachrichten)')
|
||||
p.add_argument('--interval', type=int, default=900, help='Sekunden zwischen Polls')
|
||||
p.add_argument('--once', action='store_true')
|
||||
args = p.parse_args()
|
||||
|
||||
seen = load_seen()
|
||||
while True:
|
||||
try:
|
||||
url = latest_export_url()
|
||||
df = fetch_events(url)
|
||||
except urllib.error.HTTPError as err:
|
||||
# lastupdate.txt zeigt gelegentlich auf ein Paket, das noch nicht
|
||||
# hochgeladen ist -> beim naechsten Durchlauf erneut versuchen
|
||||
print(f"noch nicht verfuegbar ({err.code}), warte...")
|
||||
except urllib.error.URLError as err:
|
||||
print(f"Netzwerkfehler: {err.reason}")
|
||||
else:
|
||||
hits = filter_events(df, args.country, args.root_codes,
|
||||
args.min_mentions, args.max_tone)
|
||||
new = hits[~hits['GLOBALEVENTID'].isin(seen)]
|
||||
print(f"{url.rsplit('/', 1)[-1]}: {len(df)} Events, "
|
||||
f"{len(hits)} passend, {len(new)} neu")
|
||||
report(new)
|
||||
seen.update(new['GLOBALEVENTID'])
|
||||
save_seen(seen)
|
||||
|
||||
if args.once:
|
||||
break
|
||||
time.sleep(args.interval)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
Reference in New Issue
Block a user