init
This commit is contained in:
+120
@@ -0,0 +1,120 @@
|
||||
"""Laedt alle GDELT-2.0-Events eines Tages und filtert auf Europa.
|
||||
|
||||
python load_europe.py # heute (UTC), bis zum letzten Paket
|
||||
python load_europe.py --date 20260901 # ein voller vergangener Tag
|
||||
python load_europe.py --no-transcontinental # ohne Russland und Tuerkei
|
||||
|
||||
Schreibt events_europe_<datum>.csv.gz ins Projektverzeichnis.
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import datetime as dt
|
||||
import gzip
|
||||
import io
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
import zipfile
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
|
||||
import pandas as pd
|
||||
|
||||
from tracker import COLUMNS, NUMERIC
|
||||
|
||||
BASE = "http://data.gdeltproject.org/gdeltv2/{}.export.CSV.zip"
|
||||
|
||||
# FIPS-10-4-Codes, wie sie in ActionGeo_CountryCode stehen (NICHT ISO!)
|
||||
EUROPE = {
|
||||
'AL': 'Albanien', 'AN': 'Andorra', 'AU': 'Oesterreich', 'BE': 'Belgien',
|
||||
'BK': 'Bosnien-Herzegowina', 'BO': 'Belarus', 'BU': 'Bulgarien',
|
||||
'CY': 'Zypern', 'DA': 'Daenemark', 'EI': 'Irland', 'EN': 'Estland',
|
||||
'EZ': 'Tschechien', 'FI': 'Finnland', 'FO': 'Faeroeer', 'FR': 'Frankreich',
|
||||
'GI': 'Gibraltar', 'GK': 'Guernsey', 'GM': 'Deutschland', 'GR': 'Griechenland',
|
||||
'HR': 'Kroatien', 'HU': 'Ungarn', 'IC': 'Island', 'IM': 'Isle of Man',
|
||||
'IT': 'Italien', 'JE': 'Jersey', 'KV': 'Kosovo', 'LG': 'Lettland',
|
||||
'LH': 'Litauen', 'LO': 'Slowakei', 'LS': 'Liechtenstein', 'LU': 'Luxemburg',
|
||||
'MD': 'Moldau', 'MJ': 'Montenegro', 'MK': 'Nordmazedonien', 'MN': 'Monaco',
|
||||
'MT': 'Malta', 'NL': 'Niederlande', 'NO': 'Norwegen', 'PL': 'Polen',
|
||||
'PO': 'Portugal', 'RI': 'Serbien', 'RO': 'Rumaenien', 'SI': 'Slowenien',
|
||||
'SM': 'San Marino', 'SP': 'Spanien', 'SV': 'Svalbard', 'SW': 'Schweden',
|
||||
'SZ': 'Schweiz', 'UK': 'Vereinigtes Koenigreich', 'UP': 'Ukraine',
|
||||
'VT': 'Vatikanstadt',
|
||||
}
|
||||
# transkontinental - je nach Fragestellung mitzaehlen oder nicht
|
||||
TRANSCONTINENTAL = {'RS': 'Russland', 'TU': 'Tuerkei'}
|
||||
|
||||
|
||||
def slots(date, until=None):
|
||||
"""Alle 15-Minuten-Zeitstempel eines Tages, optional bis 'until' (UTC)."""
|
||||
start = dt.datetime.combine(date, dt.time.min, tzinfo=dt.timezone.utc)
|
||||
out = []
|
||||
for i in range(96):
|
||||
ts = start + dt.timedelta(minutes=15 * i)
|
||||
if until and ts > until:
|
||||
break
|
||||
out.append(ts.strftime("%Y%m%d%H%M%S"))
|
||||
return out
|
||||
|
||||
|
||||
def fetch_slot(stamp):
|
||||
"""Ein Paket laden. Gibt (DataFrame|None, bytes) zurueck; None bei 404."""
|
||||
try:
|
||||
with urllib.request.urlopen(BASE.format(stamp), timeout=90) as r:
|
||||
raw = r.read()
|
||||
except urllib.error.HTTPError as err:
|
||||
if err.code == 404:
|
||||
return None, 0 # Slot fehlt oder noch nicht hochgeladen
|
||||
raise
|
||||
z = zipfile.ZipFile(io.BytesIO(raw))
|
||||
df = pd.read_csv(z.open(z.namelist()[0]), sep='\t', header=None,
|
||||
names=COLUMNS, dtype=str)
|
||||
return df, len(raw)
|
||||
|
||||
|
||||
def main():
|
||||
p = argparse.ArgumentParser(description=__doc__)
|
||||
p.add_argument('--date', help='YYYYMMDD, default heute (UTC)')
|
||||
p.add_argument('--no-transcontinental', action='store_true',
|
||||
help='Russland und Tuerkei ausschliessen')
|
||||
p.add_argument('--workers', type=int, default=8)
|
||||
p.add_argument('--out', help='Zieldatei, default events_europe_<datum>.csv.gz')
|
||||
args = p.parse_args()
|
||||
|
||||
now = dt.datetime.now(dt.timezone.utc)
|
||||
date = (dt.datetime.strptime(args.date, "%Y%m%d").date() if args.date
|
||||
else now.date())
|
||||
stamps = slots(date, until=now if date == now.date() else None)
|
||||
|
||||
codes = dict(EUROPE)
|
||||
if not args.no_transcontinental:
|
||||
codes.update(TRANSCONTINENTAL)
|
||||
|
||||
print(f"{date}: {len(stamps)} Pakete a 15 Minuten")
|
||||
with ThreadPoolExecutor(max_workers=args.workers) as pool:
|
||||
results = list(pool.map(fetch_slot, stamps))
|
||||
|
||||
frames = [df for df, _ in results if df is not None]
|
||||
missing = sum(1 for df, _ in results if df is None)
|
||||
downloaded = sum(n for _, n in results)
|
||||
|
||||
world = pd.concat(frames, ignore_index=True)
|
||||
europe = world[world['ActionGeo_CountryCode'].isin(codes)].copy()
|
||||
for col in NUMERIC:
|
||||
europe[col] = pd.to_numeric(europe[col], errors='coerce')
|
||||
europe['ActionGeo_Country'] = europe['ActionGeo_CountryCode'].map(codes)
|
||||
|
||||
out = args.out or f"events_europe_{date:%Y%m%d}.csv.gz"
|
||||
europe.to_csv(out, index=False, compression='gzip')
|
||||
|
||||
print(f"geladen: {len(world):>8,} Events weltweit "
|
||||
f"({downloaded / 1e6:.1f} MB gezippt, {missing} Pakete fehlten)")
|
||||
print(f"Europa: {len(europe):>8,} Events "
|
||||
f"({len(europe) / len(world):.1%})")
|
||||
print(f"eindeutige Event-IDs: {europe['GLOBALEVENTID'].nunique():,}")
|
||||
print(f"geschrieben: {out}")
|
||||
print()
|
||||
print("Top-Laender:")
|
||||
print(europe['ActionGeo_Country'].value_counts().head(12).to_string())
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
Reference in New Issue
Block a user