-
-
Notifications
You must be signed in to change notification settings - Fork 1
Expand file tree
/
Copy pathserver.py
More file actions
8682 lines (7710 loc) · 352 KB
/
Copy pathserver.py
File metadata and controls
8682 lines (7710 loc) · 352 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
#!/usr/bin/env python3
"""Global Command View - local server.
Serves the web client and proxies the public data feeds it uses. The proxy
exists for two reasons: the upstream APIs do not send permissive CORS headers,
and they are all rate limited, so responses are cached here instead of being
hammered by every browser tab.
Everything it talks to is public and key-free:
flights OpenSky Network https://opensky-network.org/
vessels Digitraffic AIS https://meri.digitraffic.fi/
cables TeleGeography https://www.submarinecablemap.com/
cameras Digitraffic weathercam
python server.py --port 8787
"""
import argparse
import gzip
import html as _html
import io
import array
import collections
import csv
import datetime
import email.utils
import json
import math
import os
import ssl
import re
import stat as stat_module
import threading
import time
import urllib.error
import urllib.parse
import urllib.request
import xml.etree.ElementTree as xml_tree
import zipfile
import webbrowser
from http.server import SimpleHTTPRequestHandler, ThreadingHTTPServer
VERSION = "6.0"
BUILT = "2026-09-08"
ROOT = os.path.dirname(os.path.abspath(__file__))
WEB = os.path.join(ROOT, "web")
CACHE_DIR = os.path.join(ROOT, ".cache")
# Nominatim and Overpass both ask that a client identify itself and say where
# to complain. A project URL is the contact route for something with no
# operator - anyone whose service this is rude to can find the issue tracker.
USER_AGENT = ("global-command-view/%s "
"(+https://github.com/Mochi-game/global-command-view)" % VERSION)
TIMEOUT = 30
# Optional API keys, kept out of git. See keys.example.json. Everything works
# without it; a key only adds camera networks that refuse anonymous callers.
KEYS = {}
_keys_path = os.path.join(ROOT, "keys.json")
if os.path.exists(_keys_path):
try:
with open(_keys_path, encoding="utf-8") as fh:
KEYS = {k: v for k, v in json.load(fh).items() if v}
except Exception as exc: # noqa: BLE001 - a broken key file must not stop the server
print(f"keys.json unreadable: {exc}")
# name -> (url, memory ttl seconds, disk ttl seconds or 0 for memory only)
FEEDS = {
"vessels": ("https://meri.digitraffic.fi/api/ais/v1/locations", 15, 0),
"vessel-meta": ("https://meri.digitraffic.fi/api/ais/v1/vessels", 3600, 86400),
"cameras-fi": ("https://tie.digitraffic.fi/api/weathercam/v1/stations", 900, 86400),
"cameras-uk": ("https://api.tfl.gov.uk/Place/Type/JamCam", 900, 86400),
"cables": (
"https://www.submarinecablemap.com/api/v3/cable/cable-geo.json",
86400,
604800,
),
"landings": (
"https://www.submarinecablemap.com/api/v3/landing-point/landing-point-geo.json",
86400,
604800,
),
# Orbital elements, not positions: the browser propagates them itself. They
# are re-issued roughly daily, and CelesTrak asks callers not to poll harder.
"satellites": (
"https://celestrak.org/NORAD/elements/gp.php?GROUP=active&FORMAT=tle",
7200,
86400,
),
# Every quake worldwide in the last week, magnitude 2.5 and up.
"quakes": (
"https://earthquake.usgs.gov/earthquakes/feed/v1.0/summary/2.5_week.geojson",
600,
86400,
),
}
TEXT_FEEDS = {"satellites"}
# Worldwide AIS, if the user registered for it. Digitraffic covers the Baltic and
# nothing else, so without this key the sea is empty outside northern Europe.
AIS = None
FLIGHTS_URL = "https://opensky-network.org/api/states/all"
FLIGHTS_TTL = 12 # OpenSky refuses more than one anonymous poll per ~10s
# OpenSky hands out ~400 anonymous credits per IP per day. When they run out it
# answers 429 with a retry hint measured in hours, so the air layer falls back to
# the community ADS-B feeders at adsb.lol until the quota comes back.
# adsb.lol's feeders are almost all in Europe and North America — it answers with
# nothing at all over the Gulf, Japan or South America. adsb.fi has genuine global
# coverage but serves at most a 250 nm circle per call, so a wide view is stitched
# together from a grid of them.
ADSB_URL = "https://opendata.adsb.fi/api/v2/lat/{lat:.3f}/lon/{lon:.3f}/dist/{radius:.0f}"
# adsb.lol behind it, in that order and not the other way round: adsb.lol's
# feeders are almost all in Europe and North America, so it is the better
# second opinion over Sweden and no help at all over the Gulf. Two community
# networks having a bad afternoon at once is rarer than one.
ADSB_URLS = (
ADSB_URL,
"https://api.adsb.lol/v2/lat/{lat:.3f}/lon/{lon:.3f}/dist/{radius:.0f}",
)
ADSB_RADIUS_NM = 250
ADSB_MAX_CALLS = 8
ADSB_PACE = 1.2 # seconds between calls; adsb.fi allows roughly one per second
_opensky_blocked_until = 0.0
# adsb.lol keeps a military register keyed by ICAO hex. Neither feed marks the
# aircraft it returns, so the register is pulled separately and used to tag them
# whichever source the picture came from.
MIL_URL = "https://api.adsb.lol/v2/mil"
MIL_TTL = 120
_mil_hexes = set()
_mil_stamp = 0.0
_mil_lock = threading.Lock()
OPENSKY_TOKEN_URL = (
"https://auth.opensky-network.org/auth/realms/opensky-network/"
"protocol/openid-connect/token"
)
_opensky_token = {"value": None, "expires": 0.0}
def opensky_token():
"""OAuth2 client-credentials token, if the user registered an OpenSky client.
Worth having: OpenSky answers a single call for the whole planet, where the
community feeders have to be stitched together 250 nm at a time.
"""
client_id = KEYS.get("opensky_client_id")
secret = KEYS.get("opensky_client_secret")
if not client_id or not secret:
return None
if _opensky_token["value"] and time.time() < _opensky_token["expires"]:
return _opensky_token["value"]
body = urllib.parse.urlencode({
"grant_type": "client_credentials",
"client_id": client_id,
"client_secret": secret,
}).encode()
req = urllib.request.Request(
OPENSKY_TOKEN_URL,
data=body,
headers={
"Content-Type": "application/x-www-form-urlencoded",
"User-Agent": USER_AGENT,
},
)
with urllib.request.urlopen(req, timeout=TIMEOUT) as resp:
payload = json.loads(resp.read())
key_worked("opensky_client_id")
key_worked("opensky_client_secret")
_opensky_token["value"] = payload["access_token"]
_opensky_token["expires"] = time.time() + payload.get("expires_in", 1800) - 60
log("opensky: authenticated, global snapshots available")
return _opensky_token["value"]
def military_hexes():
global _mil_stamp
if time.time() - _mil_stamp < MIL_TTL:
return _mil_hexes
with _mil_lock:
if time.time() - _mil_stamp < MIL_TTL:
return _mil_hexes
try:
data = json.loads(fetch(MIL_URL))
fresh = {ac["hex"].lower() for ac in data.get("ac") or [] if ac.get("hex")}
if fresh:
_mil_hexes.clear()
_mil_hexes.update(fresh)
_mil_stamp = time.time()
log(f"military register: {len(_mil_hexes)} airframes airborne")
except Exception as exc: # noqa: BLE001 - tagging is a bonus, not a feed
_mil_stamp = time.time() # do not retry in a tight loop
log(f"military register unavailable: {exc}")
return _mil_hexes
# The memory cache is bounded, because its keys are not. An aircraft type is
# keyed by hull, a street photo by a 100 m square, a road tile by its tile: every
# one of those grows with wherever you have been, and a plain dict would hold all
# of it for as long as the server runs. One day of flying around left 19 771
# aircraft entries on disk, and the same number would have sat in RAM alongside a
# lock object each.
#
# An OrderedDict gives least-recently-used order for free: read a key and it
# moves to the end, so eviction takes from the front. The disk cache is untouched
# by this — evicting from memory costs one file read, not a network round trip.
MEM_BUDGET_BYTES = 96 * 1024 * 1024
LOCK_BUDGET = 4096
_mem = collections.OrderedDict()
_mem_bytes = 0
_mem_guard = threading.Lock()
_locks = collections.OrderedDict()
_locks_guard = threading.Lock()
def log(*parts):
print(time.strftime("[%H:%M:%S]"), *parts, flush=True)
# A certificate failure is the machine's, and it does not explain itself.
#
# Reported from a fresh Windows install: every layer that fetches over HTTPS came
# back empty and the app said "certificate verify failed: unable to get local
# issuer certificate". That is accurate and useless. It reads like the app is
# broken, or like a key is missing, and it is neither - the app never calls a
# keyed service without a key, and a keyless install draws ninety-one radio
# stations on a healthy machine.
#
# What it means on Windows: Python verifies against the Windows certificate
# store, and Windows only fetches a root certificate the first time something
# asks for it. On a new machine, or one that cannot reach Windows Update, the
# root is simply not there yet - and unlike a browser, Python does not trigger
# the fetch itself. Opening the address in Edge or Chrome does, because they use
# that same store, and Python finds it on the next start.
#
# Checked before writing this: setting SSL_CERT_FILE, which most of the internet
# recommends, changes nothing here. Pointing it at a file that does not exist
# left verification working perfectly, so it is not what Python reads on Windows
# and following that advice only wastes an evening.
#
# What this deliberately does not do is offer to turn verification off. That
# would fix the symptom by making every connection this app makes unverified,
# on a machine whose trust store is already known to be wrong.
CERT_MARKERS = (b"CERTIFICATE_VERIFY_FAILED",
b"unable to get local issuer",
b"certificate verify failed")
CERT_ADVICE = (
"this machine cannot verify HTTPS certificates, which is why the feed is "
"empty - it is not a missing key and not this feed being down. Python on "
"Windows verifies against the Windows certificate store, and Windows only "
"fetches a root the first time something needs it, so on a new machine it "
"is not there yet. Open the failing address in Edge or Chrome once - they "
"use the same store and will make Windows fetch it - then stop and start "
"this app. If the browser also complains, something is inspecting HTTPS: "
"antivirus with 'scan encrypted connections' switched on, or a company "
"proxy whose certificate Windows does not trust."
)
# Which addresses to open, because "the failing address" is not something the
# feed log can tell you and the list is short.
#
# It is per certificate authority, not all-or-nothing: a machine reported
# shortwave, APRS and aircraft working while radio, airports, runways, weather
# and beacons were all empty. Those four sit behind three hosts, so three visits
# in a browser fix eight layers.
CERT_HOSTS = (
"https://de1.api.radio-browser.info/json/stations/search?limit=1",
"https://davidmegginson.github.io/ourairports-data/airports.csv",
"https://aviationweather.gov/api/data/metar?ids=ESSA",
"https://api.openmhz.com/systems",
)
def _with_cert_advice(data):
"""Add a plain explanation to any answer carrying a certificate failure.
Done here, once, rather than in each feed's own error path: this failure is
never about one feed. When it happens, everything fetched over HTTPS fails
the same way, and every one of them should say the same useful thing.
"""
if not data or not any(m in data for m in CERT_MARKERS):
return data
try:
body = json.loads(data)
if isinstance(body, dict) and "cert_advice" not in body:
body["cert_advice"] = CERT_ADVICE
body["cert_hosts"] = list(CERT_HOSTS)
return json.dumps(body).encode()
except Exception: # noqa: BLE001 - not JSON, or not ours to rewrite
pass
return data
# A second attempt for machines whose trust store cannot be fixed.
#
# Windows fetches a root certificate the first time something asks, through
# CryptoAPI, and Python never asks - it copies what is already in the store and
# verifies with OpenSSL. warm-certificates.ps1 asks on its behalf during
# installation and that is enough on most machines. It is not enough on one that
# cannot reach Windows Update's certificate list at all, because then the root
# does not exist locally and no amount of asking produces it.
#
# So: every request goes to the operating system's trust store first, exactly as
# before. Only one that comes back with a certificate error is tried again
# against the bundle in certs/, which is Mozilla's list as packaged by certifi.
#
# The order is the whole design. The system store is current and the bundle is a
# snapshot that ages, so a healthy machine never touches the file, and a machine
# that would otherwise show nothing at all gets a second attempt that is still
# properly verified.
#
# There is deliberately no way to switch verification off. That would turn an
# empty layer into every connection being unchecked, on a machine already known
# to have something wrong with its trust store.
CA_BUNDLE = os.path.join(ROOT, "certs", "cacert.pem")
_ca_context = None
_ca_used = [False]
def _is_cert_error(exc):
text = str(exc)
return ("CERTIFICATE_VERIFY_FAILED" in text
or "certificate verify failed" in text
or "unable to get local issuer" in text)
def _bundle_context():
global _ca_context
if _ca_context is None:
_ca_context = ssl.create_default_context(cafile=CA_BUNDLE)
return _ca_context
_system_urlopen = urllib.request.urlopen
def _urlopen_with_fallback(url, *args, **kwargs):
try:
return _system_urlopen(url, *args, **kwargs)
except Exception as exc: # noqa: BLE001 - only certificate errors are ours
if not _is_cert_error(exc) or not os.path.exists(CA_BUNDLE):
raise
retry = dict(kwargs)
retry["context"] = _bundle_context()
opened = _system_urlopen(url, *args, **retry)
if not _ca_used[0]:
_ca_used[0] = True
log("certificates: this machine could not verify a host from its own "
"store, so the bundle in certs/ is being used instead. Still "
"verified, just against Mozilla's list rather than Windows'.")
return opened
# Patched rather than threaded through fifty-six call sites, each of which would
# have had to remember to do this and one of which would have forgotten.
urllib.request.urlopen = _urlopen_with_fallback
def _mem_get(key):
"""Look up and mark as recently used."""
with _mem_guard:
hit = _mem.get(key)
if hit is not None:
_mem.move_to_end(key)
return hit
# Whether a key has actually been used for something that worked.
#
# SETUP said IN USE the moment a key was saved, which is a claim about a text
# field and not about the key. A key can be saved and wrong, saved and expired,
# saved and refused by a referrer rule - and the panel called all of those IN
# USE. In an app whose whole discipline is not stating more than it can support,
# that was the worst-supported sentence in it.
#
# Three states now: not set, saved but never seen to work, and seen to work at a
# time this records. Some keys are used by the browser rather than by the
# server - Cesium ion and Google Maps - and for those the honest answer is that
# this cannot tell, which is what it says.
_key_ok = {}
CLIENT_SIDE_KEYS = ("cesium_ion", "google_maps")
def key_worked(name):
"""Record that a call using this key came back with something."""
_key_ok[name] = time.time()
def _mem_put(key, data):
"""Store, then evict from the cold end until inside the budget."""
global _mem_bytes
with _mem_guard:
old = _mem.pop(key, None)
if old is not None:
_mem_bytes -= len(old[1])
_mem[key] = (time.time(), data)
_mem_bytes += len(data)
evicted = 0
while _mem_bytes > MEM_BUDGET_BYTES and len(_mem) > 1:
_, cold = _mem.popitem(last=False)
_mem_bytes -= len(cold[1])
evicted += 1
if evicted:
log(f"cache: dropped {evicted} cold entries, "
f"{_mem_bytes // (1024 * 1024)} MB held")
def _lock_for(key):
with _locks_guard:
lock = _locks.get(key)
if lock is None:
lock = _locks[key] = threading.Lock()
else:
_locks.move_to_end(key)
# A lock currently held must not be dropped, or two threads would each
# get a fresh one and both fetch the same URL.
while len(_locks) > LOCK_BUDGET:
cold_key, cold = _locks.popitem(last=False)
if cold.locked():
_locks[cold_key] = cold
_locks.move_to_end(cold_key)
break
return lock
def fetch(url, timeout=None):
"""GET url and return decoded bytes. Digitraffic requires gzip.
The timeout is the shared one unless a caller has measured its own. A
two-year catalogue search takes about seven seconds against a service that
is free and sometimes busy, and thirty is too tight a leash for that.
"""
req = urllib.request.Request(
url, headers={"User-Agent": USER_AGENT, "Accept-Encoding": "gzip"}
)
with urllib.request.urlopen(req, timeout=timeout or TIMEOUT) as resp:
raw = resp.read()
if resp.headers.get("Content-Encoding") == "gzip":
raw = gzip.GzipFile(fileobj=io.BytesIO(raw)).read()
return raw
# --------------------------------------------------------- disk cache bounds
# The disk cache had no bound at all and had reached 92 MB in 34 000 files.
#
# Measuring it first changed the design. It holds two populations that want
# opposite treatment:
#
# eight files mesh nodes, airports, power plants, two fire snapshots,
# satellites - 82 MB between them, and expensive to fetch again
# ~34 000 files per-query answers, median 143 bytes, ~10 MB in total
#
# So plain least-recently-used over everything would be actively wrong. Left to
# itself it would drop meshnodes.json, untouched for three days, to reclaim 30 MB
# - and the next request downloads those same 30 MB back. Small files go first,
# and the big ones are only touched if dropping every small file was not enough.
#
# Two ceilings, because bytes and file count are different problems. Half a
# gigabyte is nothing on disk, but at a 143-byte median it would hold over a
# million files, and a directory that size is slow to list, slow to back up and
# unpleasant to open. The count is what actually needs holding down.
DISK_BUDGET_BYTES = 500 * 1024 * 1024
DISK_MAX_FILES = 40000
DISK_LARGE_FILE = 1024 * 1024 # above this, evicted only as a last resort
DISK_SWEEP_EVERY = 400 # writes between sweeps; stat'ing 34 000 files
# on every write would cost more than it saves
_disk_writes = 0
def _sweep_disk(force=False):
"""Hold the cache under both ceilings, oldest and smallest first."""
global _disk_writes
if not force:
_disk_writes += 1
if _disk_writes < DISK_SWEEP_EVERY:
return
_disk_writes = 0
try:
names = os.listdir(CACHE_DIR)
except OSError:
return
small, large, total = [], [], 0
for name in names:
path = os.path.join(CACHE_DIR, name)
try:
st = os.stat(path)
except OSError:
continue
if not stat_module.S_ISREG(st.st_mode):
continue
total += st.st_size
(large if st.st_size >= DISK_LARGE_FILE else small).append(
(st.st_mtime, st.st_size, path))
count = len(small) + len(large)
if total <= DISK_BUDGET_BYTES and count <= DISK_MAX_FILES:
return
freed = dropped = 0
# Oldest first within each group, and the whole small group before any of the
# large one.
for group in (sorted(small), sorted(large)):
for mtime, size, path in group:
if total <= DISK_BUDGET_BYTES and count <= DISK_MAX_FILES:
break
try:
os.remove(path)
except OSError:
continue
total -= size
count -= 1
freed += size
dropped += 1
if dropped:
log("cache: dropped %d files, freed %.1f MB, now %.1f MB in %d files"
% (dropped, freed / 1e6, total / 1e6, count))
def _disk_path(key):
return os.path.join(CACHE_DIR, key.replace("/", "_") + ".json")
def cached(key, url, mem_ttl, disk_ttl):
"""Return (payload_bytes, source) using memory then disk then network."""
now = time.time()
hit = _mem_get(key)
if hit and now - hit[0] < mem_ttl:
return hit[1], "memory"
with _lock_for(key):
hit = _mem_get(key)
if hit and time.time() - hit[0] < mem_ttl:
return hit[1], "memory"
path = _disk_path(key)
if disk_ttl and os.path.exists(path):
age = time.time() - os.path.getmtime(path)
if age < disk_ttl:
with open(path, "rb") as fh:
data = fh.read()
_mem_put(key, data)
return data, "disk"
try:
data = fetch(url)
log(f"fetched {key} ({len(data) // 1024} kB)")
except urllib.error.HTTPError as exc:
if exc.code == 403 and key == "satellites":
# CelesTrak answers 403 when its data has not changed since your
# last download. The cached elements are the correct response.
log("celestrak: elements unchanged since last download, using cache")
else:
log(f"fetch failed for {key}: HTTP {exc.code}")
if hit:
return hit[1], "stale"
if disk_ttl and os.path.exists(path):
with open(path, "rb") as fh:
return fh.read(), "cached"
raise
except Exception as exc: # noqa: BLE001 - any failure falls back to stale
log(f"fetch failed for {key}: {exc}")
if hit:
return hit[1], "stale"
if disk_ttl and os.path.exists(path):
with open(path, "rb") as fh:
return fh.read(), "stale-disk"
raise
_mem_put(key, data)
if disk_ttl:
os.makedirs(CACHE_DIR, exist_ok=True)
with open(path, "wb") as fh:
fh.write(data)
_sweep_disk()
return data, "network"
TRAFIKVERKET_URL = "https://api.trafikinfo.trafikverket.se/v2/data.json"
TRAFIKVERKET_QUERY = """<REQUEST>
<LOGIN authenticationkey="{key}"/>
<QUERY objecttype="Camera" schemaversion="1.0">
<FILTER><EQ name="Active" value="true"/></FILTER>
<INCLUDE>Id</INCLUDE><INCLUDE>Name</INCLUDE><INCLUDE>Description</INCLUDE>
<INCLUDE>CountyNo</INCLUDE><INCLUDE>Geometry.WGS84</INCLUDE><INCLUDE>PhotoUrl</INCLUDE>
</QUERY>
</REQUEST>"""
# The camera record carries a county number rather than a place name.
COUNTIES = {
1: "Stockholm", 3: "Uppsala", 4: "Södermanland", 5: "Östergötland",
6: "Jönköping", 7: "Kronoberg", 8: "Kalmar", 9: "Gotland", 10: "Blekinge",
12: "Skåne", 13: "Halland", 14: "Västra Götaland", 17: "Värmland",
18: "Örebro", 19: "Västmanland", 20: "Dalarna", 21: "Gävleborg",
22: "Västernorrland", 23: "Jämtland", 24: "Västerbotten", 25: "Norrbotten",
}
def trafikverket_cameras():
"""Swedish road cameras. Free key, but the API refuses anonymous callers."""
hit = _mem_get("cameras-se")
path = _disk_path("cameras-se")
if hit and time.time() - hit[0] < 86400:
return json.loads(hit[1])
if os.path.exists(path) and time.time() - os.path.getmtime(path) < 86400:
with open(path, "rb") as fh:
raw = fh.read()
key_worked("trafikverket")
_mem_put("cameras-se", raw)
return json.loads(raw)
body = TRAFIKVERKET_QUERY.format(key=KEYS["trafikverket"]).encode()
req = urllib.request.Request(
TRAFIKVERKET_URL,
data=body,
headers={"Content-Type": "text/xml", "User-Agent": USER_AGENT},
)
with urllib.request.urlopen(req, timeout=TIMEOUT) as resp:
payload = json.loads(resp.read())
out = []
for camera in payload["RESPONSE"]["RESULT"][0].get("Camera", []):
point = (camera.get("Geometry") or {}).get("WGS84", "") # "POINT (18.06 59.33)"
try:
lon, lat = (float(v) for v in point.split("(")[1].rstrip(")").split())
except (IndexError, ValueError):
continue
if not camera.get("PhotoUrl"):
continue
county = (camera.get("CountyNo") or [None])[0]
out.append({
# Ids come as SE_STA_CAMERA_0_1075001058 or SE_STA_CAMERA_Orion_426
"id": camera.get("Id", "").replace("SE_STA_CAMERA_", ""),
"name": camera.get("Name") or "Camera",
"area": COUNTIES.get(county, "Sweden"),
"lat": lat,
"lon": lon,
# The bare URL serves a 10 kB thumbnail; fullsize is the real frame.
"image": camera["PhotoUrl"] + "?type=fullsize",
"source": "Trafikverket",
})
raw = json.dumps(out).encode()
key_worked("trafikverket")
_mem_put("cameras-se", raw)
os.makedirs(CACHE_DIR, exist_ok=True)
with open(path, "wb") as fh:
fh.write(raw)
_sweep_disk()
log(f"cameras SE: {len(out)} stations")
return out
WINDY_URL = (
"https://api.windy.com/webcams/api/v3/webcams"
"?limit=50&offset={offset}&include=location,images{filter}"
)
# Windy holds ~70 000 webcams but the free tier stops paging at about 1 050, so
# an unfiltered pull returns whatever is most popular — which turns out to be
# the Alps. These country buckets spend that budget on a global spread instead.
WINDY_REGIONS = [
("SE,NO,DK,FI,IS", 6),
("", 4), # unfiltered: the most-viewed webcams worldwide
("US,CA,MX", 4),
("JP,KR,TH,IN,ID", 3),
("AU,NZ,BR,AR,CL,ZA", 3),
]
def windy_cameras():
"""Windy's global webcam network. Free key after registration, metered."""
hit = _mem_get("cameras-windy")
if hit and time.time() - hit[0] < 86400:
return json.loads(hit[1])
path = _disk_path("cameras-windy")
if os.path.exists(path) and time.time() - os.path.getmtime(path) < 86400:
with open(path, "rb") as fh:
raw = fh.read()
key_worked("windy")
_mem_put("cameras-windy", raw)
return json.loads(raw)
out = []
seen = set()
for countries, pages in WINDY_REGIONS:
for page in range(pages):
url = WINDY_URL.format(
offset=page * 50,
filter=f"&countries={countries}" if countries else "",
)
req = urllib.request.Request(
url,
headers={"x-windy-api-key": KEYS["windy"], "User-Agent": USER_AGENT},
)
with urllib.request.urlopen(req, timeout=TIMEOUT) as resp:
payload = json.loads(resp.read())
webcams = payload.get("webcams") or []
if not webcams:
break
for cam in webcams:
location = cam.get("location") or {}
images = (cam.get("images") or {}).get("current") or {}
webcam_id = str(cam.get("webcamId", ""))
if not images.get("preview") or location.get("latitude") is None:
continue
if webcam_id in seen: # buckets overlap with the popular list
continue
seen.add(webcam_id)
out.append({
"id": webcam_id,
"name": cam.get("title") or "Webcam",
"area": ", ".join(
filter(None, [location.get("city"), location.get("country")])
),
"lat": location["latitude"],
"lon": location["longitude"],
# Plain imgproxy paths with no expiry token: they always
# serve that webcam's current frame, so link them directly.
"image": images["preview"],
"source": "Windy",
})
raw = json.dumps(out).encode()
key_worked("windy")
_mem_put("cameras-windy", raw)
os.makedirs(CACHE_DIR, exist_ok=True)
with open(path, "wb") as fh:
fh.write(raw)
_sweep_disk()
log(f"cameras Windy: {len(out)} stations")
return out
AIRCRAFT_URL = "https://api.adsbdb.com/v0/aircraft/{hex}"
# ICAO type designators for rotorcraft. ADS-B carries the type but never says
# "helicopter", so the list is the classifier.
ROTORCRAFT = {
"A109", "A119", "A139", "A169", "A189", "AS32", "AS50", "AS55", "AS65",
"B06", "B06T", "B105", "B212", "B222", "B230", "B407", "B412", "B427",
"B429", "B430", "B505", "BK17", "EC20", "EC25", "EC30", "EC35", "EC45",
"EC55", "EC75", "EH10", "EXPL", "GAZL", "H125", "H135", "H145", "H155",
"H160", "H175", "H500", "H60", "HUCO", "KA32", "LYNX", "MD52", "MD60",
"MI17", "MI24", "MI8", "NH90", "PUMA", "R22", "R44", "R66", "S276", "S61",
"S76", "S92", "UH1", "UH60",
}
# Owner strings that mean a police operator, in the languages the registry uses.
POLICE_WORDS = (
"police", "polizei", "politie", "polis", "politi", "policia", "polizia",
"policja", "rendorseg", "gendarmerie", "guardia civil", "carabinieri",
"sheriff", "state patrol", "state trooper", "garda", "npas",
"public safety", "law enforcement",
)
def classify_owner(owner):
"""Police, or another state operator, or nothing — from the registry text."""
low = (owner or "").lower()
if any(word in low for word in POLICE_WORDS):
return "police"
for word in ("air force", "navy", "army", "military", "defence", "defense"):
if word in low:
return "military"
for word in ("coast guard", "coastguard", "kustbevakning", "kystvakt", "border",
"maritime administration", "sjofartsverket", "sjöfartsverket",
"search and rescue", "rescue", "redningstjeneste", "raddning"):
if word in low:
return "coastguard"
for word in ("ambulance", "hems", "air rescue", "rega", "medical", "lifeflight"):
if word in low:
return "medical"
return ""
def aircraft_type(icao_hex):
"""Registry record for one airframe: type, owner and what that owner is.
ADS-B tells you a hull is there, not what it is. adsbdb keeps the registry,
and registry entries do not change, so each hex is asked for once ever.
"""
key = f"actype_{icao_hex}"
hit = _mem_get(key)
if hit:
return json.loads(hit[1])
path = _disk_path(key)
if os.path.exists(path):
with open(path, "rb") as fh:
raw = fh.read()
_mem_put(key, raw)
return json.loads(raw)
record = {}
try:
payload = json.loads(fetch(AIRCRAFT_URL.format(hex=icao_hex)))
aircraft = (payload.get("response") or {}).get("aircraft") or {}
if aircraft:
owner = aircraft.get("registered_owner") or ""
record = {
"reg": aircraft.get("registration") or "",
"icao_type": aircraft.get("icao_type") or "",
"type": aircraft.get("type") or "",
"owner": owner,
"country": aircraft.get("registered_owner_country_name") or "",
"rotorcraft": (aircraft.get("icao_type") or "") in ROTORCRAFT,
"role": classify_owner(owner),
}
except Exception: # noqa: BLE001 - an unknown airframe is a normal answer
record = {}
raw = json.dumps(record).encode()
_mem_put(key, raw)
os.makedirs(CACHE_DIR, exist_ok=True)
with open(path, "wb") as fh:
fh.write(raw)
_sweep_disk()
return record
PHOTO_URL = "https://api.planespotters.net/pub/photos/hex/{hex}"
ROUTE_URL = "https://api.adsbdb.com/v0/callsign/{callsign}"
# planespotters rejects generic library user agents and asks for a contact link.
PHOTO_AGENT = "global-command-view/0.9 (+http://localhost:8820 local research client)"
ESRI_IDENTIFY = (
"https://services.arcgisonline.com/arcgis/rest/services/World_Imagery/MapServer/identify"
)
MARKS_PATH = os.path.join(ROOT, "data", "marks.json")
_marks_lock = threading.Lock()
# Places being watched for movement. A target is a name, a point, and which
# run of pictures answers for it - a few hundred bytes each. The pictures
# themselves are never stored here; a single one is eight gigabytes and the
# processing that turns a stack of them into millimetres happens elsewhere.
TARGETS_PATH = os.path.join(ROOT, "data", "targets.json")
_targets_lock = threading.Lock()
USAGE_PATH = os.path.join(ROOT, "data", "usage.json")
# What Google gives away each month on the Photorealistic 3D Tiles SKU before
# the first cent is charged. One request buys a session, not a tile.
GOOGLE_FREE_ROOTS = 1000
def _pacific_now():
"""Now, in US Pacific time.
Google resets the free monthly allowance at midnight Pacific on the 1st, so
counting in UTC would zero the tally seven or eight hours early and report a
clean sheet while Google was still charging against the old month.
zoneinfo is the right answer where the machine has a tz database. Windows
ships none, so the US rule is spelled out as the fallback rather than adding
a dependency: DST from the second Sunday in March to the first Sunday in
November, both at 02:00 local.
"""
utc = datetime.datetime.now(datetime.timezone.utc)
try:
from zoneinfo import ZoneInfo
return utc.astimezone(ZoneInfo("America/Los_Angeles"))
except Exception: # noqa: BLE001 - no tz database, use the written rule
pass
def nth_sunday(year, month, nth):
first = datetime.date(year, month, 1)
# weekday(): Monday is 0, so Sunday is 6
first_sunday = 1 + (6 - first.weekday()) % 7
return datetime.date(year, month, first_sunday + 7 * (nth - 1))
year = utc.year
# 02:00 PST is 10:00 UTC; 02:00 PDT is 09:00 UTC
starts = datetime.datetime.combine(
nth_sunday(year, 3, 2), datetime.time(10), datetime.timezone.utc)
ends = datetime.datetime.combine(
nth_sunday(year, 11, 1), datetime.time(9), datetime.timezone.utc)
offset = -7 if starts <= utc < ends else -8
return utc + datetime.timedelta(hours=offset)
def _usage_month():
return _pacific_now().strftime("%Y-%m")
def read_usage():
"""This month's billable requests, as counted by this app.
Only this app's own asking is counted, and only since counting began. The
authority is the Google Cloud console; this is a warning light, not a bill.
"""
stored = {}
if os.path.exists(USAGE_PATH):
with open(USAGE_PATH, encoding="utf-8") as fh:
stored = json.load(fh)
month = _usage_month()
services = stored.get("services", {})
return json.dumps({
"month": month,
"google_root": stored.get("months", {}).get(month, 0),
"free_limit": GOOGLE_FREE_ROOTS,
"streetview": services.get("google_streetview", {}).get(month, 0),
"streetview_limit": USAGE_LIMITS["google_streetview"],
"since": stored.get("since", month),
}).encode(), "disk"
# Street View images are billed per request with 10 000 free a month; the
# metadata lookups the app makes first are free and unlimited, and are not
# counted here because there is nothing to count.
USAGE_LIMITS = {"google_root": GOOGLE_FREE_ROOTS, "google_streetview": 10000}
def bump_usage(raw):
"""Record one billable Google request. Months are kept, so history stays."""
incoming = json.loads(raw or b"{}")
# The browser vouching for a key the server never uses.
#
# Cesium ion and the Google key are handed to the page, not to this process,
# so nothing here ever sends a request with them and SETUP said as much:
# "used by the browser, so the server cannot vouch for it". True when it was
# written, and useless to somebody looking at photoreal 3D on screen while
# the panel declines to confirm the key it is running on.
#
# The page knows. It is what loaded the Maps API and the terrain, so it says
# so, through the channel it already uses to count billable requests. Only
# the two browser-side keys may be reported this way - anything else the
# server can and does prove for itself.
proved = incoming.get("worked")
if proved:
if proved not in CLIENT_SIDE_KEYS:
raise ValueError("not a key the browser can vouch for")
key_worked(proved)
return json.dumps({"ok": True, "worked": proved}).encode()
service = incoming.get("service", "google_root")
if service not in USAGE_LIMITS:
raise ValueError("unknown service")
stored = {}
if os.path.exists(USAGE_PATH):
with open(USAGE_PATH, encoding="utf-8") as fh:
stored = json.load(fh)
stored.setdefault("since", _usage_month())
month = _usage_month()
# Photoreal sessions kept the bare month->count shape before there was a
# second thing to count; that shape is still read, so old files still work.
months = stored.setdefault("months", {})
if service == "google_root":
months[month] = months.get(month, 0) + 1
used = months[month]
else:
per = stored.setdefault("services", {}).setdefault(service, {})
per[month] = per.get(month, 0) + 1
used = per[month]
os.makedirs(os.path.dirname(USAGE_PATH), exist_ok=True)
with open(USAGE_PATH, "w", encoding="utf-8") as fh:
json.dump(stored, fh, indent=2)
limit = USAGE_LIMITS[service]
log(f"google {service} {used}/{limit} free this month")
if used > limit:
log(f"past the free allowance for {service} — further calls are billed",
"warn")
return json.dumps({"ok": True, "service": service, "used": used,
"free_limit": limit}).encode()
ALLOWED_KEYS = (
"windy", "trafikverket", "opensky_client_id", "opensky_client_secret",
"aisstream", "cesium_ion", "google_maps", "openaq", "gfw", "tomtom",
"copernicus", "mapquest",
)
def write_keys(raw):
"""Save API keys typed into the app, and start using them without a restart.
Keys arrive from a page served on localhost only. Values are never logged and
never handed back to the browser — the page is told which are set, nothing more.
"""
global AIS
incoming = json.loads(raw)
stored = {}
if os.path.exists(_keys_path):