forked from heygen-com/hyperframes
-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathdownload_wiki.py
More file actions
37 lines (33 loc) · 1.46 KB
/
Copy pathdownload_wiki.py
File metadata and controls
37 lines (33 loc) · 1.46 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
import urllib.request
import json
import time
import os
os.makedirs("public/images/raw", exist_ok=True)
files = [
"File:Lewis_Gun_during_Emu_War.jpg",
"File:Ray_Owen_during_Emu_War.jpg",
"File:Sargent_McMurray_and_J._O'Hallroan_Emu_War.jpg",
"File:Dromaius_novaehollandiae_-_01.jpg"
]
headers = {'User-Agent': 'HyperFramesBot/1.0 (contact@heygen.com)'}
for fname in files:
url = f"https://commons.wikimedia.org/w/api.php?action=query&titles={urllib.parse.quote(fname)}&prop=imageinfo&iiprop=url&format=json"
req = urllib.request.Request(url, headers=headers)
try:
with urllib.request.urlopen(req) as resp:
data = json.loads(resp.read().decode('utf-8'))
pages = data.get('query', {}).get('pages', {})
for pid, page in pages.items():
img_info = page.get('imageinfo', [])
if img_info:
img_url = img_info[0].get('url')
print(fname, '->', img_url)
img_req = urllib.request.Request(img_url, headers=headers)
clean_name = fname.replace('File:', '')
out_path = os.path.join('public/images/raw', clean_name)
with urllib.request.urlopen(img_req) as img_resp, open(out_path, 'wb') as f:
f.write(img_resp.read())
print('Successfully saved', clean_name)
time.sleep(1)
except Exception as e:
print('Error:', fname, e)