world-density-map/wiki.py

412 lines
15 KiB
Python

#!/usr/bin/env python3
import sys, time, codecs, re
import json
import yaml
import requests
from bs4 import BeautifulSoup
from bs4 import element
host = 'https://en.wikipedia.org'
# https://de.wikipedia.org/wiki/Bevölkerungsentwicklung
# Kennzeichnend für die Bevölkerungsentwicklung der Welt, insbesondere die der letzten 200 Jahre, war ein starkes hyperexponentielles Wachstum, weshalb man auch von Bevölkerungsexplosion spricht. Seit dem Wendepunkt 1962/63 sinkt hingegen die Wachstumsrate und seit 1989 auch der absolute Zuwachs.
# Bedingt durch seinen negativen Einfluss auf die begrenzte Tragfähigkeit der Erde (siehe auch Ökologischer Fußabdruck) sowie seine Multiplikatorfunktion aller der nachhaltigen Entwicklung entgegenstehenden Aktivitäten des Menschen ist das Bevölkerungswachstum eines der zentralen globalen Probleme und mitverantwortlich für die globale Erwärmung.
# https://www.cia.gov/library/publications/the-world-factbook/geos/print_af.html
# https://www.webuildinternet.com/2015/07/09/geojson-data-of-the-netherlands/
def first_int(td):
if not td:
return -1
raw_str = td.text.replace(',', '')
int_str = re.search('^\d+', raw_str)
if int_str:
return int_str.group(0)
return -1
def readinfobox(url):
pop_total = 0
area_km2 = 0
print('reading infobox from %s ... ' % url)
soup = BeautifulSoup(html(url), 'html5lib')
table = soup.find("table", class_="infobox")
th = table.find_all('th')
for item in th:
if item.text.startswith('Area') and area_km2 == 0:
next_tr = item.parent.next_sibling.next_sibling
head = next_tr.find('th')
# print(head)
# if head:
# title = re.search('[A-Za-z]+', head.text).group(0).lower()
td = head.next_sibling.next_sibling
if td:
area_km2 = re.search('^[0-9.,]+', td.text.replace(',', '')).group(0)
elif item.text.startswith('Population'):
next_tr = item.parent.next_sibling.next_sibling
pop_total = first_int(next_tr.find('td'))
if area_km2 == 0:
print('error: no area data found')
return area_km2, pop_total
def readinfobox_raw(url):
pop_total = 0
area_raw = 0
# print('reading infobox from %s ... ' % url)
soup = BeautifulSoup(html(url), 'html5lib')
table = soup.find("table", class_="infobox")
th = table.find_all('th')
for item in th:
if item.text.startswith('Area') and area_raw == 0:
if item.next_sibling.next_sibling != None:
if item.next_sibling.next_sibling.text != None:
# print(item.next_sibling.next_sibling)
nearby_col_text = item.next_sibling.next_sibling.text
if re.search('km2', nearby_col_text):
area_raw = nearby_col_text
next_row = item.parent.next_sibling.next_sibling
next_row_head = next_row.find('th')
next_row_col = next_row_head.next_sibling.next_sibling
if next_row_col:
next_row_col_text = next_row_col.text
if re.search('km2', next_row_col_text):
area_raw = next_row_col_text
# print(nearby_col_text, next_row_nearby_col_text)
# print(head)
# if head:
# title = re.search('[A-Za-z]+', head.text).group(0).lower()
# td = head.next_sibling.next_sibling
# if td:
# area_raw = td.text
# area_raw = re.search('^[0-9.,]+', td.text.replace(',', '')).group(0)
elif item.text.startswith('Population'):
if item.next_sibling.next_sibling != None:
if item.next_sibling.next_sibling.text != None:
nearby_col_text = item.next_sibling.next_sibling.text
if re.search('[0-9,]+', nearby_col_text):
pop_total = re.search('[0-9,]+', nearby_col_text).group(0)
else:
if item.parent.next_sibling.next_sibling != None:
next_row_text = item.parent.next_sibling.next_sibling.find('td').text
if next_row_text != None:
if re.search('[0-9,]+', next_row_text):
pop_total = re.search('[0-9,]+', next_row_text).group(0)
# else:
# pop_total = first_int(next_tr.find('td'))
if area_raw != 0:
# filter the number before 'km2'
area = None
try:
area = re.search('[0-9,\.]+?(?=\skm2)', area_raw).group(0)
except:
pass
if area != None:
area_raw = area.replace(',', '')
if pop_total != 0:
pop_total = pop_total.replace(',', '')
if area_raw == 0:
print('error: no area data found')
if pop_total == 0:
print('error: no population data found')
if area_raw != 0 and pop_total != 0:
print('area = <%s>' % area_raw, 'population = <%s>' % pop_total)
return area_raw, pop_total
def readlistofstates():
# table xpath sample
# //*[@id="mw-content-text"]/div/table[1]/tbody/tr[4]/td[1]/b/a
url = '%s/wiki/List_of_sovereign_states' % host
request = requests.get(url)
if not request:
print('error: could not read contents from <%s>' % url)
return False
soup = BeautifulSoup(request.text, 'html5lib')
table = soup.find("table", class_="wikitable")
states = []
for tr in table.find_all('tr'):
# even_numbers = list(filter(lambda x: x % 2 == 0, fibonacci))
firsttd = list(filter(lambda x: x != '\n', tr.children))[0] # str_list = list(filter(None, str_list)) # fastest
firsttda = firsttd.find('a', href=re.compile('/wiki/'))
if firsttda:
states.append({ 'name' : firsttda.text, 'url' : firsttda['href'] })
# for c in tr.children:
# print(c)
# for a in table.find_all('a', href=re.compile('/wiki/')):
# print(a.text, a.href)
# for tag in soup.find_all(re.compile("^dt|dd")):
# if tag.get_text(strip=True) != '':
# result_list.append(tag)
# return result_list
return states
def readnavbox(url):
request = requests.get(url)
if not request:
print('error: could not read contents from <%s>' % url)
return False
soup = BeautifulSoup(request.text, 'html5lib')
table = soup.find("table", class_="navbox")
anchors = table.find_all('a')
provinces = []
for anchor in anchors:
provinces.append({ 'name': anchor.text, 'link' : '%s%s' % (host, anchor['href']) })
return provinces
def html(url):
print('fetching %s ... ' % url)
request = requests.get(url)
if not request:
print('error: could not read contents from <%s>' % url)
return False
return request.text
def writejson(filename, aobj):
with open(filename, 'w') as outfile:
json.dump(aobj, outfile, sort_keys=True, indent=4, separators=(',', ': '))
def germanstates():
request = requests.get('https://en.wikipedia.org/wiki/List_of_statistical_offices_in_Germany')
if not request:
print('error: could not read contents from <%s>' % url)
return False
soup = BeautifulSoup(request.text, 'html5lib')
table = soup.find("table", class_="wikitable")
rows = table.find_all('tr')
states = []
for row in rows:
children = list(row.children)
if len(children) > 1:
anchor = children[1].find('a')
if anchor:
href = anchor['href']
name = anchor.text
states.append({'name_en' : name, 'link' : 'https://en.wikipedia.org%s' % href })
return states
def netherlands():
# provinces: table class wikitable no 2 + 3, links in left column
url = 'https://en.wikipedia.org/wiki/Netherlands'
soup = BeautifulSoup(html(url), 'html5lib')
wikitables = soup.find_all("table", class_="wikitable")
tables = [wikitables[1]]
states = []
for table in tables:
tbody = table.find('tbody')
for row in tbody.find_all('tr'):
children = list(row.children)
if len(children) > 3:
anchor = children[1].find('a')
if anchor:
# normalize names
name = anchor['title']
name = re.sub('\(.*\)', '', name) # remove brackets
name = name.replace('_', '-') # only dashes
name = name.replace(' ', '-') # only dashes
if name[-1:] == '_': # remove trailing underscore
name = name[:-1]
if name[-1:] == '-': # remove trailing underscore
name = name[:-1]
name_nl = name.replace('South', 'Zuid') # in dutch
name_nl = name_nl.replace('North', 'Noord') # in dutch
# print(name)
states.append({'name_nl' : name_nl, 'name_en' : name, 'link' : 'https://en.wikipedia.org%s' % anchor['href'] })
return states
def netherlands_municipalities():
# provinces: table class wikitable no 2 + 3, links in left column
url = 'https://en.wikipedia.org/wiki/List_of_municipalities_of_the_Netherlands'
soup = BeautifulSoup(html(url), 'html5lib')
wikitables = soup.find_all("table", class_="wikitable")
# print(len(wikitables))
# return
tables = [wikitables[0]]
states = []
for table in tables:
tbody = table.find('tbody')
for row in tbody.find_all('tr'):
children = list(row.children)
if len(children) > 3:
anchor = children[1].find('a')
if anchor:
states.append({'name_en' : anchor['title'], 'link' : 'https://en.wikipedia.org%s' % anchor['href'] })
return states
def area_population_infoboxes(items):
for item in items:
area_km2, pop_total = readinfobox(item['link'])
item['area_km2'] = area_km2
item['population'] = pop_total
return items
def listregions(url, tablenum, colnum):
soup = BeautifulSoup(html(url), 'html5lib')
wikitables = soup.find_all("table", class_="wikitable")
print('found %d wikitables'% len(wikitables))
if len(wikitables) < tablenum:
print('error: table at index <%d> not available', tablenum)
return
tables = [wikitables[tablenum]]
items = []
for table in tables:
tbody = table.find('tbody')
for row in tbody.find_all('tr'):
children = list(row.children)
if len(children) > 3:
anchor = children[colnum].find('a')
if anchor != None:
if anchor.get('title', None) != None and anchor.get('href', None) != None:
item = {'name' : anchor.text, 'link' : 'https://en.wikipedia.org%s' % anchor['href'] }
print(item)
items.append(item)
return items
def main():
# readinfobox_raw('https://en.wikipedia.org/wiki/Greater_London')
# readinfobox_raw('https://en.wikipedia.org/wiki/South_Holland')
# readinfobox_raw('https://en.wikipedia.org/wiki/Saxony-Anhalt')
# readinfobox_raw('https://en.wikipedia.org/wiki/Berlin')
# readinfobox_raw('https://en.wikipedia.org/wiki/Hong_Kong')
# readinfobox_raw('https://en.wikipedia.org/wiki/Geldermalsen')
# return
items = listregions('https://en.wikipedia.org/wiki/Regions_of_England', 0, 1)
for item in items:
area_km2, pop_total = readinfobox_raw(item['link'])
item['area_km2'] = area_km2
item['population'] = pop_total
writejson('eng/regions.json', items)
return
# print(readinfobox('https://en.wikipedia.org/wiki/Geldermalsen'))
# return
# writejson('netherlands-municipalities.json', area_population_infoboxes(netherlands_municipalities()))
# # print(items)
# return
# writejson('netherlands.json', netherlands())
# return
# print(readinfobox('https://en.wikipedia.org/wiki/South_Holland'))
# print(readinfobox('https://en.wikipedia.org/wiki/Saxony-Anhalt'))
# print(readinfobox('https://en.wikipedia.org/wiki/Berlin'))
# print(readinfobox('https://en.wikipedia.org/wiki/Hong_Kong'))
# return
provinces = netherlands()
for province in provinces:
area_km2_int, pop_total_int = readinfobox(province['link'])
province['area_km2'] = area_km2_int
province['population'] = pop_total_int
writejson('netherlands/provinces.json', provinces)
return
# area_km2_int, pop_total_int = readinfobox('https://en.wikipedia.org/wiki/Hong_Kong')
# print(area_km2_int, pop_total_int)
# return
host = 'https://en.wikipedia.org'
url = '%s/wiki/China' % host
provinces = readnavbox(url)
for province in provinces:
name = province['name']
link = province['link']
area_km2_int, pop_total_int = readinfobox(link)
province['area_km2'] = area_km2_int
province['population'] = pop_total_int
writejson('china.json', provinces)
return
return
readstatepop()
return
states = readlistofstates()
writejson('en-wikipedia-%s-states.json' % len(states), states)
return 0
if __name__ == '__main__':
sys.exit(main())
#### trash
# normalize names
# items = readjson('netherlands/netherlands.json')
# for item in items:
# name = item['name_en']
# name = re.sub('\(.*\)', '', name) # remove brackets
# name = name.replace('_', '-') # only dashes
# name = name.replace(' ', '-') # only dashes
# if name[-1:] == '_': # remove trailing underscore
# name = name[:-1]
# if name[-1:] == '-': # remove trailing underscore
# name = name[:-1]
# name = name.replace('South', 'Zuid') # in english
# name = name.replace('North', 'Noord') # in english
# print(name)
# next_tr = item.parent.next_sibling.next_sibling
# next_tr_td = next_tr.find('td')
# pop_total_raw_str = next_tr_td.text.replace(',', '')
# pop_total = re.search('^\d+', pop_total_raw_str)
# if pop_total:
# pop_total_int = pop_total.group(0)
# next_tr_th = next_tr.find('th')
# if re.search('Total', next_tr_th.text) or re.search('Municipality', next_tr_th.text):
# next_tr_td = next_tr_th.next_sibling.next_sibling
# pop_total_raw_str = next_tr_td.text.replace(',', '')
# pop_total = re.search('^\d+', pop_total_raw_str)
# if pop_total:
# pop_total_int = pop_total.group(0)
# popa = table.find("a", text=re.compile('Population'))
# area_a = table.find("a", text=re.compile('Area'))
# # print(popa, area_a)
# return
# pop = list(popa.parent.parent.next_sibling.next_sibling.find('td').children)[0]
# area_km2 = list(area_a.parent.parent.next_sibling.next_sibling.find('td').children)[0]
# print(pop, area_km2)
# pop_int = int(pop.replace(',', ''))
# area_km2_int = int(area_km2.replace(',', ''))
# print(pop_int, area_km2_int)
# density = pop_int / area_km2_int
# print(density)