Skip to content
Merged
Show file tree
Hide file tree
Changes from 2 commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 7 additions & 0 deletions ACKNOWLEDGMENTS.md
Original file line number Diff line number Diff line change
Expand Up @@ -155,6 +155,13 @@ provided by [IPinfo](https://ipinfo.io) and released the under [Creative Commons
Attribution-ShareAlike 4.0 International
License](https://creativecommons.org/licenses/by-sa/4.0/).

## MaxMind

We use the free [GeoLite
Country](https://www.maxmind.com/en/geolite-free-ip-geolocation-data) database
provided by [MaxMind](https://www.maxmind.com) and released the under the
Comment thread
m-appel marked this conversation as resolved.
Outdated
[GeoLite End User License Agreement](https://www.maxmind.com/en/geolite/eula).

## Number Resource Organization

We use the [extended allocation and assignment
Expand Down
5 changes: 5 additions & 0 deletions config.json.example
Original file line number Diff line number Diff line change
Expand Up @@ -33,6 +33,11 @@
"token": ""
},

"maxmind": {
"account_id": "",
"license_key": ""
},

"pch": {
"parallel_downloads": 1,
"parallel_parsers": 8
Expand Down
1 change: 1 addition & 0 deletions documentation/data-sources.md
Original file line number Diff line number Diff line change
Expand Up @@ -44,6 +44,7 @@
| Internet Assigned Numbers Authority (IANA) | DNS Root Zone File | https://www.iana.org/domains/root/files | [README](https://github.com/InternetHealthReport/internet-yellow-pages/tree/main/iyp/crawlers/iana#readme) | iana.root_zone |
| | IPv4/IPv6 Address Allocations (extract) | https://www.iana.org/numbers | [README](https://github.com/InternetHealthReport/internet-yellow-pages/tree/main/iyp/crawlers/iana#readme) | iana.address_space |
| IPinfo | IP to Country Mapping| https://ipinfo.io | [README](https://github.com/InternetHealthReport/internet-yellow-pages/tree/main/iyp/crawlers/ipinfo#readme) | ipinfo.ip_country |
| MaxMind | IP to Country Mapping| https://www.maxmind.com/ | [README](https://github.com/InternetHealthReport/internet-yellow-pages/tree/main/iyp/crawlers/maxmind#readme) | maxmind.geolite_country |
| NRO | Extended allocation and assignment reports| https://www.nro.net/about/rirs/statistics | [README](https://github.com/InternetHealthReport/internet-yellow-pages/tree/main/iyp/crawlers/nro#readme) | nro.delegated_stats |
| OONI | Facebook Messenger | https://github.com/ooni/spec/blob/master/nettests/ts-019-facebook-messenger.md | [README](https://github.com/InternetHealthReport/internet-yellow-pages/tree/main/iyp/crawlers/ooni#readme) | ooni.facebookmessenger |
| | Header Field Manipulation Test | https://github.com/ooni/spec/blob/master/nettests/ts-006-header-field-manipulation.md | [README](https://github.com/InternetHealthReport/internet-yellow-pages/tree/main/iyp/crawlers/ooni#readme) | ooni.httpheaderfieldmanipulation |
Expand Down
6 changes: 1 addition & 5 deletions iyp/crawlers/ihr/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -48,14 +48,12 @@ means that Japan ASes depends strongly (AS Hegemony equals 0.19) on AS2497.

### Prefixes' RPKI and IRR status - `rov.py`

Connect prefixes to their origin AS, their AS dependencies, their RPKI/IRR status, and their country
(provided by Maxmind).
Connect prefixes to their origin AS, their AS dependencies, and their RPKI/IRR status.

```Cypher
(:BGPPrefix {prefix: '8.8.8.0/24'})<-[:ORIGINATE]-(:AS {asn: 15169})
(:BGPPrefix {prefix: '8.8.8.0/24'})-[:DEPENDS_ON]->(:AS {asn: 15169})
(:BGPPrefix {prefix: '8.8.8.0/24'})-[:CATEGORIZED]->(:Tag {label: 'RPKI Valid'})
(:BGPPrefix {prefix: '8.8.8.0/24'})-[:COUNTRY]->(:Country {country_code: 'US'})
```

Tag labels (possibly) added by this crawler:
Expand All @@ -69,8 +67,6 @@ Tag labels (possibly) added by this crawler:
- `IRR Invalid,more-specific`
- `IRR NotFound`

The country geo-location is provided by Maxmind.

## Dependence

These crawlers are not depending on other crawlers.
15 changes: 1 addition & 14 deletions iyp/crawlers/ihr/rov.py
Original file line number Diff line number Diff line change
Expand Up @@ -61,12 +61,10 @@ def run(self):
asns = set()
prefixes = set()
tags = set()
countries = set()

orig_links = list()
tag_links = list()
dep_links = list()
country_links = list()

logging.info('Computing links...')
for rec in csv.DictReader(csv_lines):
Expand All @@ -92,12 +90,10 @@ def run(self):
originasn = int(rec['originasn_id'])
rpki_status = 'RPKI ' + rec['rpki_status']
irr_status = 'IRR ' + rec['irr_status']
cc = rec['country_id']

asns.add(originasn)
tags.add(rpki_status)
tags.add(irr_status)
countries.add(cc)

# Compute links
orig_links.append({
Expand All @@ -118,12 +114,6 @@ def run(self):
'props': [self.reference, rec]
})

country_links.append({
'src_id': prefix,
'dst_id': cc,
'props': [self.reference]
})

# Dependency links
asn = int(rec['asn_id'])
asns.add(asn)
Expand All @@ -138,20 +128,17 @@ def run(self):
prefix_id = self.iyp.batch_get_nodes_by_single_prop('BGPPrefix', 'prefix', prefixes, all=False)
self.iyp.batch_add_node_label(list(prefix_id.values()), 'Prefix')
tag_id = self.iyp.batch_get_nodes_by_single_prop('Tag', 'label', tags, all=False)
country_id = self.iyp.batch_get_nodes_by_single_prop('Country', 'country_code', countries)

replace_link_ids(orig_links, asn_id, prefix_id)
replace_link_ids(tag_links, prefix_id, tag_id)
replace_link_ids(country_links, prefix_id, country_id)
replace_link_ids(dep_links, prefix_id, asn_id)

self.iyp.batch_add_links('ORIGINATE', orig_links)
self.iyp.batch_add_links('CATEGORIZED', tag_links)
self.iyp.batch_add_links('DEPENDS_ON', dep_links)
self.iyp.batch_add_links('COUNTRY', country_links)

def unit_test(self):
return super().unit_test(['ORIGINATE', 'CATEGORIZED', 'DEPENDS_ON', 'COUNTRY'])
return super().unit_test(['ORIGINATE', 'CATEGORIZED', 'DEPENDS_ON'])


def main() -> None:
Expand Down
20 changes: 20 additions & 0 deletions iyp/crawlers/maxmind/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,20 @@
# MaxMind -- https://www.maxmind.com/

MaxMind is an IP geolocation service, that provides different kinds of IP
Comment thread
m-appel marked this conversation as resolved.
Outdated
databases, including a [free
tier](https://www.maxmind.com/en/geolite-free-ip-geolocation-data) that maps IP
prefixes to countries. We import the free database into IYP.

## Graph representation

A prefix can also be just a single IP, resulting in /32 or /128 prefixes, which
is intended. We also import the auxiliary country data provided by MaxMind as
relationship properties.

```cypher
(pfx:GeoPrefix {prefix: '202.208.0.0/12'})-[:COUNTRY {continent_name: 'Asia', is_in_european_union: 0}]->(:Country {country_code: 'JP'})
```

## Dependence

This crawler is not depending on other crawlers.
173 changes: 173 additions & 0 deletions iyp/crawlers/maxmind/geolite_country.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,173 @@
import argparse
import hashlib
import json
import logging
import os
import sys
from io import BytesIO
from ipaddress import ip_network
from zipfile import ZipFile

import pandas as pd
import requests

from iyp import BaseCrawler, set_modification_time_from_last_modified_header

ORG = 'MaxMind'
URL = 'https://download.maxmind.com/geoip/databases/GeoLite2-Country-CSV/download?suffix=zip'
NAME = 'maxmind.geolite_country'

SHA256_URL = 'https://download.maxmind.com/geoip/databases/GeoLite2-Country-CSV/download?suffix=zip.sha256'

MAXMIND_ACCOUNT_ID = ''
MAXMIND_LICENSE_KEY = ''
if os.path.exists('config.json'):
MAXMIND_ACCOUNT_ID = json.load(open('config.json', 'r'))['maxmind']['account_id']
MAXMIND_LICENSE_KEY = json.load(open('config.json', 'r'))['maxmind']['license_key']


class Crawler(BaseCrawler):
def __init__(self, organization, url, name):
super().__init__(organization, url, name)
self.reference['reference_url_info'] = 'https://dev.maxmind.com/geoip/geolite2-free-geolocation-data/'

def run(self):
"""Fetch data and push to IYP."""
session = requests.Session()
session.auth = (MAXMIND_ACCOUNT_ID, MAXMIND_LICENSE_KEY)

logging.info('Fetching files...')

hash_res = session.get(SHA256_URL)
if hash_res.status_code != 200:
logging.error('Failed to fetch SHA256 hash.')
hash_res.raise_for_status()
sha256_hash, expected_filename = hash_res.text.split()

db_res = session.get(URL)
if db_res.status_code != 200:
logging.error('Failed to fetch database.')
db_res.raise_for_status()

session.close()

sha256 = hashlib.file_digest(BytesIO(db_res.content), 'sha256').hexdigest()
filename = db_res.headers['Content-Disposition'].removeprefix('attachment; filename=')
if sha256 != sha256_hash or filename != expected_filename:
logging.error('Validation of SHA256 hash or filename failed:')
logging.error(f' SHA256: {sha256}')
logging.error(f' Expected: {sha256_hash}')
logging.error(f' Filename: {filename}')
logging.error(f' Expected: {expected_filename}')
raise ValueError('Validation of SHA256 hash or filename failed.')

set_modification_time_from_last_modified_header(self.reference, db_res)
logging.info(f'Downloaded file: {filename} Last-Modified: {self.reference["reference_time_modification"]}')

logging.info('Parsing data...')

with ZipFile(BytesIO(db_res.content)) as zf:
path_prefix = filename.removesuffix('.zip')
try:
v4_file = zf.getinfo(f'{path_prefix}/GeoLite2-Country-Blocks-IPv4.csv')
v6_file = zf.getinfo(f'{path_prefix}/GeoLite2-Country-Blocks-IPv6.csv')
geoname_file = zf.getinfo(f'{path_prefix}/GeoLite2-Country-Locations-en.csv')
except KeyError as e:
logging.error(f'Failed to extract data from ZIP file: {e}')
raise ValueError(f'Failed to extract data from ZIP file: {e}')

# The geoname file maps the geoname_id to the actual country data
# (e.g., country code).
# Do not parse NA country code as NaN field...
with zf.open(geoname_file) as f:
geoname_df = pd.read_csv(f, keep_default_na=False, na_values=[''])
# Locale is static "en", so not interesting.
geoname_df.pop('locale_code')
# Data contains Asia and Europe as locations with only a continent
# code, which we do not model.
geoname_df = geoname_df[geoname_df['country_iso_code'].notna()]

# Only load network and geoname_id columns. Remaining columns are
# either empty, deprecated, or we do not care about them:
# https://dev.maxmind.com/geoip/docs/databases/city-and-country/#blocks-files
with zf.open(v4_file) as f:
ipv4_df = pd.read_csv(f, usecols=(0, 1))
with zf.open(v6_file) as f:
ipv6_df = pd.read_csv(f, usecols=(0, 1))
ip_df = pd.concat([ipv4_df, ipv6_df])
# Some blocks only have an entry for the registered country, which
# we already cover with the delegated stats file.
ip_df = ip_df[ip_df['geoname_id'].notna()]

# Join IP data with geoname data based on geoname_id field (which we do
# not need afterwards).
ip_merged_df = ip_df.merge(geoname_df)
ip_merged_df.pop('geoname_id')

countries = set(ip_merged_df['country_iso_code'].unique())
prefixes = set()
links = list()

for r in ip_merged_df.itertuples(index=False):
prefix = ip_network(r.network).compressed
prefixes.add(prefix)
links.append(
{
'src_id': prefix,
'dst_id': r.country_iso_code,
'props': [
self.reference,
{
'continent_code': r.continent_code,
'continent_name': r.continent_name,
'country_iso_code': r.country_iso_code,
'country_name': r.country_name,
'is_in_european_union': r.is_in_european_union,
}
]
}
)

logging.info('Pushing data...')

country_id = self.iyp.batch_get_nodes_by_single_prop('Country', 'country_code', countries, all=False)
prefix_id = self.iyp.batch_get_nodes_by_single_prop('GeoPrefix', 'prefix', prefixes, all=False)
self.iyp.batch_add_node_label(list(prefix_id.values()), 'Prefix')

for link in links:
link['src_id'] = prefix_id[link['src_id']]
link['dst_id'] = country_id[link['dst_id']]

self.iyp.batch_add_links('COUNTRY', links)

def unit_test(self):
return super().unit_test(['COUNTRY'])


def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument('--unit-test', action='store_true')
args = parser.parse_args()

FORMAT = '%(asctime)s %(levelname)s %(message)s'
logging.basicConfig(
format=FORMAT,
filename='log/' + NAME + '.log',
level=logging.INFO,
datefmt='%Y-%m-%d %H:%M:%S',
)

logging.info(f'Started: {sys.argv}')

crawler = Crawler(ORG, URL, NAME)
if args.unit_test:
crawler.unit_test()
else:
crawler.run()
crawler.close()
logging.info(f'Finished: {sys.argv}')


if __name__ == '__main__':
main()
sys.exit(0)
Loading