From 7f2b62bb73ad84ccc6a98027abf86ee9353e435a Mon Sep 17 00:00:00 2001 From: Malte Tashiro Date: Mon, 13 Jul 2026 07:37:11 +0000 Subject: [PATCH 1/3] Add MaxMind GeoLite Country crawler Adds support for the free MaxMind IP geolocation database. This data is already integrated via the ihr.rov crawler, but our modeling has changed and geolocation is not really the purpose of the ROV crawler. Separating this into its own crawler makes things more logical, so the country information is not imported by ihr.rov anymore. --- ACKNOWLEDGMENTS.md | 7 + config.json.example | 5 + documentation/data-sources.md | 1 + iyp/crawlers/ihr/README.md | 6 +- iyp/crawlers/ihr/rov.py | 15 +-- iyp/crawlers/maxmind/README.md | 20 +++ iyp/crawlers/maxmind/geolite_country.py | 164 ++++++++++++++++++++++++ 7 files changed, 199 insertions(+), 19 deletions(-) create mode 100644 iyp/crawlers/maxmind/README.md create mode 100644 iyp/crawlers/maxmind/geolite_country.py diff --git a/ACKNOWLEDGMENTS.md b/ACKNOWLEDGMENTS.md index 88ac10b3..9bd5c2da 100644 --- a/ACKNOWLEDGMENTS.md +++ b/ACKNOWLEDGMENTS.md @@ -155,6 +155,13 @@ provided by [IPinfo](https://ipinfo.io) and released the under [Creative Commons Attribution-ShareAlike 4.0 International License](https://creativecommons.org/licenses/by-sa/4.0/). +## MaxMind + +We use the free [GeoLite +Country](https://www.maxmind.com/en/geolite-free-ip-geolocation-data) database +provided by [MaxMind](https://www.maxmind.com) and released the under the +[GeoLite End User License Agreement](https://www.maxmind.com/en/geolite/eula). + ## Number Resource Organization We use the [extended allocation and assignment diff --git a/config.json.example b/config.json.example index 82f21609..30ff219c 100644 --- a/config.json.example +++ b/config.json.example @@ -33,6 +33,11 @@ "token": "" }, + "maxmind": { + "account_id": "", + "license_key": "" + }, + "pch": { "parallel_downloads": 1, "parallel_parsers": 8 diff --git a/documentation/data-sources.md b/documentation/data-sources.md index a2fd8945..1e7a1121 100644 --- a/documentation/data-sources.md +++ b/documentation/data-sources.md @@ -44,6 +44,7 @@ | Internet Assigned Numbers Authority (IANA) | DNS Root Zone File | https://www.iana.org/domains/root/files | [README](https://github.com/InternetHealthReport/internet-yellow-pages/tree/main/iyp/crawlers/iana#readme) | iana.root_zone | | | IPv4/IPv6 Address Allocations (extract) | https://www.iana.org/numbers | [README](https://github.com/InternetHealthReport/internet-yellow-pages/tree/main/iyp/crawlers/iana#readme) | iana.address_space | | IPinfo | IP to Country Mapping| https://ipinfo.io | [README](https://github.com/InternetHealthReport/internet-yellow-pages/tree/main/iyp/crawlers/ipinfo#readme) | ipinfo.ip_country | +| MaxMind | IP to Country Mapping| https://www.maxmind.com/ | [README](https://github.com/InternetHealthReport/internet-yellow-pages/tree/main/iyp/crawlers/maxmind#readme) | maxmind.geolite_country | | NRO | Extended allocation and assignment reports| https://www.nro.net/about/rirs/statistics | [README](https://github.com/InternetHealthReport/internet-yellow-pages/tree/main/iyp/crawlers/nro#readme) | nro.delegated_stats | | OONI | Facebook Messenger | https://github.com/ooni/spec/blob/master/nettests/ts-019-facebook-messenger.md | [README](https://github.com/InternetHealthReport/internet-yellow-pages/tree/main/iyp/crawlers/ooni#readme) | ooni.facebookmessenger | | | Header Field Manipulation Test | https://github.com/ooni/spec/blob/master/nettests/ts-006-header-field-manipulation.md | [README](https://github.com/InternetHealthReport/internet-yellow-pages/tree/main/iyp/crawlers/ooni#readme) | ooni.httpheaderfieldmanipulation | diff --git a/iyp/crawlers/ihr/README.md b/iyp/crawlers/ihr/README.md index fb8fccad..ab3fe7cf 100644 --- a/iyp/crawlers/ihr/README.md +++ b/iyp/crawlers/ihr/README.md @@ -48,14 +48,12 @@ means that Japan ASes depends strongly (AS Hegemony equals 0.19) on AS2497. ### Prefixes' RPKI and IRR status - `rov.py` -Connect prefixes to their origin AS, their AS dependencies, their RPKI/IRR status, and their country -(provided by Maxmind). +Connect prefixes to their origin AS, their AS dependencies, and their RPKI/IRR status. ```Cypher (:BGPPrefix {prefix: '8.8.8.0/24'})<-[:ORIGINATE]-(:AS {asn: 15169}) (:BGPPrefix {prefix: '8.8.8.0/24'})-[:DEPENDS_ON]->(:AS {asn: 15169}) (:BGPPrefix {prefix: '8.8.8.0/24'})-[:CATEGORIZED]->(:Tag {label: 'RPKI Valid'}) -(:BGPPrefix {prefix: '8.8.8.0/24'})-[:COUNTRY]->(:Country {country_code: 'US'}) ``` Tag labels (possibly) added by this crawler: @@ -69,8 +67,6 @@ Tag labels (possibly) added by this crawler: - `IRR Invalid,more-specific` - `IRR NotFound` -The country geo-location is provided by Maxmind. - ## Dependence These crawlers are not depending on other crawlers. diff --git a/iyp/crawlers/ihr/rov.py b/iyp/crawlers/ihr/rov.py index 0f02637e..0a14d7ea 100644 --- a/iyp/crawlers/ihr/rov.py +++ b/iyp/crawlers/ihr/rov.py @@ -61,12 +61,10 @@ def run(self): asns = set() prefixes = set() tags = set() - countries = set() orig_links = list() tag_links = list() dep_links = list() - country_links = list() logging.info('Computing links...') for rec in csv.DictReader(csv_lines): @@ -92,12 +90,10 @@ def run(self): originasn = int(rec['originasn_id']) rpki_status = 'RPKI ' + rec['rpki_status'] irr_status = 'IRR ' + rec['irr_status'] - cc = rec['country_id'] asns.add(originasn) tags.add(rpki_status) tags.add(irr_status) - countries.add(cc) # Compute links orig_links.append({ @@ -118,12 +114,6 @@ def run(self): 'props': [self.reference, rec] }) - country_links.append({ - 'src_id': prefix, - 'dst_id': cc, - 'props': [self.reference] - }) - # Dependency links asn = int(rec['asn_id']) asns.add(asn) @@ -138,20 +128,17 @@ def run(self): prefix_id = self.iyp.batch_get_nodes_by_single_prop('BGPPrefix', 'prefix', prefixes, all=False) self.iyp.batch_add_node_label(list(prefix_id.values()), 'Prefix') tag_id = self.iyp.batch_get_nodes_by_single_prop('Tag', 'label', tags, all=False) - country_id = self.iyp.batch_get_nodes_by_single_prop('Country', 'country_code', countries) replace_link_ids(orig_links, asn_id, prefix_id) replace_link_ids(tag_links, prefix_id, tag_id) - replace_link_ids(country_links, prefix_id, country_id) replace_link_ids(dep_links, prefix_id, asn_id) self.iyp.batch_add_links('ORIGINATE', orig_links) self.iyp.batch_add_links('CATEGORIZED', tag_links) self.iyp.batch_add_links('DEPENDS_ON', dep_links) - self.iyp.batch_add_links('COUNTRY', country_links) def unit_test(self): - return super().unit_test(['ORIGINATE', 'CATEGORIZED', 'DEPENDS_ON', 'COUNTRY']) + return super().unit_test(['ORIGINATE', 'CATEGORIZED', 'DEPENDS_ON']) def main() -> None: diff --git a/iyp/crawlers/maxmind/README.md b/iyp/crawlers/maxmind/README.md new file mode 100644 index 00000000..0ec629d6 --- /dev/null +++ b/iyp/crawlers/maxmind/README.md @@ -0,0 +1,20 @@ +# MaxMind -- https://www.maxmind.com/ + +MaxMind is an IP geolocation service, that provides different kinds of IP +databases, including a [free +tier](https://www.maxmind.com/en/geolite-free-ip-geolocation-data) that maps IP +prefixes to countries. We import the free database into IYP. + +## Graph representation + +A prefix can also be just a single IP, resulting in /32 or /128 prefixes, which +is intended. We also import the auxiliary country data provided by MaxMind as +relationship properties. + +```cypher +(pfx:GeoPrefix {prefix: '202.208.0.0/12'})-[:COUNTRY {continent_name: 'Asia', is_in_european_union: 0}]->(:Country {country_code: 'JP'}) +``` + +## Dependence + +This crawler is not depending on other crawlers. diff --git a/iyp/crawlers/maxmind/geolite_country.py b/iyp/crawlers/maxmind/geolite_country.py new file mode 100644 index 00000000..a61024ca --- /dev/null +++ b/iyp/crawlers/maxmind/geolite_country.py @@ -0,0 +1,164 @@ +import argparse +import hashlib +import json +import logging +import os +import sys +from io import BytesIO +from ipaddress import ip_network +from zipfile import ZipFile + +import pandas as pd +import requests + +from iyp import BaseCrawler, set_modification_time_from_last_modified_header + +ORG = 'MaxMind' +URL = 'https://download.maxmind.com/geoip/databases/GeoLite2-Country-CSV/download?suffix=zip' +NAME = 'maxmind.geolite_country' + +SHA256_URL = 'https://download.maxmind.com/geoip/databases/GeoLite2-Country-CSV/download?suffix=zip.sha256' + +MAXMIND_ACCOUNT_ID = '' +MAXMIND_LICENSE_KEY = '' +if os.path.exists('config.json'): + MAXMIND_ACCOUNT_ID = json.load(open('config.json', 'r'))['maxmind']['account_id'] + MAXMIND_LICENSE_KEY = json.load(open('config.json', 'r'))['maxmind']['license_key'] + + +class Crawler(BaseCrawler): + def __init__(self, organization, url, name): + super().__init__(organization, url, name) + self.reference['reference_url_info'] = 'https://dev.maxmind.com/geoip/geolite2-free-geolocation-data/' + + def run(self): + """Fetch data and push to IYP.""" + session = requests.Session() + session.auth = (MAXMIND_ACCOUNT_ID, MAXMIND_LICENSE_KEY) + + logging.info('Fetching files...') + + hash_res = session.get(SHA256_URL) + if hash_res.status_code != 200: + logging.error('Failed to fetch SHA256 hash.') + hash_res.raise_for_status() + sha256_hash, expected_filename = hash_res.text.split() + + db_res = session.get(URL) + if db_res.status_code != 200: + logging.error('Failed to fetch database.') + db_res.raise_for_status() + + session.close() + + sha256 = hashlib.file_digest(BytesIO(db_res.content), 'sha256').hexdigest() + filename = db_res.headers['Content-Disposition'].removeprefix('attachment; filename=') + if sha256 != sha256_hash or filename != expected_filename: + logging.error('Validation of SHA256 hash or filename failed:') + logging.error(f' SHA256: {sha256}') + logging.error(f' Expected: {sha256_hash}') + logging.error(f' Filename: {filename}') + logging.error(f' Expected: {expected_filename}') + raise ValueError('Validation of SHA256 hash or filename failed.') + + set_modification_time_from_last_modified_header(self.reference, db_res) + logging.info(f'Downloaded file: {filename} Last-Modified: {self.reference["reference_time_modification"]}') + + logging.info('Parsing data...') + + with ZipFile(BytesIO(db_res.content)) as zf: + path_prefix = filename.removesuffix('.zip') + try: + v4_file = zf.getinfo(f'{path_prefix}/GeoLite2-Country-Blocks-IPv4.csv') + v6_file = zf.getinfo(f'{path_prefix}/GeoLite2-Country-Blocks-IPv6.csv') + geoname_file = zf.getinfo(f'{path_prefix}/GeoLite2-Country-Locations-en.csv') + except KeyError as e: + logging.error(f'Failed to extract data from ZIP file: {e}') + raise ValueError(f'Failed to extract data from ZIP file: {e}') + + with zf.open(geoname_file) as f: + geoname_df = pd.read_csv(f, keep_default_na=False, na_values=['']) + geoname_df.pop('locale_code') + # Data contains Asia and Europe as locations with only a continent + # code, which we do not model. + geoname_df = geoname_df[geoname_df['country_iso_code'].notna()] + + with zf.open(v4_file) as f: + ipv4_df = pd.read_csv(f, usecols=(0, 1)) + with zf.open(v6_file) as f: + ipv6_df = pd.read_csv(f, usecols=(0, 1)) + ip_df = pd.concat([ipv4_df, ipv6_df]) + # Some blocks only have an entry for the registered country, which + # we already cover with the delegated stats file. + ip_df = ip_df[ip_df['geoname_id'].notna()] + + ip_merged_df = ip_df.merge(geoname_df) + ip_merged_df.pop('geoname_id') + + countries = set(ip_merged_df['country_iso_code'].unique()) + prefixes = set() + links = list() + + for r in ip_merged_df.itertuples(index=False): + prefix = ip_network(r.network).compressed + prefixes.add(prefix) + links.append( + { + 'src_id': prefix, + 'dst_id': r.country_iso_code, + 'props': [ + self.reference, + { + 'continent_code': r.continent_code, + 'continent_name': r.continent_name, + 'country_iso_code': r.country_iso_code, + 'country_name': r.country_name, + 'is_in_european_union': r.is_in_european_union, + } + ] + } + ) + + logging.info('Pushing data...') + + country_id = self.iyp.batch_get_nodes_by_single_prop('Country', 'country_code', countries, all=False) + prefix_id = self.iyp.batch_get_nodes_by_single_prop('GeoPrefix', 'prefix', prefixes, all=False) + self.iyp.batch_add_node_label(list(prefix_id.values()), 'Prefix') + + for link in links: + link['src_id'] = prefix_id[link['src_id']] + link['dst_id'] = country_id[link['dst_id']] + + self.iyp.batch_add_links('COUNTRY', links) + + def unit_test(self): + return super().unit_test(['COUNTRY']) + + +def main() -> None: + parser = argparse.ArgumentParser() + parser.add_argument('--unit-test', action='store_true') + args = parser.parse_args() + + FORMAT = '%(asctime)s %(levelname)s %(message)s' + logging.basicConfig( + format=FORMAT, + filename='log/' + NAME + '.log', + level=logging.INFO, + datefmt='%Y-%m-%d %H:%M:%S', + ) + + logging.info(f'Started: {sys.argv}') + + crawler = Crawler(ORG, URL, NAME) + if args.unit_test: + crawler.unit_test() + else: + crawler.run() + crawler.close() + logging.info(f'Finished: {sys.argv}') + + +if __name__ == '__main__': + main() + sys.exit(0) From 1c4cac6c82ae9dde3dc7bd4be0b5567bdb4cc663 Mon Sep 17 00:00:00 2001 From: Malte Tashiro Date: Mon, 13 Jul 2026 11:55:48 +0000 Subject: [PATCH 2/3] Add some more comments --- iyp/crawlers/maxmind/geolite_country.py | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/iyp/crawlers/maxmind/geolite_country.py b/iyp/crawlers/maxmind/geolite_country.py index a61024ca..dbd682e2 100644 --- a/iyp/crawlers/maxmind/geolite_country.py +++ b/iyp/crawlers/maxmind/geolite_country.py @@ -76,13 +76,20 @@ def run(self): logging.error(f'Failed to extract data from ZIP file: {e}') raise ValueError(f'Failed to extract data from ZIP file: {e}') + # The geoname file maps the geoname_id to the actual country data + # (e.g., country code). + # Do not parse NA country code as NaN field... with zf.open(geoname_file) as f: geoname_df = pd.read_csv(f, keep_default_na=False, na_values=['']) + # Locale is static "en", so not interesting. geoname_df.pop('locale_code') # Data contains Asia and Europe as locations with only a continent # code, which we do not model. geoname_df = geoname_df[geoname_df['country_iso_code'].notna()] + # Only load network and geoname_id columns. Remaining columns are + # either empty, deprecated, or we do not care about them: + # https://dev.maxmind.com/geoip/docs/databases/city-and-country/#blocks-files with zf.open(v4_file) as f: ipv4_df = pd.read_csv(f, usecols=(0, 1)) with zf.open(v6_file) as f: @@ -92,6 +99,8 @@ def run(self): # we already cover with the delegated stats file. ip_df = ip_df[ip_df['geoname_id'].notna()] + # Join IP data with geoname data based on geoname_id field (which we do + # not need afterwards). ip_merged_df = ip_df.merge(geoname_df) ip_merged_df.pop('geoname_id') From 3a73c3fb28464a8355edf73ae4dc7fbfc1707d60 Mon Sep 17 00:00:00 2001 From: Malte Tashiro Date: Wed, 15 Jul 2026 10:59:47 +0900 Subject: [PATCH 3/3] Apply suggestions from code review Co-authored-by: Romain Fontugne --- ACKNOWLEDGMENTS.md | 2 +- iyp/crawlers/maxmind/README.md | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/ACKNOWLEDGMENTS.md b/ACKNOWLEDGMENTS.md index 9bd5c2da..1ee8f3eb 100644 --- a/ACKNOWLEDGMENTS.md +++ b/ACKNOWLEDGMENTS.md @@ -159,7 +159,7 @@ License](https://creativecommons.org/licenses/by-sa/4.0/). We use the free [GeoLite Country](https://www.maxmind.com/en/geolite-free-ip-geolocation-data) database -provided by [MaxMind](https://www.maxmind.com) and released the under the +provided by [MaxMind](https://www.maxmind.com) and released under the [GeoLite End User License Agreement](https://www.maxmind.com/en/geolite/eula). ## Number Resource Organization diff --git a/iyp/crawlers/maxmind/README.md b/iyp/crawlers/maxmind/README.md index 0ec629d6..afe3df03 100644 --- a/iyp/crawlers/maxmind/README.md +++ b/iyp/crawlers/maxmind/README.md @@ -1,6 +1,6 @@ # MaxMind -- https://www.maxmind.com/ -MaxMind is an IP geolocation service, that provides different kinds of IP +MaxMind is an IP geolocation service that provides different kinds of IP databases, including a [free tier](https://www.maxmind.com/en/geolite-free-ip-geolocation-data) that maps IP prefixes to countries. We import the free database into IYP.