summaryrefslogtreecommitdiff
path: root/data_gathering.ipynb
diff options
context:
space:
mode:
authorbicker <bickerkards@gmail.com>2017-07-14 01:23:11 +0200
committerbicker <bickerkards@gmail.com>2017-07-14 01:23:11 +0200
commit092af5e8fefe1dc0f7e5d02bb3050d05e22dd475 (patch)
treeada870c384ac8ac559c74e697fddfbf5c0bc412a /data_gathering.ipynb
parent9a3b2702bbcbd67eab592b043b7dcd50b48afe3b (diff)
Diffstat (limited to 'data_gathering.ipynb')
-rw-r--r--data_gathering.ipynb1184
1 files changed, 579 insertions, 605 deletions
diff --git a/data_gathering.ipynb b/data_gathering.ipynb
index 9070f80..cdbaf61 100644
--- a/data_gathering.ipynb
+++ b/data_gathering.ipynb
@@ -9,8 +9,21 @@
},
"outputs": [],
"source": [
- "# Constantijn Bicker Caarten\n",
- "# Last updated: 13-06-2017"
+ "# Written by Constantijn Bicker Caarten\n",
+ "# Last updated: 07-07-2017\n",
+ "#\n",
+ "#\n",
+ "# This code gathers data on TLDs in the DNS. This data \n",
+ "# consists out of A, AAAA, DNSKEY, DS and NS records, \n",
+ "# as well as TCP and UDP support, response time and\n",
+ "# anycast support."
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "metadata": {},
+ "source": [
+ "# Imports"
]
},
{
@@ -21,6 +34,7 @@
},
"outputs": [],
"source": [
+ "from dns.resolver import Resolver, NXDOMAIN, query\n",
"from socket import error as socket_error\n",
"from urllib.request import urlopen\n",
"from dns.query import udp, tcp\n",
@@ -29,50 +43,28 @@
"from uuid import uuid4\n",
"from tqdm import tqdm\n",
"\n",
+ "import matplotlib as plt\n",
"import pandas as pd\n",
"import numpy as np\n",
- "import matplotlib as plt\n",
"\n",
- "import subprocess\n",
+ "import datetime\n",
+ "import requests\n",
"import socket\n",
"import copy\n",
- "import time\n",
"import json\n",
- "import os\n",
- "\n",
- "%matplotlib inline"
+ "import os"
]
},
{
- "cell_type": "code",
- "execution_count": 76,
+ "cell_type": "markdown",
"metadata": {},
- "outputs": [
- {
- "name": "stdout",
- "output_type": "stream",
- "text": [
- "mkdir: cannot create directory ‘data’: File exists\n",
- "mkdir: cannot create directory ‘data/whois’: File exists\n",
- "mkdir: cannot create directory ‘data/dig’: File exists\n",
- "mkdir: cannot create directory ‘data/backup’: File exists\n",
- "mkdir: cannot create directory ‘data/lists’: File exists\n"
- ]
- }
- ],
"source": [
- "# First time run\n",
- "!mkdir data\n",
- "!mkdir data/whois\n",
- "!mkdir data/cymru\n",
- "!mkdir data/dig\n",
- "!mkdir data/backup\n",
- "!mkdir data/lists"
+ "# Constants"
]
},
{
"cell_type": "code",
- "execution_count": 199,
+ "execution_count": 3,
"metadata": {
"collapsed": true
},
@@ -81,9 +73,54 @@
"ZSK = 256\n",
"KSK = 257\n",
"\n",
- "pie = (6, 6)\n",
+ "ATLAS_API_KEY = '' # Add your Atlas API key\n",
+ "ATLAS_BILL_TO = '' # Add your Atlas account email\n",
+ "\n",
+ "URL_DNS_MEASUREMENT_CREATE = 'https://atlas.ripe.net:443/api/v2/measurements/dns/'\n",
+ "URL_DNS_MEASUREMENT_GET = 'https://atlas.ripe.net:443/api/v2/measurements/dns/'\n",
+ "\n",
+ "HEADERS = {'Content-type': 'application/json', 'Accept': 'text/plain'}\n",
"\n",
- "newline = '\\n'"
+ "NEWLINE = '\\n'"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 4,
+ "metadata": {
+ "collapsed": true
+ },
+ "outputs": [],
+ "source": [
+ "payload = {\n",
+ " 'bill_to': ATLAS_BILL_TO,\n",
+ " 'is_oneoff': True,\n",
+ " 'definitions': [],\n",
+ " 'probes': []\n",
+ "}\n",
+ " \n",
+ "definition = {\n",
+ " 'af': 4,\n",
+ " 'query_class': 'IN',\n",
+ " 'query_type': '',\n",
+ " 'query_argument': '',\n",
+ " 'description': '',\n",
+ " 'use_probe_resolver': True,\n",
+ " 'resolve_on_probe': False,\n",
+ " 'set_nsid_bit': True,\n",
+ " 'protocol': 'UDP',\n",
+ " 'udp_payload_size': 512,\n",
+ " 'retry': 0,\n",
+ " 'skip_dns_check': False,\n",
+ " 'include_qbuf': False,\n",
+ " 'include_abuf': True,\n",
+ " 'prepend_probe_id': False,\n",
+ " 'set_rd_bit': False,\n",
+ " 'set_do_bit': False,\n",
+ " 'set_cd_bit': False,\n",
+ " 'type': 'dns',\n",
+ " 'is_public': True\n",
+ "}"
]
},
{
@@ -95,7 +132,7 @@
},
{
"cell_type": "code",
- "execution_count": 157,
+ "execution_count": 5,
"metadata": {
"collapsed": true,
"scrolled": false
@@ -106,18 +143,18 @@
" '''Writes a list to a file with each value on a new line'''\n",
" with open(fn, 'w') as f:\n",
" for datum in data:\n",
- " f.write(datum + newline)\n",
+ " f.write(datum + NEWLINE)\n",
" \n",
"def append_list(fn, data):\n",
" '''Appends a list to a file with each value on a new line'''\n",
" with open(fn, 'a') as f:\n",
" for datum in data:\n",
- " f.write(datum + newline)\n",
+ " f.write(datum + NEWLINE)\n",
" \n",
"def read_list(fn):\n",
" '''Reads a file and '''\n",
" with open(fn, 'r') as f:\n",
- " return [line.strip(newline) for line in f]\n",
+ " return [line.strip(NEWLINE) for line in f]\n",
" \n",
"def write_json(fn, data):\n",
" with open(fn, 'w') as f:\n",
@@ -126,20 +163,23 @@
"def read_json(fn):\n",
" '''Read a json file (fn) and returns it as a dictionary'''\n",
" with open(fn, 'r') as f:\n",
- " return json.dumps(f.read())"
+ " return json.loads(f.read())"
]
},
{
"cell_type": "code",
- "execution_count": 134,
- "metadata": {},
+ "execution_count": 6,
+ "metadata": {
+ "collapsed": true
+ },
"outputs": [],
"source": [
"def write_data(fn, data):\n",
" \"\"\"Backs up the previous version of the data if it exists and writes the new data to a file.\"\"\"\n",
" # Backs up the previous data if it exists.\n",
" try:\n",
- " write_json(\"data/backup/{}.json \".format(fn) + time.ctime().replace(' ', '-'), \n",
+ " now = datetime.datetime.now().strftime('%H:%M-%d-%m-%Y')\n",
+ " write_json(\"data/backup/{}_{}.json \".format(fn, now), \n",
" read_json(\"data/{}.json\".format(fn)))\n",
" except:\n",
" pass\n",
@@ -149,25 +189,25 @@
},
{
"cell_type": "code",
- "execution_count": 6,
+ "execution_count": 7,
"metadata": {
"collapsed": true
},
"outputs": [],
"source": [
"def find(lst, key, value):\n",
- " for i, dic in enumerate(lst):\n",
+ " '''Finds the first index of a list \n",
+ " lst where the key matches the value'''\n",
+ " for index, dic in enumerate(lst):\n",
" if dic[key] == value:\n",
- " return i\n",
- " return None\n",
- "\n",
- "def sort_dict_list(data, x):\n",
- " return sorted(data, key=lambda k: k[x]) "
+ " return index\n",
+ " \n",
+ " return None"
]
},
{
"cell_type": "code",
- "execution_count": 7,
+ "execution_count": 8,
"metadata": {
"collapsed": true
},
@@ -192,7 +232,7 @@
},
{
"cell_type": "code",
- "execution_count": 119,
+ "execution_count": 9,
"metadata": {
"collapsed": true,
"scrolled": true
@@ -203,25 +243,98 @@
" pass\n",
"\n",
"def test_tcp_udp(data, timeout = 5):\n",
- " pbar = tqdm(total=len(data))\n",
+ " data_copy = copy.deepcopy(data)\n",
+ " \n",
+ " pbar = tqdm(total=len(data_copy))\n",
"\n",
- " for datum in data:\n",
- " for p in (udp, tcp):\n",
+ " for datum in data_copy:\n",
+ " protocols = []\n",
+ " \n",
+ " if 'tcp' in datum and not datum['tcp']:\n",
+ " protocols.append(udp)\n",
+ " \n",
+ " if ('udp' in datum and not datum['udp']):\n",
+ " protocols.append(tcp)\n",
+ " \n",
+ " for p in protocols:\n",
" # Create SOA query\n",
" m = dns.message.make_query(datum['tld'], dns.rdatatype.SOA)\n",
+ " \n",
" try: \n",
" a = p(m, datum['ip'], timeout = timeout)\n",
+ " \n",
" # We expect NOERROR RCODE (0) and an answer\n",
" if a.rcode() == 0 and len(a.answer) > 0:\n",
" datum[p.__name__] = True\n",
- "\n",
" else:\n",
" raise CustomDNSException('failed')\n",
+ " \n",
" except (dns.exception.Timeout, socket_error, CustomDNSException):\n",
" datum[p.__name__] = False\n",
"\n",
" pbar.update(1)\n",
- " pbar.close()"
+ " pbar.close()\n",
+ " \n",
+ " return data_copy"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 10,
+ "metadata": {
+ "collapsed": true
+ },
+ "outputs": [],
+ "source": [
+ "def find_nxdomain(tld, max_tries = 3):\n",
+ " for _ in range(max_tries):\n",
+ " domain = '{}.{}'.format(str(uuid4()), tld)\n",
+ " \n",
+ " try:\n",
+ " query(domain)\n",
+ " except NXDOMAIN:\n",
+ " return domain\n",
+ " except:\n",
+ " pass\n",
+ " \n",
+ " return None\n",
+ "\n",
+ "def find_nxdomain_wildcard(tld, max_tries = 3):\n",
+ " for _ in range(max_tries):\n",
+ " domain = '{}.{}'.format(str(uuid4()), tld)\n",
+ "\n",
+ " response = !dig soa +noall +authority +noidn {domain}\n",
+ "\n",
+ " if len(response) > 0 and response[0].startswith(tld):\n",
+ " return domain\n",
+ " \n",
+ " return None"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "metadata": {},
+ "source": [
+ "# Init"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": null,
+ "metadata": {},
+ "outputs": [],
+ "source": [
+ "# First time run\n",
+ "!mkdir data\n",
+ "\n",
+ "!mkdir data/dig\n",
+ "!mkdir data/lists\n",
+ "!mkdir data/cymru\n",
+ "!mkdir data/whois\n",
+ "!mkdir data/atlas\n",
+ "!mkdir data/atlas/ns\n",
+ "!mkdir data/atlas/soa\n",
+ "!mkdir data/backup"
]
},
{
@@ -233,24 +346,24 @@
},
{
"cell_type": "code",
- "execution_count": 23,
+ "execution_count": 11,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
- "--2017-06-13 13:09:02-- https://data.iana.org/TLD/tlds-alpha-by-domain.txt\n",
+ "--2017-07-12 20:37:59-- https://data.iana.org/TLD/tlds-alpha-by-domain.txt\n",
"Loaded CA certificate '/etc/ssl/certs/ca-certificates.crt'\n",
- "Resolving data.iana.org... 2606:2800:11f:bb5:f27:227f:1bbf:a0e, 72.21.81.189\n",
- "Connecting to data.iana.org|2606:2800:11f:bb5:f27:227f:1bbf:a0e|:443... connected.\n",
+ "Resolving data.iana.org... 72.21.81.189, 2606:2800:11f:bb5:f27:227f:1bbf:a0e\n",
+ "Connecting to data.iana.org|72.21.81.189|:443... connected.\n",
"HTTP request sent, awaiting response... 200 OK\n",
- "Length: 10295 (10K) [text/plain]\n",
+ "Length: 10433 (10K) [text/plain]\n",
"Saving to: ‘data/lists/tlds’\n",
"\n",
- "data/lists/tlds 100%[===================>] 10.05K --.-KB/s in 0s \n",
+ "data/lists/tlds 100%[===================>] 10.19K --.-KB/s in 0s \n",
"\n",
- "2017-06-13 13:09:08 (128 MB/s) - ‘data/lists/tlds’ saved [10295/10295]\n",
+ "2017-07-12 20:37:59 (189 MB/s) - ‘data/lists/tlds’ saved [10433/10433]\n",
"\n"
]
}
@@ -262,7 +375,7 @@
},
{
"cell_type": "code",
- "execution_count": 24,
+ "execution_count": 12,
"metadata": {
"collapsed": true
},
@@ -274,11 +387,11 @@
{
"cell_type": "code",
"execution_count": null,
- "metadata": {
- "collapsed": true
- },
+ "metadata": {},
"outputs": [],
"source": [
+ "# Gathers WHOIS records for each TLD.\n",
+ "\n",
"pbar = tqdm(total=len(tlds))\n",
"\n",
"for tld in tlds:\n",
@@ -290,7 +403,7 @@
},
{
"cell_type": "code",
- "execution_count": 92,
+ "execution_count": 13,
"metadata": {},
"outputs": [
{
@@ -303,6 +416,8 @@
}
],
"source": [
+ "# Gathers empty or missing WHOIS records.\n",
+ "\n",
"indir = 'data/whois/'\n",
"\n",
"for root, dirs, filenames in os.walk(indir):\n",
@@ -321,10 +436,12 @@
},
{
"cell_type": "code",
- "execution_count": 126,
+ "execution_count": 14,
"metadata": {},
"outputs": [],
"source": [
+ "# Extracts the creation date and organisations for each TLD from the WHOIS record.\n",
+ "\n",
"data_tlds = [{'tld': tld, 'organisations': []} for tld in tlds]\n",
"\n",
"for root, dirs, filenames in os.walk(indir):\n",
@@ -337,27 +454,53 @@
" elif line.startswith('organisation'):\n",
" _, org = line.split('rganisation:')\n",
" index = find(data_tlds, 'tld', fn)\n",
- " data_tlds[index]['organisations'].append(org.strip(newline))"
+ " data_tlds[index]['organisations'].append(org.strip(NEWLINE))"
]
},
{
"cell_type": "code",
- "execution_count": 136,
+ "execution_count": 15,
"metadata": {},
"outputs": [],
"source": [
+ "# Gets the type of each TLD listed in the table of the url.\n",
+ "\n",
+ "url = \"https://www.iana.org/domains/root/db/\"\n",
+ "html = urlopen(url)\n",
+ "soup = BeautifulSoup(html, 'html5lib')\n",
+ "\n",
+ "for table in soup.find_all(attrs={'class': 'iana-table'}):\n",
+ " values = [td.get_text(strip=True) for td in table.find_all('td')]\n",
+ " values = [td for td in table.find_all('td')]\n",
+ "\n",
+ "for i in range(0, len(values), 3):\n",
+ " tld = str(values[i].findAll('a', href=True)[0]).split('.html')[0][26:].upper()\n",
+ " index = find(data_tlds, 'tld', tld)\n",
+ " \n",
+ " if index != None:\n",
+ " data_tlds[index]['type'] = values[i + 1].get_text(strip = True)"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 16,
+ "metadata": {
+ "collapsed": true
+ },
+ "outputs": [],
+ "source": [
"write_data('data_tlds', data_tlds)"
]
},
{
"cell_type": "code",
- "execution_count": 152,
+ "execution_count": 17,
"metadata": {
"collapsed": true
},
"outputs": [],
"source": [
- "# Special TLDs are the same as record types or classes which do not work in bulk.\n",
+ "# Special TLDs are the same as record types or classes which do not work in some bulk operations.\n",
"special_tlds = ['CH', 'IN', 'MD', 'MG', 'MR', 'MX']\n",
"write_list('data/lists/tlds', [tld for tld in tlds if tld not in special_tlds])"
]
@@ -371,18 +514,9 @@
},
{
"cell_type": "code",
- "execution_count": 18,
+ "execution_count": null,
"metadata": {},
- "outputs": [
- {
- "name": "stdout",
- "output_type": "stream",
- "text": [
- "Gathering name servers.\n",
- "Done.\n"
- ]
- }
- ],
+ "outputs": [],
"source": [
"# Gathers the name servers of every TLD using dig.\n",
"print('Gathering name servers.')\n",
@@ -396,15 +530,16 @@
},
{
"cell_type": "code",
- "execution_count": 139,
+ "execution_count": null,
"metadata": {
"collapsed": true
},
"outputs": [],
"source": [
+ "# Parses the NS records.\n",
+ "\n",
"data_ns = []\n",
"\n",
- "# Parses the answers of dig.\n",
"with open('data/dig/tld_nss', 'r') as f:\n",
" for line in f:\n",
" if not line.startswith('.'):\n",
@@ -423,31 +558,18 @@
},
{
"cell_type": "code",
- "execution_count": 20,
- "metadata": {
- "collapsed": true
- },
+ "execution_count": null,
+ "metadata": {},
"outputs": [],
"source": [
- "write_list('data/lists/nss', set([datum['ns'] for datum in data]))"
+ "write_list('data/lists/nss', set([datum['ns'] for datum in data_ns]))"
]
},
{
"cell_type": "code",
- "execution_count": 21,
+ "execution_count": null,
"metadata": {},
- "outputs": [
- {
- "name": "stdout",
- "output_type": "stream",
- "text": [
- "Gathering IPv4 addresses.\n",
- "Done.\n",
- "Gathering IPv6 addresses.\n",
- "Done.\n"
- ]
- }
- ],
+ "outputs": [],
"source": [
"print('Gathering IPv4 addresses.')\n",
"!dig +noall +answer +noidn A -f data/lists/nss > data/dig/ns_ipv4s\n",
@@ -460,10 +582,8 @@
},
{
"cell_type": "code",
- "execution_count": 143,
- "metadata": {
- "collapsed": true
- },
+ "execution_count": null,
+ "metadata": {},
"outputs": [],
"source": [
"ns_ipv4s = ns_ips('data/dig/ns_ipv4s')\n",
@@ -471,7 +591,10 @@
"\n",
"data_ips = []\n",
"\n",
- "for datum in data:\n",
+ "# Adds the IP adress to the dictionary and creates a copy \n",
+ "# in case a name server has multiple IP addresses or has \n",
+ "# both a IPv4 and IPv6 address.\n",
+ "for datum in data_ns:\n",
" if datum['ns'] in ns_ipv4s:\n",
" for ip in ns_ipv4s[datum['ns']]:\n",
" new_datum = copy.deepcopy(datum)\n",
@@ -487,7 +610,7 @@
},
{
"cell_type": "code",
- "execution_count": 144,
+ "execution_count": null,
"metadata": {
"collapsed": true
},
@@ -505,7 +628,7 @@
},
{
"cell_type": "code",
- "execution_count": 73,
+ "execution_count": null,
"metadata": {
"collapsed": true
},
@@ -518,8 +641,9 @@
},
{
"cell_type": "code",
- "execution_count": 77,
+ "execution_count": null,
"metadata": {
+ "collapsed": true,
"scrolled": false
},
"outputs": [],
@@ -529,12 +653,15 @@
},
{
"cell_type": "code",
- "execution_count": 145,
+ "execution_count": null,
"metadata": {
+ "collapsed": true,
"scrolled": false
},
"outputs": [],
"source": [
+ "# Makes a dictionary with the IP addresses as key and a list of ASNs as value.\n",
+ "\n",
"ip_asns = {}\n",
"\n",
"with open('data/cymru/ip_asns', 'r') as f:\n",
@@ -551,23 +678,12 @@
},
{
"cell_type": "code",
- "execution_count": 146,
+ "execution_count": null,
"metadata": {
"collapsed": true
},
"outputs": [],
"source": [
- "for datum in data_ips:\n",
- " if datum['ip'] in ip_asns:\n",
- " datum['asn'] = ip_asns[datum['ip']]"
- ]
- },
- {
- "cell_type": "code",
- "execution_count": 147,
- "metadata": {},
- "outputs": [],
- "source": [
"data_asns = []\n",
"\n",
"for datum in data_ips:\n",
@@ -580,7 +696,7 @@
},
{
"cell_type": "code",
- "execution_count": 148,
+ "execution_count": null,
"metadata": {
"collapsed": true
},
@@ -598,68 +714,34 @@
},
{
"cell_type": "code",
- "execution_count": 149,
- "metadata": {
- "scrolled": true
- },
- "outputs": [],
- "source": [
- "test_tcp_udp(data_ips)"
- ]
- },
- {
- "cell_type": "code",
"execution_count": null,
"metadata": {
"collapsed": true
},
"outputs": [],
"source": [
- "data_no_tcp_and_udp = [datum for datum in data_ips if not datum['tcp'] and not datum['udp']]\n",
- "\n",
- "test_tcp_udp(data_no_tcp_and_udp, timeout = 10)"
+ "data_ips = read_json('data/data_ips.json')"
]
},
{
"cell_type": "code",
"execution_count": null,
- "metadata": {
- "collapsed": true
- },
- "outputs": [],
- "source": [
- "len([datum for datum in data_ips if not datum['tcp'] and not datum['udp']]), len(data_ips)"
- ]
- },
- {
- "cell_type": "code",
- "execution_count": 142,
- "metadata": {
- "collapsed": true
- },
+ "metadata": {},
"outputs": [],
"source": [
- "write_data('data_prot', data_ips)"
+ "data_reach = test_tcp_udp(data_ips)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
- "collapsed": true
- },
- "outputs": [],
- "source": []
- },
- {
- "cell_type": "code",
- "execution_count": null,
- "metadata": {
- "collapsed": true
+ "scrolled": true
},
"outputs": [],
"source": [
- "len(set([datum['tld'] for datum in data_ips if not datum['tcp'] and not datum['udp']]))"
+ "# Retries testing TCP or UDP\n",
+ "data_reach = test_tcp_udp(data_reach, timeout = 15)"
]
},
{
@@ -670,32 +752,48 @@
},
"outputs": [],
"source": [
- "with open('data/backup/data_tcp_udp', 'w') as f:\n",
- " f.write(json.dumps(data_ips))"
+ "write_data('data_reach', data_reach)"
]
},
{
- "cell_type": "code",
- "execution_count": null,
- "metadata": {
- "collapsed": true
- },
- "outputs": [],
+ "cell_type": "markdown",
+ "metadata": {},
"source": [
- "# with open('data/backup/data_tcp_udp', 'r') as f:\n",
- "# data_ips = json.loads(f.read())"
+ "# Credibility"
]
},
{
"cell_type": "code",
- "execution_count": null,
- "metadata": {
- "collapsed": true,
- "scrolled": false
- },
- "outputs": [],
+ "execution_count": 20,
+ "metadata": {},
+ "outputs": [
+ {
+ "name": "stdout",
+ "output_type": "stream",
+ "text": [
+ "Gathering DNSKEY records.\n",
+ "Done\n",
+ "Gathering DS records.\n",
+ "Done.\n",
+ "Gathering DNSKEY and DS records for special TLDs.\n",
+ "Done\n"
+ ]
+ }
+ ],
"source": [
- "df = pd.DataFrame(data_ips)"
+ "print('Gathering DNSKEY records.')\n",
+ "!dig +noall +answer +noidn -t DNSKEY -f data/lists/tlds > data/dig/tld_dnskeys\n",
+ "print('Done')\n",
+ "\n",
+ "print('Gathering DS records.')\n",
+ "!dig +noall +answer +noidn -t DS -f data/lists/tlds > data/dig/tld_dss\n",
+ "print('Done.')\n",
+ "\n",
+ "print('Gathering DNSKEY and DS records for special TLDs.')\n",
+ "for tld in special_tlds:\n",
+ " !dig +noall +answer +noidn -t DNSKEY {tld} >> data/dig/tld_dnskeys\n",
+ " !dig +noall +answer +noidn -t DS {tld} >> data/dig/tld_dss\n",
+ "print('Done')"
]
},
{
@@ -706,29 +804,24 @@
},
"outputs": [],
"source": [
- "xtlds = []\n",
- "xnss = []\n",
- "xips = []\n",
- "xasns = []\n",
- "xtcp = []\n",
- "xudp = []\n",
+ "# Parses the DNSKEY records.\n",
"\n",
- "for datum in data_ips:\n",
- " xtlds.append(datum['tld'])\n",
- " xnss.append(datum['ns'])\n",
- " xips.append(datum['ip'])\n",
- " xtcp.append(datum['tcp'])\n",
- " xudp.append(datum['udp'])\n",
- " if 'asn' in datum:\n",
- " xasns.append(datum['asn'])\n",
- " else:\n",
- " xasns.append([])\n",
- " \n",
- "print(len(data_ips), len(xtlds), len(xnss), len(xips))\n",
+ "data_cred = [{'tld': tld, 'ds': False, 'dnskey': False, 'jsj': None} for tld in tlds]\n",
+ "\n",
+ "for answer in read_list('data/dig/tld_dnskeys'):\n",
+ " answer_fields = answer.split()\n",
+ " tld = answer_fields[0][:-1].upper()\n",
+ " index = find(data_cred, 'tld', tld)\n",
" \n",
- "ix = pd.MultiIndex.from_arrays([xtlds, xnss, xips], names=['tld', 'ns', 'ip'])\n",
- "dg = pd.DataFrame({'asn': xasns, 'tcp': xtcp, 'udp': xudp}, index = ix)\n",
- "# dg.head(10)"
+ " try:\n",
+ " data_cred[index]['dnskey'] = True\n",
+ " \n",
+ " if int(answer_fields[4]) == KSK:\n",
+ " data_cred[index]['ksk'] = answer_fields[6]\n",
+ " elif int(answer_fields[4]) == ZSK:\n",
+ " data_cred[index]['zsk'] = answer_fields[6]\n",
+ " except:\n",
+ " print(tld)"
]
},
{
@@ -739,45 +832,18 @@
},
"outputs": [],
"source": [
- "# dage = {}\n",
+ "# Parses the DS records.\n",
"\n",
- "# for datum in data_age:\n",
- "# dage[datum['tld']] = datum['age']\n",
- " \n",
- "# for datum in data_ips:\n",
- "# if datum['tld'][:-1].upper() in dage:\n",
- "# if dage[datum['tld'][:-1].upper()] == 'new':\n",
- "# datum['age'] = 'new'\n",
- "# else:\n",
- "# datum['age'] = 'old'\n",
- "# else:\n",
- "# datum['age'] = None\n",
- "\n",
- "# xtlds = []\n",
- "# xnss = []\n",
- "# xips = []\n",
- "# xasns = []\n",
- "# xtcp = []\n",
- "# xudp = []\n",
- "\n",
- "# for datum in data_ips:\n",
- "# if datum['age'] == 'old':\n",
+ "for answer in read_list('data/dig/tld_dss'):\n",
+ " v = answer.split()\n",
+ " tld = v[0][:-1].upper()\n",
" \n",
- "# xtlds.append(datum['tld'])\n",
- "# xnss.append(datum['ns'])\n",
- "# xips.append(datum['ip'])\n",
- "# xtcp.append(datum['tcp'])\n",
- "# xudp.append(datum['udp'])\n",
- "# if 'asn' in datum:\n",
- "# xasns.append(datum['asn'])\n",
- "# else:\n",
- "# xasns.append([])\n",
- " \n",
- "# print(len(data_ips), len(xtlds), len(xnss), len(xips))\n",
+ " index = find(data_cred, 'tld', tld)\n",
" \n",
- "# ix = pd.MultiIndex.from_arrays([xtlds, xnss, xips], names=['tld', 'ns', 'ip'])\n",
- "# dg = pd.DataFrame({'asn': xasns, 'tcp': xtcp, 'udp': xudp}, index = ix)\n",
- "# # dg.head(10)"
+ " try:\n",
+ " data_cred[index]['ds'] = True\n",
+ " except:\n",
+ " print(tld)"
]
},
{
@@ -788,235 +854,123 @@
},
"outputs": [],
"source": [
- "# data_ips[0]"
+ "write_data('data_cred', data_cred)"
]
},
{
- "cell_type": "code",
- "execution_count": null,
- "metadata": {
- "collapsed": true
- },
- "outputs": [],
+ "cell_type": "markdown",
+ "metadata": {},
"source": [
- "ns_tcp_udp = [{'ns': datum['ns'], 'tcp': False, 'udp': False} for datum in data_ips if datum['tld'] in dtype and dtype[datum['tld']] == 'country-code']\n",
- "\n",
- "for ns in ns_tcp_udp:\n",
- " for datum in data_ips:\n",
- " if datum['ns'] == ns['ns']:\n",
- " if datum['tcp']:\n",
- " ns['tcp'] = True\n",
- " \n",
- " if datum['udp']:\n",
- " ns['udp'] = True "
+ "# Performance"
]
},
{
"cell_type": "code",
- "execution_count": null,
- "metadata": {
- "collapsed": true
- },
- "outputs": [],
+ "execution_count": 22,
+ "metadata": {},
+ "outputs": [
+ {
+ "name": "stderr",
+ "output_type": "stream",
+ "text": [
+ "100%|██████████| 1547/1547 [01:51<00:00, 13.85it/s]\n"
+ ]
+ }
+ ],
"source": [
- "ns_tcp_udp = [{'ns': datum['ns'], 'tcp': False, 'udp': False} for datum in data_ips]\n",
+ "# Generates domains that result in a NXDOMAIN response for each TLD.\n",
+ "data_test_perf = []\n",
+ "pbar = tqdm(total=len(tlds))\n",
"\n",
- "for ns in ns_tcp_udp:\n",
- " for datum in data_ips:\n",
- " if datum['ns'] == ns['ns']:\n",
- " if datum['tcp']:\n",
- " ns['tcp'] = True\n",
- " \n",
- " if datum['udp']:\n",
- " ns['udp'] = True "
+ "for tld in tlds:\n",
+ " data_test_perf.append({'tld': tld, 'domain': find_nxdomain(tld)})\n",
+ " \n",
+ " pbar.update(1)\n",
+ "pbar.close()"
]
},
{
"cell_type": "code",
- "execution_count": null,
- "metadata": {
- "collapsed": true
- },
- "outputs": [],
+ "execution_count": 23,
+ "metadata": {},
+ "outputs": [
+ {
+ "name": "stderr",
+ "output_type": "stream",
+ "text": [
+ "100%|██████████| 38/38 [00:05<00:00, 3.85it/s]\n"
+ ]
+ }
+ ],
"source": [
- "ip = ':'\n",
- "\n",
- "ns_tcp_udp = [{'ns': datum['ns'], 'tcp': False, 'udp': False} for datum in data_ips if ip in datum['ip']]\n",
+ "# Generates domains that result in a NXDOMAIN response for each TLD that uses wildcards.\n",
+ "pbar = tqdm(total=len([datum for datum in data_test_perf if not datum['domain']]))\n",
"\n",
- "for ns in ns_tcp_udp:\n",
- " for datum in data_ips:\n",
- " if datum['ns'] == ns['ns'] and ip in datum['ip']:\n",
- " if datum['tcp']:\n",
- " ns['tcp'] = True\n",
- " \n",
- " if datum['udp']:\n",
- " ns['udp'] = True"
- ]
- },
- {
- "cell_type": "code",
- "execution_count": null,
- "metadata": {
- "collapsed": true
- },
- "outputs": [],
- "source": [
- "dg = pd.DataFrame(ns_tcp_udp)"
- ]
- },
- {
- "cell_type": "code",
- "execution_count": null,
- "metadata": {
- "collapsed": true
- },
- "outputs": [],
- "source": [
- "ax = dg.tcp.value_counts().plot.pie(autopct=lambda p : '{:.2f}% ({:.0f})'.format(p, p * dg.count().udp / 100), figsize = pie_size)\n",
- "# ax = dg.tcp.plot.bar()\n",
- "ax.set_ylabel('')\n",
- "fig = ax.get_figure()\n",
- "fig.savefig(\"imgs/tcp.pdf\")"
- ]
- },
- {
- "cell_type": "code",
- "execution_count": null,
- "metadata": {
- "collapsed": true
- },
- "outputs": [],
- "source": [
- "ax = dg.udp.value_counts().plot.pie(autopct=lambda p : '{:.2f}% ({:.0f})'.format(p, p * dg.count().udp / 100), \n",
- " figsize = pie_size)\n",
- "ax.set_ylabel('')\n",
- "# ax.set_title('UDP')\n",
- "fig = ax.get_figure()\n",
- "fig.savefig(\"imgs/udp.pdf\")"
- ]
- },
- {
- "cell_type": "code",
- "execution_count": null,
- "metadata": {
- "collapsed": true
- },
- "outputs": [],
- "source": [
- "df_ftcp = dg.loc[dg.tcp == False]\n",
- "df_fudp = dg.loc[dg.udp == False]\n",
- "df_ttcp = dg.loc[dg.tcp == True]\n",
- "\n",
- "ff = df_ftcp.loc[df_ftcp.udp == False].count().tcp\n",
- "ft = df_ftcp.loc[df_ftcp.udp == True].count().tcp\n",
- "tf = df_fudp.loc[df_fudp.tcp == True].count().tcp\n",
- "tt = df_ttcp.loc[df_ttcp.udp == True].count().tcp\n",
- "\n",
- "print(ff, ft, tf, tt)\n",
- "\n",
- "ut_data = [{'name': 'none', 'count': ff}, \n",
- " {'name': 'tcp', 'count': tf}, \n",
- " {'name': 'udp', 'count': ft}, \n",
- " {'name': 'tcp + udp', 'count': tt}\n",
- " ]\n",
- "\n",
- "total = ff + ft + tf + tt\n",
- "\n",
- "dfgh = pd.DataFrame(ut_data)\n",
- "dfgh.index = dfgh['name']\n",
- "del dfgh['name']\n",
- "# ax = dfgh.plot.pie('count',\n",
- "# # autopct='s(%.2f)',\n",
- "# autopct=lambda p : '{:.2f}% ({:.0f})'.format(p, p * total / 100),\n",
- "# # radius = 2.5,\n",
- "# # pctdistance=1.2,\n",
- "# # labeldistance=1.2,\n",
- "# # explode = True,\n",
- "# figsize = pie_size, \n",
- "# legend=False, \n",
- "# labels=['','','',''])\n",
- "\n",
- "ax = dfgh.plot.barh()\n",
- "# ax.set_xlim([0,10000])\n",
- "\n",
- "ax.legend(loc='best', labels=dfgh.index)\n",
- "ax.set_xlabel('Number of name servers')\n",
- "ax.set_ylabel('Protocol(s) supported')\n",
- "# ax.set_title('name server udp/tcp support')\n",
- "ax.legend_.remove()\n",
- "ax.set_xscale('log')\n",
- "fig = ax.get_figure()\n",
- "fig.tight_layout()\n",
- "fig.savefig(\"imgs/tcp_udp_generic.pdf\")"
- ]
- },
- {
- "cell_type": "code",
- "execution_count": null,
- "metadata": {
- "collapsed": true
- },
- "outputs": [],
- "source": [
- "dfgh.index"
- ]
- },
- {
- "cell_type": "code",
- "execution_count": null,
- "metadata": {
- "collapsed": true
- },
- "outputs": [],
- "source": [
- "df.loc[df.tld == 'actor.']"
- ]
- },
- {
- "cell_type": "code",
- "execution_count": null,
- "metadata": {
- "collapsed": true
- },
- "outputs": [],
- "source": [
- "# def write_to_file(fn, indir, content):\n",
- "# with open(indir + fn, 'w') as f:\n",
- "# f.write(content)"
- ]
- },
- {
- "cell_type": "code",
- "execution_count": null,
- "metadata": {
- "collapsed": true
- },
- "outputs": [],
- "source": [
- "# write_to_file('tcp_udp_not_working', 'data/temp/', df_ftcp.loc[df_ftcp.udp == False].to_csv())"
+ "for datum in [datum for datum in data_test_perf if not datum['domain']]:\n",
+ " datum['domain'] = find_nxdomain_wildcard(datum['tld'])\n",
+ " \n",
+ " pbar.update(1)\n",
+ "pbar.close()"
]
},
{
"cell_type": "code",
- "execution_count": null,
+ "execution_count": 24,
"metadata": {
"collapsed": true
},
"outputs": [],
"source": [
- "# df_ftcp.loc[df_ftcp.udp == False]"
+ "# Set the probes.\n",
+ "probe_ids = [10262, 10287, 11040, 11429, 12515, 12873, 12956, 13623, 13728, 13769, 13788, 13799, 13804, \n",
+ " 13805, 13810, 14237, 26057, 14564, 15156, 14691, 15594, 15799, 4205, 18131, 18195, 18691, \n",
+ " 19326, 19740, 20111, 20353, 20493, 20531, 20621, 21003, 21035, 21122, 21251, 21345, 21703, \n",
+ " 22286, 22695, 23031, 23085, 28240, 27972, 23697, 24807, 25011, 25148, 25323, 26936, 26378, \n",
+ " 26627, 4155, 26823, 28355, 30676, 4829, 29006, 29183, 29405, 30225, 30324, 31201, 19306, \n",
+ " 19634, 6025, 11660, 22388, 25182, 4123, 3812, 20923, 14384, 12389]\n",
+ "\n",
+ "probes = [\n",
+ " {\n",
+ " \"value\": str(probe_ids)[1:-1],\n",
+ " \"type\": \"probes\",\n",
+ " \"requested\": len(probe_ids)\n",
+ " }\n",
+ "]"
]
},
{
"cell_type": "code",
- "execution_count": null,
- "metadata": {
- "collapsed": true
- },
+ "execution_count": 28,
+ "metadata": {},
"outputs": [],
"source": [
- "# df.loc[df.ip.str.contains(':')].count().tcp #ipv6\n",
- "# df[~df[\"ip\"].str.contains(\":\")].count().tcp #ipv4"
+ "payloads = []\n",
+ "payload_size = 100 # max 100\n",
+ "step_size = int(payload_size / 2)\n",
+ "\n",
+ "for i in range(0, len(data_test_perf), step_size):\n",
+ " defintions = []\n",
+ " \n",
+ " for datum in data_test_perf[i:i + step_size]:\n",
+ " # Create caching measurement\n",
+ " definition_caching = definition.copy()\n",
+ " definition_caching['query_type'] = \"NS\"\n",
+ " definition_caching['query_argument'] = datum['tld']\n",
+ " definition_caching['description'] = \"caching \" + datum['tld']\n",
+ " defintions.append(definition_caching)\n",
+ " \n",
+ " # Create response time measurement\n",
+ " definition_measuring = definition.copy()\n",
+ " definition_measuring['query_type'] = \"SOA\"\n",
+ " definition_measuring['query_argument'] = datum['domain']\n",
+ " definition_measuring['description'] = \"measuring \" + datum['tld']\n",
+ " defintions.append(definition_measuring) \n",
+ "\n",
+ " new_payload = payload.copy()\n",
+ " new_payload['probes'] = probes\n",
+ " new_payload['definitions'] = defintions\n",
+ "\n",
+ " payloads.append(new_payload)"
]
},
{
@@ -1027,10 +981,7 @@
},
"outputs": [],
"source": [
- "# ax = df[~df[\"ip\"].str.contains(\":\")].tcp.value_counts().plot.pie(autopct='%.2f', figsize = pie_size)\n",
- "# ax.set_ylabel('')\n",
- "# fig = ax.get_figure()\n",
- "# fig.savefig(\"imgs/tcp_ipv4.pdf\")"
+ "write_data('payloads', payloads)"
]
},
{
@@ -1041,10 +992,9 @@
},
"outputs": [],
"source": [
- "# ax = df[~df[\"ip\"].str.contains(\":\")].udp.value_counts().plot.pie(autopct='%.2f', figsize = pie_size)\n",
- "# ax.set_ylabel('')\n",
- "# fig = ax.get_figure()\n",
- "# fig.savefig(\"imgs/udp_ipv4.pdf\")"
+ "measurement_ids = []\n",
+ "measurement_responses = []\n",
+ "url = URL_DNS_MEASUREMENT_CREATE + '?key=' + ATLAS_API_KEY"
]
},
{
@@ -1055,10 +1005,21 @@
},
"outputs": [],
"source": [
- "# ax = df.loc[df.ip.str.contains(':')].tcp.value_counts().plot.pie(autopct='%.2f', figsize = pie_size)\n",
- "# ax.set_ylabel('')\n",
- "# fig = ax.get_figure()\n",
- "# fig.savefig(\"imgs/tcp_ipv6.pdf\")"
+ "# Only 10 payloads can be sent per day.\n",
+ "# Manually set this slice each day.\n",
+ "for payload in payloads[:10]:\n",
+ " \n",
+ " request = requests.post(url, data = json.dumps(payload), headers = HEADERS)\n",
+ " print(request.status_code)\n",
+ " \n",
+ " while request.status_code == 400:\n",
+ " print(request.json())\n",
+ " request = requests.post(url, data = json.dumps(payload), headers = HEADERS)\n",
+ " time.sleep(300)\n",
+ " print(request.status_code)\n",
+ " \n",
+ " measurement_ids += measurement_ids + request.json()\n",
+ " measurement_ids = [min(measurement_ids), max(measurement_ids)]"
]
},
{
@@ -1069,10 +1030,40 @@
},
"outputs": [],
"source": [
- "# ax = df.loc[df.ip.str.contains(':')].udp.value_counts().plot.pie(autopct='%.2f', figsize = pie_size)\n",
- "# ax.set_ylabel('')\n",
- "# fig = ax.get_figure()\n",
- "# fig.savefig(\"imgs/udp_ipv6.pdf\")"
+ "# Retrieves the result of each measurement and writes it to a file.\n",
+ "\n",
+ "next_result = True\n",
+ "url = '{}?id__gte={}&id__lte={}&mine=true'.format(URL_DNS_MEASUREMENT_GET, min(measurement_ids), max(measurement_ids))\n",
+ "measurements = requests.get(url).json()\n",
+ "pbar = tqdm(total=len(tlds))\n",
+ "\n",
+ "while next_result:\n",
+ " for result in measurements['results']:\n",
+ " if len(result['description'].split()) == 2:\n",
+ " measurement_type, tld = result['description'].split()\n",
+ " request = requests.get(result['result']).json()\n",
+ "\n",
+ " if measurement_type == 'measuring':\n",
+ " with open('data/atlas/soa/{}.json'.format(tld.upper()), 'w') as f:\n",
+ " temp = copy.deepcopy(result)\n",
+ " temp['result'] = request\n",
+ "\n",
+ " f.write(json.dumps(temp))\n",
+ "\n",
+ " elif measurement_type == 'caching':\n",
+ " with open('data/atlas/ns/{}.json'.format(tld.upper()), 'w') as f:\n",
+ " temp = copy.deepcopy(result)\n",
+ " temp['result'] = request\n",
+ "\n",
+ " f.write(json.dumps(temp))\n",
+ " \n",
+ " if measurements['next']:\n",
+ " measurements = requests.get(measurements['next']).json() \n",
+ " else:\n",
+ " next_result = False\n",
+ "\n",
+ " pbar.update(1)\n",
+ "pbar.close()"
]
},
{
@@ -1083,109 +1074,48 @@
},
"outputs": [],
"source": [
- "df.head()"
- ]
- },
- {
- "cell_type": "markdown",
- "metadata": {},
- "source": [
- "# Credibility"
- ]
- },
- {
- "cell_type": "code",
- "execution_count": 201,
- "metadata": {
- "collapsed": true
- },
- "outputs": [
- {
- "name": "stderr",
- "output_type": "stream",
- "text": [
- "\n",
- "\n",
- "\n",
- "\n",
- "\n",
- "\n",
- "\n",
- "\n",
- "\n",
- "\n",
- "\n",
- "\n",
- "\n",
- "\n"
- ]
- }
- ],
- "source": [
- "!dig +noall +answer +noidn -t DNSKEY -f data/lists/tlds > data/dig/tld_dnskeys\n",
- "!dig +noall +answer +noidn -t DS -f data/lists/tlds > data/dig/tld_dss\n",
+ "# Extracts the response time from the measurement results.\n",
"\n",
- "for tld in special_tlds:\n",
- " !dig +noall +answer +noidn -t DNSKEY {tld} >> data/dig/tld_dnskeys\n",
- " !dig +noall +answer +noidn -t DS {tld} >> data/dig/tld_dss"
- ]
- },
- {
- "cell_type": "code",
- "execution_count": 235,
- "metadata": {},
- "outputs": [],
- "source": [
- "data_cred = [{'tld': tld, 'ds': False, 'dnskey': False, 'algorithm': None} for tld in tlds]\n",
+ "indir = 'data/atlas/soa/'\n",
"\n",
- "temp = []\n",
+ "data_perf = []\n",
"\n",
- "for answer in read_list('data/dig/tld_dnskeys'):\n",
- " v = answer.split()\n",
- " tld = v[0][:-1].upper()\n",
- " index = find(data_cred, 'tld', tld)\n",
- " \n",
- " try:\n",
- " data_cred[index]['dnskey'] = True\n",
- " data_cred[index]['algorithm'] = v[6]\n",
- " except:\n",
- " print(tld)"
- ]
- },
- {
- "cell_type": "code",
- "execution_count": 236,
- "metadata": {},
- "outputs": [],
- "source": [
- "for answer in read_list('data/dig/tld_dss'):\n",
- " v = answer.split()\n",
- " tld = v[0][:-1].upper()\n",
- " \n",
- " index = find(data_cred, 'tld', tld)\n",
- " \n",
- " try:\n",
- " data_cred[index]['ds'] = True\n",
- " except:\n",
- " print(tld)"
+ "for root, dirs, filenames in os.walk(indir):\n",
+ " for f in filenames:\n",
+ " tld, _ = f.split('.')\n",
+ " \n",
+ " datum = {'tld': tld, 'rt': [], 'timeouts': 0}\n",
+ " \n",
+ " with open(indir + f, 'r') as f:\n",
+ " tld_results = json.loads(f.read())\n",
+ " \n",
+ " for probe in tld_results['result']:\n",
+ " for result in probe['resultset']:\n",
+ " if 'result' in result:\n",
+ " datum['rt'].append(result['result']['rt']) \n",
+ " elif 'error' in result and 'timeout' in result['error']:\n",
+ " datum['timeouts'] += 1\n",
+ " \n",
+ " datum['rt'] = np.mean(datum['rt'])\n",
+ " data_perf.append(datum)"
]
},
{
"cell_type": "code",
- "execution_count": 237,
+ "execution_count": null,
"metadata": {
"collapsed": true
},
"outputs": [],
"source": [
- "write_data('data_cred', data_cred)"
+ "write_data('data_perf', data_perf)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
- "# Organisations per TLD"
+ "# Anycast"
]
},
{
@@ -1195,17 +1125,13 @@
"collapsed": true
},
"outputs": [],
- "source": []
- },
- {
- "cell_type": "code",
- "execution_count": null,
- "metadata": {
- "collapsed": true
- },
- "outputs": [],
"source": [
- "# tld_orgs"
+ "test_set = [\n",
+ " {'target': '185.49.140.60', 'argument': 'nlnetlabs.nl', 'ns': 'ns.nlnetlabs.nl'},\n",
+ " {'target': '192.16.197.229', 'argument': 'nlnet.nl', 'ns': 'mcvax.nlnet.nl'},\n",
+ " {'target': '194.0.28.53', 'argument': 'nl', 'ns': 'ns5.dns.nl'},\n",
+ " {'target': '204.61.216.4', 'argument': 'nlnetlabs.nl', 'ns': 'anyns.pch.net'}\n",
+ "]"
]
},
{
@@ -1216,10 +1142,21 @@
},
"outputs": [],
"source": [
- "df_orgs = pd.DataFrame(tld_orgs)\n",
- "df_orgs.index = df_orgs['tld']\n",
- "del df_orgs['tld']\n",
- "df_orgs.head()"
+ "definitions = []\n",
+ "n = 3 # number of measurements\n",
+ "\n",
+ "for test in test_set:\n",
+ " for i in range(1, n + 1): \n",
+ " new_definition = copy.deepcopy(definition)\n",
+ " new_definition['type'] = 'SOA'\n",
+ " new_definition['query_argument'] = test['argument']\n",
+ " new_definition['use_probe_resolver'] = False\n",
+ " new_definition['target'] = test['target']\n",
+ " new_definition['description'] = 'anycast {} {}'.format(i, test['ns'])\n",
+ "\n",
+ " definitions.append(new_definition)\n",
+ "\n",
+ "payload['definitions'] = definitions"
]
},
{
@@ -1230,13 +1167,44 @@
},
"outputs": [],
"source": [
- "df_orgs.reset_index(inplace=True)\n",
- "rows = []\n",
- "_ = df_orgs.apply(lambda row: [rows.append([row['tld'], nn]) \n",
- " for nn in row.organisations], axis=1)\n",
- "df_orgs_new = pd.DataFrame(rows, columns=df_orgs.columns).set_index(['tld'])\n",
+ "anycast_probes = [\n",
+ " # North America\n",
+ " {'id': 22447, 'country-code': 'US', 'city': 'San Francisco'},\n",
+ " {'id': 14233, 'country-code': 'US', 'city': 'Denver'},\n",
+ " {'id': 25081, 'country-code': 'US', 'city': 'Washington'},\n",
+ " # South America\n",
+ " {'id': 31450, 'country-code': 'CR', 'city': 'San Jose'},\n",
+ " {'id': 30185, 'country-code': 'BR', 'city': 'Sao Paulo'},\n",
+ " {'id': 30123, 'country-code': 'CL', 'city': 'Santiago'},\n",
+ " # Europe\n",
+ " {'id': 32669, 'country-code': 'GR', 'city': 'Athens'},\n",
+ " {'id': 31479, 'country-code': 'RU', 'city': 'Moscow'},\n",
+ " {'id': 29762, 'country-code': 'ES', 'city': 'Madrid'},\n",
+ " {'id': 26610, 'country-code': 'NL', 'city': 'Utrecht'},\n",
+ " # Africa\n",
+ " {'id': 22458, 'country-code': 'ZA', 'city': 'Cape Town'},\n",
+ " {'id': 13258, 'country-code': 'AE', 'city': 'Dubai'},\n",
+ " {'id': 28493, 'country-code': 'SN', 'city': 'Dakar'},\n",
+ " # Asia\n",
+ " {'id': 28819, 'country-code': 'JP', 'city': 'Tokyo'},\n",
+ " {'id': 28964, 'country-code': 'KR', 'city': 'Seoul'},\n",
+ " {'id': 6107, 'country-code': 'IN', 'city': 'Mumbai'},\n",
+ " {'id': 25047, 'country-code': 'HK', 'city': 'Hong Kong'},\n",
+ " {'id': 26378, 'country-code': 'KG', 'city': 'Bishkek'},\n",
+ " # Oceania\n",
+ " {'id': 25208, 'country-code': 'AU', 'city': 'Sydney'},\n",
+ " {'id': 28226, 'country-code': 'NZ', 'city': 'Welington'}\n",
+ "]\n",
+ "\n",
+ "probes = [\n",
+ " {\n",
+ " \"value\": str([probe['id'] for probe in anycast_probes])[1:-1],\n",
+ " \"type\": \"probes\",\n",
+ " \"requested\": len(anycast_probes)\n",
+ " }\n",
+ "]\n",
"\n",
- "df_orgs_new.head()"
+ "payload['probes'] = probes"
]
},
{
@@ -1247,7 +1215,8 @@
},
"outputs": [],
"source": [
- "df_orgs_new.organisations.value_counts(ascending=False).head(80).plot.barh(figsize = (10,20))"
+ "url = '{}?key={}'.format(URL_DNS_MEASUREMENT_CREATE, atlas_api_key)\n",
+ "request = requests.post(url, data = json.dumps(payload), headers = HEADERS)"
]
},
{
@@ -1258,9 +1227,7 @@
},
"outputs": [],
"source": [
- "bins = df_orgs_new.organisations.value_counts().nunique() - 1\n",
- "ax = df_orgs_new.organisations.value_counts().hist(bins = bins)\n",
- "ax.set_yscale('log')"
+ "measurement_ids = request.json()"
]
},
{
@@ -1271,7 +1238,20 @@
},
"outputs": [],
"source": [
- "df_orgs_new.organisations.value_counts().value_counts().plot.pie()"
+ "# Only 10 payloads can be sent per day.\n",
+ "# Manually set this slice each day.\n",
+ "for payload in payloads[:10]:\n",
+ " \n",
+ " request = requests.post(url, data = json.dumps(payload), headers = HEADERS)\n",
+ " print(request.status_code)\n",
+ " \n",
+ " while request.status_code == 400:\n",
+ " print(request.json())\n",
+ " request = requests.post(url, data = json.dumps(payload), headers = HEADERS)\n",
+ " time.sleep(300)\n",
+ " print(request.status_code)\n",
+ " \n",
+ " measurement_ids += measurement_ids + request.json()"
]
},
{
@@ -1282,45 +1262,49 @@
},
"outputs": [],
"source": [
- "# df_orgs_new.organisations.value_counts()\n",
+ "next_result = True\n",
+ "url = '{}?id__in={}&mine=true'.format(URL_DNS_MEASUREMENT_GET, str(measurement_ids)[1:-1])\n",
+ "measurements = requests.get(url).json()\n",
"\n",
- "# df2[df2['rr_quality'] > 0]].groupby([df2.index.hour,'sleep_summary_id')"
- ]
- },
- {
- "cell_type": "code",
- "execution_count": null,
- "metadata": {
- "collapsed": true
- },
- "outputs": [],
- "source": [
- "ax = df_tld_orgs.organisation.value_counts().head(20).plot.barh(figsize = bar_size, fontsize=12)\n",
- "ax.set_xlabel('Number of TLDs', fontsize = 16)\n",
- "ax.set_ylabel('Organisation',fontsize = 16)\n",
- "fig = ax.get_figure()\n",
- "fig.savefig(\"imgs/orgs.png\")"
+ "while next_result:\n",
+ " for result in measurements['results']:\n",
+ " if len(result['description'].split()) == 3:\n",
+ " measurement_type, i, ns = result['description'].split()\n",
+ " request = requests.get(result['result']).json()\n",
+ "\n",
+ " if measurement_type == 'anycast':\n",
+ " temp = copy.deepcopy(result)\n",
+ " temp['result'] = request\n",
+ " \n",
+ " write_json('data/atlas/anycast/{}.{}.json'.format(ns, i), temp)\n",
+ " \n",
+ " if measurements['next']:\n",
+ " measurements = requests.get(measurements['next']).json() \n",
+ " else:\n",
+ " next_result = False"
]
},
{
"cell_type": "code",
"execution_count": null,
- "metadata": {
- "collapsed": true
- },
- "outputs": [],
- "source": [
- "ax = df_tld_orgs.type.value_counts().plot.pie(figsize = pie_size, legend=True)\n",
- "ax.set_ylabel('')\n",
- "fig = ax.get_figure()\n",
- "fig.savefig(\"imgs/types.png\")"
- ]
- },
- {
- "cell_type": "markdown",
"metadata": {},
+ "outputs": [],
"source": [
- "# Growth"
+ "indir = 'data/atlas/anycast/'\n",
+ "probe_data = {k:v for k, v in [(probe_id, []) for probe_id in [probe['id'] for probe in anycast_probes]]}\n",
+ "data_ac = [{'ns': test['ns'], 'probes': copy.deepcopy(probe_data)} for test in test_set]\n",
+ "\n",
+ "for root, dirs, filenames in os.walk(indir):\n",
+ " for f in filenames:\n",
+ " result = read_json(indir + f)\n",
+ " _, _, ns = result['description'].split()\n",
+ " index = find(data_ac, 'ns', ns)\n",
+ " \n",
+ " for probe in result['result']:\n",
+ " try:\n",
+ " data_ac[index]['probes'][probe['prb_id']].append(probe['result']['rt'])\n",
+ " except:\n",
+ " print('x')"
]
},
{
@@ -1331,17 +1315,7 @@
},
"outputs": [],
"source": [
- "data_age = []\n",
- "dage = {}\n",
- "\n",
- "for datum in tld_creation:\n",
- " y, m, d = datum['date_created'].split('-')\n",
- " if y in ['2014', '2015', '2016', '2017'] or y == '2013' and int(m) >= 10:\n",
- " data_age.append({'tld': datum['tld'], 'age': 'new'})\n",
- " dage[datum['tld']] = 'new'\n",
- " else:\n",
- " data_age.append({'tld': datum['tld'], 'age': 'old'})\n",
- " dage[datum['tld']] = 'old'"
+ "write_data('data_ac', data_ac)"
]
}
],