diff options
| author | bicker <bickerkards@gmail.com> | 2017-07-14 01:23:11 +0200 |
|---|---|---|
| committer | bicker <bickerkards@gmail.com> | 2017-07-14 01:23:11 +0200 |
| commit | 092af5e8fefe1dc0f7e5d02bb3050d05e22dd475 (patch) | |
| tree | ada870c384ac8ac559c74e697fddfbf5c0bc412a /data_gathering.ipynb | |
| parent | 9a3b2702bbcbd67eab592b043b7dcd50b48afe3b (diff) | |
Diffstat (limited to 'data_gathering.ipynb')
| -rw-r--r-- | data_gathering.ipynb | 1184 |
1 files changed, 579 insertions, 605 deletions
diff --git a/data_gathering.ipynb b/data_gathering.ipynb index 9070f80..cdbaf61 100644 --- a/data_gathering.ipynb +++ b/data_gathering.ipynb @@ -9,8 +9,21 @@ }, "outputs": [], "source": [ - "# Constantijn Bicker Caarten\n", - "# Last updated: 13-06-2017" + "# Written by Constantijn Bicker Caarten\n", + "# Last updated: 07-07-2017\n", + "#\n", + "#\n", + "# This code gathers data on TLDs in the DNS. This data \n", + "# consists out of A, AAAA, DNSKEY, DS and NS records, \n", + "# as well as TCP and UDP support, response time and\n", + "# anycast support." + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "# Imports" ] }, { @@ -21,6 +34,7 @@ }, "outputs": [], "source": [ + "from dns.resolver import Resolver, NXDOMAIN, query\n", "from socket import error as socket_error\n", "from urllib.request import urlopen\n", "from dns.query import udp, tcp\n", @@ -29,50 +43,28 @@ "from uuid import uuid4\n", "from tqdm import tqdm\n", "\n", + "import matplotlib as plt\n", "import pandas as pd\n", "import numpy as np\n", - "import matplotlib as plt\n", "\n", - "import subprocess\n", + "import datetime\n", + "import requests\n", "import socket\n", "import copy\n", - "import time\n", "import json\n", - "import os\n", - "\n", - "%matplotlib inline" + "import os" ] }, { - "cell_type": "code", - "execution_count": 76, + "cell_type": "markdown", "metadata": {}, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "mkdir: cannot create directory ‘data’: File exists\n", - "mkdir: cannot create directory ‘data/whois’: File exists\n", - "mkdir: cannot create directory ‘data/dig’: File exists\n", - "mkdir: cannot create directory ‘data/backup’: File exists\n", - "mkdir: cannot create directory ‘data/lists’: File exists\n" - ] - } - ], "source": [ - "# First time run\n", - "!mkdir data\n", - "!mkdir data/whois\n", - "!mkdir data/cymru\n", - "!mkdir data/dig\n", - "!mkdir data/backup\n", - "!mkdir data/lists" + "# Constants" ] }, { "cell_type": "code", - "execution_count": 199, + "execution_count": 3, "metadata": { "collapsed": true }, @@ -81,9 +73,54 @@ "ZSK = 256\n", "KSK = 257\n", "\n", - "pie = (6, 6)\n", + "ATLAS_API_KEY = '' # Add your Atlas API key\n", + "ATLAS_BILL_TO = '' # Add your Atlas account email\n", + "\n", + "URL_DNS_MEASUREMENT_CREATE = 'https://atlas.ripe.net:443/api/v2/measurements/dns/'\n", + "URL_DNS_MEASUREMENT_GET = 'https://atlas.ripe.net:443/api/v2/measurements/dns/'\n", + "\n", + "HEADERS = {'Content-type': 'application/json', 'Accept': 'text/plain'}\n", "\n", - "newline = '\\n'" + "NEWLINE = '\\n'" + ] + }, + { + "cell_type": "code", + "execution_count": 4, + "metadata": { + "collapsed": true + }, + "outputs": [], + "source": [ + "payload = {\n", + " 'bill_to': ATLAS_BILL_TO,\n", + " 'is_oneoff': True,\n", + " 'definitions': [],\n", + " 'probes': []\n", + "}\n", + " \n", + "definition = {\n", + " 'af': 4,\n", + " 'query_class': 'IN',\n", + " 'query_type': '',\n", + " 'query_argument': '',\n", + " 'description': '',\n", + " 'use_probe_resolver': True,\n", + " 'resolve_on_probe': False,\n", + " 'set_nsid_bit': True,\n", + " 'protocol': 'UDP',\n", + " 'udp_payload_size': 512,\n", + " 'retry': 0,\n", + " 'skip_dns_check': False,\n", + " 'include_qbuf': False,\n", + " 'include_abuf': True,\n", + " 'prepend_probe_id': False,\n", + " 'set_rd_bit': False,\n", + " 'set_do_bit': False,\n", + " 'set_cd_bit': False,\n", + " 'type': 'dns',\n", + " 'is_public': True\n", + "}" ] }, { @@ -95,7 +132,7 @@ }, { "cell_type": "code", - "execution_count": 157, + "execution_count": 5, "metadata": { "collapsed": true, "scrolled": false @@ -106,18 +143,18 @@ " '''Writes a list to a file with each value on a new line'''\n", " with open(fn, 'w') as f:\n", " for datum in data:\n", - " f.write(datum + newline)\n", + " f.write(datum + NEWLINE)\n", " \n", "def append_list(fn, data):\n", " '''Appends a list to a file with each value on a new line'''\n", " with open(fn, 'a') as f:\n", " for datum in data:\n", - " f.write(datum + newline)\n", + " f.write(datum + NEWLINE)\n", " \n", "def read_list(fn):\n", " '''Reads a file and '''\n", " with open(fn, 'r') as f:\n", - " return [line.strip(newline) for line in f]\n", + " return [line.strip(NEWLINE) for line in f]\n", " \n", "def write_json(fn, data):\n", " with open(fn, 'w') as f:\n", @@ -126,20 +163,23 @@ "def read_json(fn):\n", " '''Read a json file (fn) and returns it as a dictionary'''\n", " with open(fn, 'r') as f:\n", - " return json.dumps(f.read())" + " return json.loads(f.read())" ] }, { "cell_type": "code", - "execution_count": 134, - "metadata": {}, + "execution_count": 6, + "metadata": { + "collapsed": true + }, "outputs": [], "source": [ "def write_data(fn, data):\n", " \"\"\"Backs up the previous version of the data if it exists and writes the new data to a file.\"\"\"\n", " # Backs up the previous data if it exists.\n", " try:\n", - " write_json(\"data/backup/{}.json \".format(fn) + time.ctime().replace(' ', '-'), \n", + " now = datetime.datetime.now().strftime('%H:%M-%d-%m-%Y')\n", + " write_json(\"data/backup/{}_{}.json \".format(fn, now), \n", " read_json(\"data/{}.json\".format(fn)))\n", " except:\n", " pass\n", @@ -149,25 +189,25 @@ }, { "cell_type": "code", - "execution_count": 6, + "execution_count": 7, "metadata": { "collapsed": true }, "outputs": [], "source": [ "def find(lst, key, value):\n", - " for i, dic in enumerate(lst):\n", + " '''Finds the first index of a list \n", + " lst where the key matches the value'''\n", + " for index, dic in enumerate(lst):\n", " if dic[key] == value:\n", - " return i\n", - " return None\n", - "\n", - "def sort_dict_list(data, x):\n", - " return sorted(data, key=lambda k: k[x]) " + " return index\n", + " \n", + " return None" ] }, { "cell_type": "code", - "execution_count": 7, + "execution_count": 8, "metadata": { "collapsed": true }, @@ -192,7 +232,7 @@ }, { "cell_type": "code", - "execution_count": 119, + "execution_count": 9, "metadata": { "collapsed": true, "scrolled": true @@ -203,25 +243,98 @@ " pass\n", "\n", "def test_tcp_udp(data, timeout = 5):\n", - " pbar = tqdm(total=len(data))\n", + " data_copy = copy.deepcopy(data)\n", + " \n", + " pbar = tqdm(total=len(data_copy))\n", "\n", - " for datum in data:\n", - " for p in (udp, tcp):\n", + " for datum in data_copy:\n", + " protocols = []\n", + " \n", + " if 'tcp' in datum and not datum['tcp']:\n", + " protocols.append(udp)\n", + " \n", + " if ('udp' in datum and not datum['udp']):\n", + " protocols.append(tcp)\n", + " \n", + " for p in protocols:\n", " # Create SOA query\n", " m = dns.message.make_query(datum['tld'], dns.rdatatype.SOA)\n", + " \n", " try: \n", " a = p(m, datum['ip'], timeout = timeout)\n", + " \n", " # We expect NOERROR RCODE (0) and an answer\n", " if a.rcode() == 0 and len(a.answer) > 0:\n", " datum[p.__name__] = True\n", - "\n", " else:\n", " raise CustomDNSException('failed')\n", + " \n", " except (dns.exception.Timeout, socket_error, CustomDNSException):\n", " datum[p.__name__] = False\n", "\n", " pbar.update(1)\n", - " pbar.close()" + " pbar.close()\n", + " \n", + " return data_copy" + ] + }, + { + "cell_type": "code", + "execution_count": 10, + "metadata": { + "collapsed": true + }, + "outputs": [], + "source": [ + "def find_nxdomain(tld, max_tries = 3):\n", + " for _ in range(max_tries):\n", + " domain = '{}.{}'.format(str(uuid4()), tld)\n", + " \n", + " try:\n", + " query(domain)\n", + " except NXDOMAIN:\n", + " return domain\n", + " except:\n", + " pass\n", + " \n", + " return None\n", + "\n", + "def find_nxdomain_wildcard(tld, max_tries = 3):\n", + " for _ in range(max_tries):\n", + " domain = '{}.{}'.format(str(uuid4()), tld)\n", + "\n", + " response = !dig soa +noall +authority +noidn {domain}\n", + "\n", + " if len(response) > 0 and response[0].startswith(tld):\n", + " return domain\n", + " \n", + " return None" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "# Init" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# First time run\n", + "!mkdir data\n", + "\n", + "!mkdir data/dig\n", + "!mkdir data/lists\n", + "!mkdir data/cymru\n", + "!mkdir data/whois\n", + "!mkdir data/atlas\n", + "!mkdir data/atlas/ns\n", + "!mkdir data/atlas/soa\n", + "!mkdir data/backup" ] }, { @@ -233,24 +346,24 @@ }, { "cell_type": "code", - "execution_count": 23, + "execution_count": 11, "metadata": {}, "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ - "--2017-06-13 13:09:02-- https://data.iana.org/TLD/tlds-alpha-by-domain.txt\n", + "--2017-07-12 20:37:59-- https://data.iana.org/TLD/tlds-alpha-by-domain.txt\n", "Loaded CA certificate '/etc/ssl/certs/ca-certificates.crt'\n", - "Resolving data.iana.org... 2606:2800:11f:bb5:f27:227f:1bbf:a0e, 72.21.81.189\n", - "Connecting to data.iana.org|2606:2800:11f:bb5:f27:227f:1bbf:a0e|:443... connected.\n", + "Resolving data.iana.org... 72.21.81.189, 2606:2800:11f:bb5:f27:227f:1bbf:a0e\n", + "Connecting to data.iana.org|72.21.81.189|:443... connected.\n", "HTTP request sent, awaiting response... 200 OK\n", - "Length: 10295 (10K) [text/plain]\n", + "Length: 10433 (10K) [text/plain]\n", "Saving to: ‘data/lists/tlds’\n", "\n", - "data/lists/tlds 100%[===================>] 10.05K --.-KB/s in 0s \n", + "data/lists/tlds 100%[===================>] 10.19K --.-KB/s in 0s \n", "\n", - "2017-06-13 13:09:08 (128 MB/s) - ‘data/lists/tlds’ saved [10295/10295]\n", + "2017-07-12 20:37:59 (189 MB/s) - ‘data/lists/tlds’ saved [10433/10433]\n", "\n" ] } @@ -262,7 +375,7 @@ }, { "cell_type": "code", - "execution_count": 24, + "execution_count": 12, "metadata": { "collapsed": true }, @@ -274,11 +387,11 @@ { "cell_type": "code", "execution_count": null, - "metadata": { - "collapsed": true - }, + "metadata": {}, "outputs": [], "source": [ + "# Gathers WHOIS records for each TLD.\n", + "\n", "pbar = tqdm(total=len(tlds))\n", "\n", "for tld in tlds:\n", @@ -290,7 +403,7 @@ }, { "cell_type": "code", - "execution_count": 92, + "execution_count": 13, "metadata": {}, "outputs": [ { @@ -303,6 +416,8 @@ } ], "source": [ + "# Gathers empty or missing WHOIS records.\n", + "\n", "indir = 'data/whois/'\n", "\n", "for root, dirs, filenames in os.walk(indir):\n", @@ -321,10 +436,12 @@ }, { "cell_type": "code", - "execution_count": 126, + "execution_count": 14, "metadata": {}, "outputs": [], "source": [ + "# Extracts the creation date and organisations for each TLD from the WHOIS record.\n", + "\n", "data_tlds = [{'tld': tld, 'organisations': []} for tld in tlds]\n", "\n", "for root, dirs, filenames in os.walk(indir):\n", @@ -337,27 +454,53 @@ " elif line.startswith('organisation'):\n", " _, org = line.split('rganisation:')\n", " index = find(data_tlds, 'tld', fn)\n", - " data_tlds[index]['organisations'].append(org.strip(newline))" + " data_tlds[index]['organisations'].append(org.strip(NEWLINE))" ] }, { "cell_type": "code", - "execution_count": 136, + "execution_count": 15, "metadata": {}, "outputs": [], "source": [ + "# Gets the type of each TLD listed in the table of the url.\n", + "\n", + "url = \"https://www.iana.org/domains/root/db/\"\n", + "html = urlopen(url)\n", + "soup = BeautifulSoup(html, 'html5lib')\n", + "\n", + "for table in soup.find_all(attrs={'class': 'iana-table'}):\n", + " values = [td.get_text(strip=True) for td in table.find_all('td')]\n", + " values = [td for td in table.find_all('td')]\n", + "\n", + "for i in range(0, len(values), 3):\n", + " tld = str(values[i].findAll('a', href=True)[0]).split('.html')[0][26:].upper()\n", + " index = find(data_tlds, 'tld', tld)\n", + " \n", + " if index != None:\n", + " data_tlds[index]['type'] = values[i + 1].get_text(strip = True)" + ] + }, + { + "cell_type": "code", + "execution_count": 16, + "metadata": { + "collapsed": true + }, + "outputs": [], + "source": [ "write_data('data_tlds', data_tlds)" ] }, { "cell_type": "code", - "execution_count": 152, + "execution_count": 17, "metadata": { "collapsed": true }, "outputs": [], "source": [ - "# Special TLDs are the same as record types or classes which do not work in bulk.\n", + "# Special TLDs are the same as record types or classes which do not work in some bulk operations.\n", "special_tlds = ['CH', 'IN', 'MD', 'MG', 'MR', 'MX']\n", "write_list('data/lists/tlds', [tld for tld in tlds if tld not in special_tlds])" ] @@ -371,18 +514,9 @@ }, { "cell_type": "code", - "execution_count": 18, + "execution_count": null, "metadata": {}, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "Gathering name servers.\n", - "Done.\n" - ] - } - ], + "outputs": [], "source": [ "# Gathers the name servers of every TLD using dig.\n", "print('Gathering name servers.')\n", @@ -396,15 +530,16 @@ }, { "cell_type": "code", - "execution_count": 139, + "execution_count": null, "metadata": { "collapsed": true }, "outputs": [], "source": [ + "# Parses the NS records.\n", + "\n", "data_ns = []\n", "\n", - "# Parses the answers of dig.\n", "with open('data/dig/tld_nss', 'r') as f:\n", " for line in f:\n", " if not line.startswith('.'):\n", @@ -423,31 +558,18 @@ }, { "cell_type": "code", - "execution_count": 20, - "metadata": { - "collapsed": true - }, + "execution_count": null, + "metadata": {}, "outputs": [], "source": [ - "write_list('data/lists/nss', set([datum['ns'] for datum in data]))" + "write_list('data/lists/nss', set([datum['ns'] for datum in data_ns]))" ] }, { "cell_type": "code", - "execution_count": 21, + "execution_count": null, "metadata": {}, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "Gathering IPv4 addresses.\n", - "Done.\n", - "Gathering IPv6 addresses.\n", - "Done.\n" - ] - } - ], + "outputs": [], "source": [ "print('Gathering IPv4 addresses.')\n", "!dig +noall +answer +noidn A -f data/lists/nss > data/dig/ns_ipv4s\n", @@ -460,10 +582,8 @@ }, { "cell_type": "code", - "execution_count": 143, - "metadata": { - "collapsed": true - }, + "execution_count": null, + "metadata": {}, "outputs": [], "source": [ "ns_ipv4s = ns_ips('data/dig/ns_ipv4s')\n", @@ -471,7 +591,10 @@ "\n", "data_ips = []\n", "\n", - "for datum in data:\n", + "# Adds the IP adress to the dictionary and creates a copy \n", + "# in case a name server has multiple IP addresses or has \n", + "# both a IPv4 and IPv6 address.\n", + "for datum in data_ns:\n", " if datum['ns'] in ns_ipv4s:\n", " for ip in ns_ipv4s[datum['ns']]:\n", " new_datum = copy.deepcopy(datum)\n", @@ -487,7 +610,7 @@ }, { "cell_type": "code", - "execution_count": 144, + "execution_count": null, "metadata": { "collapsed": true }, @@ -505,7 +628,7 @@ }, { "cell_type": "code", - "execution_count": 73, + "execution_count": null, "metadata": { "collapsed": true }, @@ -518,8 +641,9 @@ }, { "cell_type": "code", - "execution_count": 77, + "execution_count": null, "metadata": { + "collapsed": true, "scrolled": false }, "outputs": [], @@ -529,12 +653,15 @@ }, { "cell_type": "code", - "execution_count": 145, + "execution_count": null, "metadata": { + "collapsed": true, "scrolled": false }, "outputs": [], "source": [ + "# Makes a dictionary with the IP addresses as key and a list of ASNs as value.\n", + "\n", "ip_asns = {}\n", "\n", "with open('data/cymru/ip_asns', 'r') as f:\n", @@ -551,23 +678,12 @@ }, { "cell_type": "code", - "execution_count": 146, + "execution_count": null, "metadata": { "collapsed": true }, "outputs": [], "source": [ - "for datum in data_ips:\n", - " if datum['ip'] in ip_asns:\n", - " datum['asn'] = ip_asns[datum['ip']]" - ] - }, - { - "cell_type": "code", - "execution_count": 147, - "metadata": {}, - "outputs": [], - "source": [ "data_asns = []\n", "\n", "for datum in data_ips:\n", @@ -580,7 +696,7 @@ }, { "cell_type": "code", - "execution_count": 148, + "execution_count": null, "metadata": { "collapsed": true }, @@ -598,68 +714,34 @@ }, { "cell_type": "code", - "execution_count": 149, - "metadata": { - "scrolled": true - }, - "outputs": [], - "source": [ - "test_tcp_udp(data_ips)" - ] - }, - { - "cell_type": "code", "execution_count": null, "metadata": { "collapsed": true }, "outputs": [], "source": [ - "data_no_tcp_and_udp = [datum for datum in data_ips if not datum['tcp'] and not datum['udp']]\n", - "\n", - "test_tcp_udp(data_no_tcp_and_udp, timeout = 10)" + "data_ips = read_json('data/data_ips.json')" ] }, { "cell_type": "code", "execution_count": null, - "metadata": { - "collapsed": true - }, - "outputs": [], - "source": [ - "len([datum for datum in data_ips if not datum['tcp'] and not datum['udp']]), len(data_ips)" - ] - }, - { - "cell_type": "code", - "execution_count": 142, - "metadata": { - "collapsed": true - }, + "metadata": {}, "outputs": [], "source": [ - "write_data('data_prot', data_ips)" + "data_reach = test_tcp_udp(data_ips)" ] }, { "cell_type": "code", "execution_count": null, "metadata": { - "collapsed": true - }, - "outputs": [], - "source": [] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": { - "collapsed": true + "scrolled": true }, "outputs": [], "source": [ - "len(set([datum['tld'] for datum in data_ips if not datum['tcp'] and not datum['udp']]))" + "# Retries testing TCP or UDP\n", + "data_reach = test_tcp_udp(data_reach, timeout = 15)" ] }, { @@ -670,32 +752,48 @@ }, "outputs": [], "source": [ - "with open('data/backup/data_tcp_udp', 'w') as f:\n", - " f.write(json.dumps(data_ips))" + "write_data('data_reach', data_reach)" ] }, { - "cell_type": "code", - "execution_count": null, - "metadata": { - "collapsed": true - }, - "outputs": [], + "cell_type": "markdown", + "metadata": {}, "source": [ - "# with open('data/backup/data_tcp_udp', 'r') as f:\n", - "# data_ips = json.loads(f.read())" + "# Credibility" ] }, { "cell_type": "code", - "execution_count": null, - "metadata": { - "collapsed": true, - "scrolled": false - }, - "outputs": [], + "execution_count": 20, + "metadata": {}, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "Gathering DNSKEY records.\n", + "Done\n", + "Gathering DS records.\n", + "Done.\n", + "Gathering DNSKEY and DS records for special TLDs.\n", + "Done\n" + ] + } + ], "source": [ - "df = pd.DataFrame(data_ips)" + "print('Gathering DNSKEY records.')\n", + "!dig +noall +answer +noidn -t DNSKEY -f data/lists/tlds > data/dig/tld_dnskeys\n", + "print('Done')\n", + "\n", + "print('Gathering DS records.')\n", + "!dig +noall +answer +noidn -t DS -f data/lists/tlds > data/dig/tld_dss\n", + "print('Done.')\n", + "\n", + "print('Gathering DNSKEY and DS records for special TLDs.')\n", + "for tld in special_tlds:\n", + " !dig +noall +answer +noidn -t DNSKEY {tld} >> data/dig/tld_dnskeys\n", + " !dig +noall +answer +noidn -t DS {tld} >> data/dig/tld_dss\n", + "print('Done')" ] }, { @@ -706,29 +804,24 @@ }, "outputs": [], "source": [ - "xtlds = []\n", - "xnss = []\n", - "xips = []\n", - "xasns = []\n", - "xtcp = []\n", - "xudp = []\n", + "# Parses the DNSKEY records.\n", "\n", - "for datum in data_ips:\n", - " xtlds.append(datum['tld'])\n", - " xnss.append(datum['ns'])\n", - " xips.append(datum['ip'])\n", - " xtcp.append(datum['tcp'])\n", - " xudp.append(datum['udp'])\n", - " if 'asn' in datum:\n", - " xasns.append(datum['asn'])\n", - " else:\n", - " xasns.append([])\n", - " \n", - "print(len(data_ips), len(xtlds), len(xnss), len(xips))\n", + "data_cred = [{'tld': tld, 'ds': False, 'dnskey': False, 'jsj': None} for tld in tlds]\n", + "\n", + "for answer in read_list('data/dig/tld_dnskeys'):\n", + " answer_fields = answer.split()\n", + " tld = answer_fields[0][:-1].upper()\n", + " index = find(data_cred, 'tld', tld)\n", " \n", - "ix = pd.MultiIndex.from_arrays([xtlds, xnss, xips], names=['tld', 'ns', 'ip'])\n", - "dg = pd.DataFrame({'asn': xasns, 'tcp': xtcp, 'udp': xudp}, index = ix)\n", - "# dg.head(10)" + " try:\n", + " data_cred[index]['dnskey'] = True\n", + " \n", + " if int(answer_fields[4]) == KSK:\n", + " data_cred[index]['ksk'] = answer_fields[6]\n", + " elif int(answer_fields[4]) == ZSK:\n", + " data_cred[index]['zsk'] = answer_fields[6]\n", + " except:\n", + " print(tld)" ] }, { @@ -739,45 +832,18 @@ }, "outputs": [], "source": [ - "# dage = {}\n", + "# Parses the DS records.\n", "\n", - "# for datum in data_age:\n", - "# dage[datum['tld']] = datum['age']\n", - " \n", - "# for datum in data_ips:\n", - "# if datum['tld'][:-1].upper() in dage:\n", - "# if dage[datum['tld'][:-1].upper()] == 'new':\n", - "# datum['age'] = 'new'\n", - "# else:\n", - "# datum['age'] = 'old'\n", - "# else:\n", - "# datum['age'] = None\n", - "\n", - "# xtlds = []\n", - "# xnss = []\n", - "# xips = []\n", - "# xasns = []\n", - "# xtcp = []\n", - "# xudp = []\n", - "\n", - "# for datum in data_ips:\n", - "# if datum['age'] == 'old':\n", + "for answer in read_list('data/dig/tld_dss'):\n", + " v = answer.split()\n", + " tld = v[0][:-1].upper()\n", " \n", - "# xtlds.append(datum['tld'])\n", - "# xnss.append(datum['ns'])\n", - "# xips.append(datum['ip'])\n", - "# xtcp.append(datum['tcp'])\n", - "# xudp.append(datum['udp'])\n", - "# if 'asn' in datum:\n", - "# xasns.append(datum['asn'])\n", - "# else:\n", - "# xasns.append([])\n", - " \n", - "# print(len(data_ips), len(xtlds), len(xnss), len(xips))\n", + " index = find(data_cred, 'tld', tld)\n", " \n", - "# ix = pd.MultiIndex.from_arrays([xtlds, xnss, xips], names=['tld', 'ns', 'ip'])\n", - "# dg = pd.DataFrame({'asn': xasns, 'tcp': xtcp, 'udp': xudp}, index = ix)\n", - "# # dg.head(10)" + " try:\n", + " data_cred[index]['ds'] = True\n", + " except:\n", + " print(tld)" ] }, { @@ -788,235 +854,123 @@ }, "outputs": [], "source": [ - "# data_ips[0]" + "write_data('data_cred', data_cred)" ] }, { - "cell_type": "code", - "execution_count": null, - "metadata": { - "collapsed": true - }, - "outputs": [], + "cell_type": "markdown", + "metadata": {}, "source": [ - "ns_tcp_udp = [{'ns': datum['ns'], 'tcp': False, 'udp': False} for datum in data_ips if datum['tld'] in dtype and dtype[datum['tld']] == 'country-code']\n", - "\n", - "for ns in ns_tcp_udp:\n", - " for datum in data_ips:\n", - " if datum['ns'] == ns['ns']:\n", - " if datum['tcp']:\n", - " ns['tcp'] = True\n", - " \n", - " if datum['udp']:\n", - " ns['udp'] = True " + "# Performance" ] }, { "cell_type": "code", - "execution_count": null, - "metadata": { - "collapsed": true - }, - "outputs": [], + "execution_count": 22, + "metadata": {}, + "outputs": [ + { + "name": "stderr", + "output_type": "stream", + "text": [ + "100%|██████████| 1547/1547 [01:51<00:00, 13.85it/s]\n" + ] + } + ], "source": [ - "ns_tcp_udp = [{'ns': datum['ns'], 'tcp': False, 'udp': False} for datum in data_ips]\n", + "# Generates domains that result in a NXDOMAIN response for each TLD.\n", + "data_test_perf = []\n", + "pbar = tqdm(total=len(tlds))\n", "\n", - "for ns in ns_tcp_udp:\n", - " for datum in data_ips:\n", - " if datum['ns'] == ns['ns']:\n", - " if datum['tcp']:\n", - " ns['tcp'] = True\n", - " \n", - " if datum['udp']:\n", - " ns['udp'] = True " + "for tld in tlds:\n", + " data_test_perf.append({'tld': tld, 'domain': find_nxdomain(tld)})\n", + " \n", + " pbar.update(1)\n", + "pbar.close()" ] }, { "cell_type": "code", - "execution_count": null, - "metadata": { - "collapsed": true - }, - "outputs": [], + "execution_count": 23, + "metadata": {}, + "outputs": [ + { + "name": "stderr", + "output_type": "stream", + "text": [ + "100%|██████████| 38/38 [00:05<00:00, 3.85it/s]\n" + ] + } + ], "source": [ - "ip = ':'\n", - "\n", - "ns_tcp_udp = [{'ns': datum['ns'], 'tcp': False, 'udp': False} for datum in data_ips if ip in datum['ip']]\n", + "# Generates domains that result in a NXDOMAIN response for each TLD that uses wildcards.\n", + "pbar = tqdm(total=len([datum for datum in data_test_perf if not datum['domain']]))\n", "\n", - "for ns in ns_tcp_udp:\n", - " for datum in data_ips:\n", - " if datum['ns'] == ns['ns'] and ip in datum['ip']:\n", - " if datum['tcp']:\n", - " ns['tcp'] = True\n", - " \n", - " if datum['udp']:\n", - " ns['udp'] = True" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": { - "collapsed": true - }, - "outputs": [], - "source": [ - "dg = pd.DataFrame(ns_tcp_udp)" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": { - "collapsed": true - }, - "outputs": [], - "source": [ - "ax = dg.tcp.value_counts().plot.pie(autopct=lambda p : '{:.2f}% ({:.0f})'.format(p, p * dg.count().udp / 100), figsize = pie_size)\n", - "# ax = dg.tcp.plot.bar()\n", - "ax.set_ylabel('')\n", - "fig = ax.get_figure()\n", - "fig.savefig(\"imgs/tcp.pdf\")" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": { - "collapsed": true - }, - "outputs": [], - "source": [ - "ax = dg.udp.value_counts().plot.pie(autopct=lambda p : '{:.2f}% ({:.0f})'.format(p, p * dg.count().udp / 100), \n", - " figsize = pie_size)\n", - "ax.set_ylabel('')\n", - "# ax.set_title('UDP')\n", - "fig = ax.get_figure()\n", - "fig.savefig(\"imgs/udp.pdf\")" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": { - "collapsed": true - }, - "outputs": [], - "source": [ - "df_ftcp = dg.loc[dg.tcp == False]\n", - "df_fudp = dg.loc[dg.udp == False]\n", - "df_ttcp = dg.loc[dg.tcp == True]\n", - "\n", - "ff = df_ftcp.loc[df_ftcp.udp == False].count().tcp\n", - "ft = df_ftcp.loc[df_ftcp.udp == True].count().tcp\n", - "tf = df_fudp.loc[df_fudp.tcp == True].count().tcp\n", - "tt = df_ttcp.loc[df_ttcp.udp == True].count().tcp\n", - "\n", - "print(ff, ft, tf, tt)\n", - "\n", - "ut_data = [{'name': 'none', 'count': ff}, \n", - " {'name': 'tcp', 'count': tf}, \n", - " {'name': 'udp', 'count': ft}, \n", - " {'name': 'tcp + udp', 'count': tt}\n", - " ]\n", - "\n", - "total = ff + ft + tf + tt\n", - "\n", - "dfgh = pd.DataFrame(ut_data)\n", - "dfgh.index = dfgh['name']\n", - "del dfgh['name']\n", - "# ax = dfgh.plot.pie('count',\n", - "# # autopct='s(%.2f)',\n", - "# autopct=lambda p : '{:.2f}% ({:.0f})'.format(p, p * total / 100),\n", - "# # radius = 2.5,\n", - "# # pctdistance=1.2,\n", - "# # labeldistance=1.2,\n", - "# # explode = True,\n", - "# figsize = pie_size, \n", - "# legend=False, \n", - "# labels=['','','',''])\n", - "\n", - "ax = dfgh.plot.barh()\n", - "# ax.set_xlim([0,10000])\n", - "\n", - "ax.legend(loc='best', labels=dfgh.index)\n", - "ax.set_xlabel('Number of name servers')\n", - "ax.set_ylabel('Protocol(s) supported')\n", - "# ax.set_title('name server udp/tcp support')\n", - "ax.legend_.remove()\n", - "ax.set_xscale('log')\n", - "fig = ax.get_figure()\n", - "fig.tight_layout()\n", - "fig.savefig(\"imgs/tcp_udp_generic.pdf\")" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": { - "collapsed": true - }, - "outputs": [], - "source": [ - "dfgh.index" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": { - "collapsed": true - }, - "outputs": [], - "source": [ - "df.loc[df.tld == 'actor.']" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": { - "collapsed": true - }, - "outputs": [], - "source": [ - "# def write_to_file(fn, indir, content):\n", - "# with open(indir + fn, 'w') as f:\n", - "# f.write(content)" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": { - "collapsed": true - }, - "outputs": [], - "source": [ - "# write_to_file('tcp_udp_not_working', 'data/temp/', df_ftcp.loc[df_ftcp.udp == False].to_csv())" + "for datum in [datum for datum in data_test_perf if not datum['domain']]:\n", + " datum['domain'] = find_nxdomain_wildcard(datum['tld'])\n", + " \n", + " pbar.update(1)\n", + "pbar.close()" ] }, { "cell_type": "code", - "execution_count": null, + "execution_count": 24, "metadata": { "collapsed": true }, "outputs": [], "source": [ - "# df_ftcp.loc[df_ftcp.udp == False]" + "# Set the probes.\n", + "probe_ids = [10262, 10287, 11040, 11429, 12515, 12873, 12956, 13623, 13728, 13769, 13788, 13799, 13804, \n", + " 13805, 13810, 14237, 26057, 14564, 15156, 14691, 15594, 15799, 4205, 18131, 18195, 18691, \n", + " 19326, 19740, 20111, 20353, 20493, 20531, 20621, 21003, 21035, 21122, 21251, 21345, 21703, \n", + " 22286, 22695, 23031, 23085, 28240, 27972, 23697, 24807, 25011, 25148, 25323, 26936, 26378, \n", + " 26627, 4155, 26823, 28355, 30676, 4829, 29006, 29183, 29405, 30225, 30324, 31201, 19306, \n", + " 19634, 6025, 11660, 22388, 25182, 4123, 3812, 20923, 14384, 12389]\n", + "\n", + "probes = [\n", + " {\n", + " \"value\": str(probe_ids)[1:-1],\n", + " \"type\": \"probes\",\n", + " \"requested\": len(probe_ids)\n", + " }\n", + "]" ] }, { "cell_type": "code", - "execution_count": null, - "metadata": { - "collapsed": true - }, + "execution_count": 28, + "metadata": {}, "outputs": [], "source": [ - "# df.loc[df.ip.str.contains(':')].count().tcp #ipv6\n", - "# df[~df[\"ip\"].str.contains(\":\")].count().tcp #ipv4" + "payloads = []\n", + "payload_size = 100 # max 100\n", + "step_size = int(payload_size / 2)\n", + "\n", + "for i in range(0, len(data_test_perf), step_size):\n", + " defintions = []\n", + " \n", + " for datum in data_test_perf[i:i + step_size]:\n", + " # Create caching measurement\n", + " definition_caching = definition.copy()\n", + " definition_caching['query_type'] = \"NS\"\n", + " definition_caching['query_argument'] = datum['tld']\n", + " definition_caching['description'] = \"caching \" + datum['tld']\n", + " defintions.append(definition_caching)\n", + " \n", + " # Create response time measurement\n", + " definition_measuring = definition.copy()\n", + " definition_measuring['query_type'] = \"SOA\"\n", + " definition_measuring['query_argument'] = datum['domain']\n", + " definition_measuring['description'] = \"measuring \" + datum['tld']\n", + " defintions.append(definition_measuring) \n", + "\n", + " new_payload = payload.copy()\n", + " new_payload['probes'] = probes\n", + " new_payload['definitions'] = defintions\n", + "\n", + " payloads.append(new_payload)" ] }, { @@ -1027,10 +981,7 @@ }, "outputs": [], "source": [ - "# ax = df[~df[\"ip\"].str.contains(\":\")].tcp.value_counts().plot.pie(autopct='%.2f', figsize = pie_size)\n", - "# ax.set_ylabel('')\n", - "# fig = ax.get_figure()\n", - "# fig.savefig(\"imgs/tcp_ipv4.pdf\")" + "write_data('payloads', payloads)" ] }, { @@ -1041,10 +992,9 @@ }, "outputs": [], "source": [ - "# ax = df[~df[\"ip\"].str.contains(\":\")].udp.value_counts().plot.pie(autopct='%.2f', figsize = pie_size)\n", - "# ax.set_ylabel('')\n", - "# fig = ax.get_figure()\n", - "# fig.savefig(\"imgs/udp_ipv4.pdf\")" + "measurement_ids = []\n", + "measurement_responses = []\n", + "url = URL_DNS_MEASUREMENT_CREATE + '?key=' + ATLAS_API_KEY" ] }, { @@ -1055,10 +1005,21 @@ }, "outputs": [], "source": [ - "# ax = df.loc[df.ip.str.contains(':')].tcp.value_counts().plot.pie(autopct='%.2f', figsize = pie_size)\n", - "# ax.set_ylabel('')\n", - "# fig = ax.get_figure()\n", - "# fig.savefig(\"imgs/tcp_ipv6.pdf\")" + "# Only 10 payloads can be sent per day.\n", + "# Manually set this slice each day.\n", + "for payload in payloads[:10]:\n", + " \n", + " request = requests.post(url, data = json.dumps(payload), headers = HEADERS)\n", + " print(request.status_code)\n", + " \n", + " while request.status_code == 400:\n", + " print(request.json())\n", + " request = requests.post(url, data = json.dumps(payload), headers = HEADERS)\n", + " time.sleep(300)\n", + " print(request.status_code)\n", + " \n", + " measurement_ids += measurement_ids + request.json()\n", + " measurement_ids = [min(measurement_ids), max(measurement_ids)]" ] }, { @@ -1069,10 +1030,40 @@ }, "outputs": [], "source": [ - "# ax = df.loc[df.ip.str.contains(':')].udp.value_counts().plot.pie(autopct='%.2f', figsize = pie_size)\n", - "# ax.set_ylabel('')\n", - "# fig = ax.get_figure()\n", - "# fig.savefig(\"imgs/udp_ipv6.pdf\")" + "# Retrieves the result of each measurement and writes it to a file.\n", + "\n", + "next_result = True\n", + "url = '{}?id__gte={}&id__lte={}&mine=true'.format(URL_DNS_MEASUREMENT_GET, min(measurement_ids), max(measurement_ids))\n", + "measurements = requests.get(url).json()\n", + "pbar = tqdm(total=len(tlds))\n", + "\n", + "while next_result:\n", + " for result in measurements['results']:\n", + " if len(result['description'].split()) == 2:\n", + " measurement_type, tld = result['description'].split()\n", + " request = requests.get(result['result']).json()\n", + "\n", + " if measurement_type == 'measuring':\n", + " with open('data/atlas/soa/{}.json'.format(tld.upper()), 'w') as f:\n", + " temp = copy.deepcopy(result)\n", + " temp['result'] = request\n", + "\n", + " f.write(json.dumps(temp))\n", + "\n", + " elif measurement_type == 'caching':\n", + " with open('data/atlas/ns/{}.json'.format(tld.upper()), 'w') as f:\n", + " temp = copy.deepcopy(result)\n", + " temp['result'] = request\n", + "\n", + " f.write(json.dumps(temp))\n", + " \n", + " if measurements['next']:\n", + " measurements = requests.get(measurements['next']).json() \n", + " else:\n", + " next_result = False\n", + "\n", + " pbar.update(1)\n", + "pbar.close()" ] }, { @@ -1083,109 +1074,48 @@ }, "outputs": [], "source": [ - "df.head()" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "# Credibility" - ] - }, - { - "cell_type": "code", - "execution_count": 201, - "metadata": { - "collapsed": true - }, - "outputs": [ - { - "name": "stderr", - "output_type": "stream", - "text": [ - "\n", - "\n", - "\n", - "\n", - "\n", - "\n", - "\n", - "\n", - "\n", - "\n", - "\n", - "\n", - "\n", - "\n" - ] - } - ], - "source": [ - "!dig +noall +answer +noidn -t DNSKEY -f data/lists/tlds > data/dig/tld_dnskeys\n", - "!dig +noall +answer +noidn -t DS -f data/lists/tlds > data/dig/tld_dss\n", + "# Extracts the response time from the measurement results.\n", "\n", - "for tld in special_tlds:\n", - " !dig +noall +answer +noidn -t DNSKEY {tld} >> data/dig/tld_dnskeys\n", - " !dig +noall +answer +noidn -t DS {tld} >> data/dig/tld_dss" - ] - }, - { - "cell_type": "code", - "execution_count": 235, - "metadata": {}, - "outputs": [], - "source": [ - "data_cred = [{'tld': tld, 'ds': False, 'dnskey': False, 'algorithm': None} for tld in tlds]\n", + "indir = 'data/atlas/soa/'\n", "\n", - "temp = []\n", + "data_perf = []\n", "\n", - "for answer in read_list('data/dig/tld_dnskeys'):\n", - " v = answer.split()\n", - " tld = v[0][:-1].upper()\n", - " index = find(data_cred, 'tld', tld)\n", - " \n", - " try:\n", - " data_cred[index]['dnskey'] = True\n", - " data_cred[index]['algorithm'] = v[6]\n", - " except:\n", - " print(tld)" - ] - }, - { - "cell_type": "code", - "execution_count": 236, - "metadata": {}, - "outputs": [], - "source": [ - "for answer in read_list('data/dig/tld_dss'):\n", - " v = answer.split()\n", - " tld = v[0][:-1].upper()\n", - " \n", - " index = find(data_cred, 'tld', tld)\n", - " \n", - " try:\n", - " data_cred[index]['ds'] = True\n", - " except:\n", - " print(tld)" + "for root, dirs, filenames in os.walk(indir):\n", + " for f in filenames:\n", + " tld, _ = f.split('.')\n", + " \n", + " datum = {'tld': tld, 'rt': [], 'timeouts': 0}\n", + " \n", + " with open(indir + f, 'r') as f:\n", + " tld_results = json.loads(f.read())\n", + " \n", + " for probe in tld_results['result']:\n", + " for result in probe['resultset']:\n", + " if 'result' in result:\n", + " datum['rt'].append(result['result']['rt']) \n", + " elif 'error' in result and 'timeout' in result['error']:\n", + " datum['timeouts'] += 1\n", + " \n", + " datum['rt'] = np.mean(datum['rt'])\n", + " data_perf.append(datum)" ] }, { "cell_type": "code", - "execution_count": 237, + "execution_count": null, "metadata": { "collapsed": true }, "outputs": [], "source": [ - "write_data('data_cred', data_cred)" + "write_data('data_perf', data_perf)" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ - "# Organisations per TLD" + "# Anycast" ] }, { @@ -1195,17 +1125,13 @@ "collapsed": true }, "outputs": [], - "source": [] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": { - "collapsed": true - }, - "outputs": [], "source": [ - "# tld_orgs" + "test_set = [\n", + " {'target': '185.49.140.60', 'argument': 'nlnetlabs.nl', 'ns': 'ns.nlnetlabs.nl'},\n", + " {'target': '192.16.197.229', 'argument': 'nlnet.nl', 'ns': 'mcvax.nlnet.nl'},\n", + " {'target': '194.0.28.53', 'argument': 'nl', 'ns': 'ns5.dns.nl'},\n", + " {'target': '204.61.216.4', 'argument': 'nlnetlabs.nl', 'ns': 'anyns.pch.net'}\n", + "]" ] }, { @@ -1216,10 +1142,21 @@ }, "outputs": [], "source": [ - "df_orgs = pd.DataFrame(tld_orgs)\n", - "df_orgs.index = df_orgs['tld']\n", - "del df_orgs['tld']\n", - "df_orgs.head()" + "definitions = []\n", + "n = 3 # number of measurements\n", + "\n", + "for test in test_set:\n", + " for i in range(1, n + 1): \n", + " new_definition = copy.deepcopy(definition)\n", + " new_definition['type'] = 'SOA'\n", + " new_definition['query_argument'] = test['argument']\n", + " new_definition['use_probe_resolver'] = False\n", + " new_definition['target'] = test['target']\n", + " new_definition['description'] = 'anycast {} {}'.format(i, test['ns'])\n", + "\n", + " definitions.append(new_definition)\n", + "\n", + "payload['definitions'] = definitions" ] }, { @@ -1230,13 +1167,44 @@ }, "outputs": [], "source": [ - "df_orgs.reset_index(inplace=True)\n", - "rows = []\n", - "_ = df_orgs.apply(lambda row: [rows.append([row['tld'], nn]) \n", - " for nn in row.organisations], axis=1)\n", - "df_orgs_new = pd.DataFrame(rows, columns=df_orgs.columns).set_index(['tld'])\n", + "anycast_probes = [\n", + " # North America\n", + " {'id': 22447, 'country-code': 'US', 'city': 'San Francisco'},\n", + " {'id': 14233, 'country-code': 'US', 'city': 'Denver'},\n", + " {'id': 25081, 'country-code': 'US', 'city': 'Washington'},\n", + " # South America\n", + " {'id': 31450, 'country-code': 'CR', 'city': 'San Jose'},\n", + " {'id': 30185, 'country-code': 'BR', 'city': 'Sao Paulo'},\n", + " {'id': 30123, 'country-code': 'CL', 'city': 'Santiago'},\n", + " # Europe\n", + " {'id': 32669, 'country-code': 'GR', 'city': 'Athens'},\n", + " {'id': 31479, 'country-code': 'RU', 'city': 'Moscow'},\n", + " {'id': 29762, 'country-code': 'ES', 'city': 'Madrid'},\n", + " {'id': 26610, 'country-code': 'NL', 'city': 'Utrecht'},\n", + " # Africa\n", + " {'id': 22458, 'country-code': 'ZA', 'city': 'Cape Town'},\n", + " {'id': 13258, 'country-code': 'AE', 'city': 'Dubai'},\n", + " {'id': 28493, 'country-code': 'SN', 'city': 'Dakar'},\n", + " # Asia\n", + " {'id': 28819, 'country-code': 'JP', 'city': 'Tokyo'},\n", + " {'id': 28964, 'country-code': 'KR', 'city': 'Seoul'},\n", + " {'id': 6107, 'country-code': 'IN', 'city': 'Mumbai'},\n", + " {'id': 25047, 'country-code': 'HK', 'city': 'Hong Kong'},\n", + " {'id': 26378, 'country-code': 'KG', 'city': 'Bishkek'},\n", + " # Oceania\n", + " {'id': 25208, 'country-code': 'AU', 'city': 'Sydney'},\n", + " {'id': 28226, 'country-code': 'NZ', 'city': 'Welington'}\n", + "]\n", + "\n", + "probes = [\n", + " {\n", + " \"value\": str([probe['id'] for probe in anycast_probes])[1:-1],\n", + " \"type\": \"probes\",\n", + " \"requested\": len(anycast_probes)\n", + " }\n", + "]\n", "\n", - "df_orgs_new.head()" + "payload['probes'] = probes" ] }, { @@ -1247,7 +1215,8 @@ }, "outputs": [], "source": [ - "df_orgs_new.organisations.value_counts(ascending=False).head(80).plot.barh(figsize = (10,20))" + "url = '{}?key={}'.format(URL_DNS_MEASUREMENT_CREATE, atlas_api_key)\n", + "request = requests.post(url, data = json.dumps(payload), headers = HEADERS)" ] }, { @@ -1258,9 +1227,7 @@ }, "outputs": [], "source": [ - "bins = df_orgs_new.organisations.value_counts().nunique() - 1\n", - "ax = df_orgs_new.organisations.value_counts().hist(bins = bins)\n", - "ax.set_yscale('log')" + "measurement_ids = request.json()" ] }, { @@ -1271,7 +1238,20 @@ }, "outputs": [], "source": [ - "df_orgs_new.organisations.value_counts().value_counts().plot.pie()" + "# Only 10 payloads can be sent per day.\n", + "# Manually set this slice each day.\n", + "for payload in payloads[:10]:\n", + " \n", + " request = requests.post(url, data = json.dumps(payload), headers = HEADERS)\n", + " print(request.status_code)\n", + " \n", + " while request.status_code == 400:\n", + " print(request.json())\n", + " request = requests.post(url, data = json.dumps(payload), headers = HEADERS)\n", + " time.sleep(300)\n", + " print(request.status_code)\n", + " \n", + " measurement_ids += measurement_ids + request.json()" ] }, { @@ -1282,45 +1262,49 @@ }, "outputs": [], "source": [ - "# df_orgs_new.organisations.value_counts()\n", + "next_result = True\n", + "url = '{}?id__in={}&mine=true'.format(URL_DNS_MEASUREMENT_GET, str(measurement_ids)[1:-1])\n", + "measurements = requests.get(url).json()\n", "\n", - "# df2[df2['rr_quality'] > 0]].groupby([df2.index.hour,'sleep_summary_id')" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": { - "collapsed": true - }, - "outputs": [], - "source": [ - "ax = df_tld_orgs.organisation.value_counts().head(20).plot.barh(figsize = bar_size, fontsize=12)\n", - "ax.set_xlabel('Number of TLDs', fontsize = 16)\n", - "ax.set_ylabel('Organisation',fontsize = 16)\n", - "fig = ax.get_figure()\n", - "fig.savefig(\"imgs/orgs.png\")" + "while next_result:\n", + " for result in measurements['results']:\n", + " if len(result['description'].split()) == 3:\n", + " measurement_type, i, ns = result['description'].split()\n", + " request = requests.get(result['result']).json()\n", + "\n", + " if measurement_type == 'anycast':\n", + " temp = copy.deepcopy(result)\n", + " temp['result'] = request\n", + " \n", + " write_json('data/atlas/anycast/{}.{}.json'.format(ns, i), temp)\n", + " \n", + " if measurements['next']:\n", + " measurements = requests.get(measurements['next']).json() \n", + " else:\n", + " next_result = False" ] }, { "cell_type": "code", "execution_count": null, - "metadata": { - "collapsed": true - }, - "outputs": [], - "source": [ - "ax = df_tld_orgs.type.value_counts().plot.pie(figsize = pie_size, legend=True)\n", - "ax.set_ylabel('')\n", - "fig = ax.get_figure()\n", - "fig.savefig(\"imgs/types.png\")" - ] - }, - { - "cell_type": "markdown", "metadata": {}, + "outputs": [], "source": [ - "# Growth" + "indir = 'data/atlas/anycast/'\n", + "probe_data = {k:v for k, v in [(probe_id, []) for probe_id in [probe['id'] for probe in anycast_probes]]}\n", + "data_ac = [{'ns': test['ns'], 'probes': copy.deepcopy(probe_data)} for test in test_set]\n", + "\n", + "for root, dirs, filenames in os.walk(indir):\n", + " for f in filenames:\n", + " result = read_json(indir + f)\n", + " _, _, ns = result['description'].split()\n", + " index = find(data_ac, 'ns', ns)\n", + " \n", + " for probe in result['result']:\n", + " try:\n", + " data_ac[index]['probes'][probe['prb_id']].append(probe['result']['rt'])\n", + " except:\n", + " print('x')" ] }, { @@ -1331,17 +1315,7 @@ }, "outputs": [], "source": [ - "data_age = []\n", - "dage = {}\n", - "\n", - "for datum in tld_creation:\n", - " y, m, d = datum['date_created'].split('-')\n", - " if y in ['2014', '2015', '2016', '2017'] or y == '2013' and int(m) >= 10:\n", - " data_age.append({'tld': datum['tld'], 'age': 'new'})\n", - " dage[datum['tld']] = 'new'\n", - " else:\n", - " data_age.append({'tld': datum['tld'], 'age': 'old'})\n", - " dage[datum['tld']] = 'old'" + "write_data('data_ac', data_ac)" ] } ], |