{ "cells": [ { "cell_type": "code", "execution_count": 1, "metadata": { "collapsed": true }, "outputs": [], "source": [ "from sortedcontainers import SortedDict\n", "from urllib.request import urlopen\n", "from bs4 import BeautifulSoup\n", "from pprint import pprint\n", "\n", "import pandas as pd\n", "import numpy as np\n", "import matplotlib as plt\n", "\n", "import json\n", "\n", "%matplotlib inline" ] }, { "cell_type": "code", "execution_count": 2, "metadata": { "collapsed": true }, "outputs": [], "source": [ "data = []" ] }, { "cell_type": "code", "execution_count": 3, "metadata": { "collapsed": true }, "outputs": [], "source": [ "d = SortedDict()\n", "\n", "with open('data/root.zone') as f:\n", " for i in range(21):\n", " next(f)\n", " \n", " for line in f:\n", " values = line.split('\\t')\n", " if 'NS' in values:\n", " tld = values[0][:-1]\n", " ns = values[-1][:-2]\n", " if tld in d:\n", " d[tld].append(ns)\n", " else:\n", " d[tld] = [ns]" ] }, { "cell_type": "code", "execution_count": 4, "metadata": { "collapsed": true }, "outputs": [], "source": [ "f = open('test.csv', 'w')\n", "f.write('tld,ns\\n')\n", "for i in d:\n", " for j in d[i]:\n", " f.write(i + ',' + j + '\\n')\n", " \n", "f.close()" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "collapsed": true }, "outputs": [], "source": [ "df = pd.read_csv(\"test.csv\")\n", "df2 = df.groupby('tld')['ns'].nunique()\n", "df2.head()" ] }, { "cell_type": "code", "execution_count": 5, "metadata": { "collapsed": true }, "outputs": [], "source": [ "url = \"https://www.iana.org/domains/root/db/\"\n", "html = urlopen(url)\n", "soup = BeautifulSoup(html, 'html5lib')\n", "\n", "for item in soup.find_all(attrs={'class': 'iana-table'}):\n", " for table in soup.find_all(attrs={'class': 'iana-table'}):\n", " values = [td.get_text(strip=True) for td in table.find_all('td')]\n", "\n", " for i in range(0, len(values), 3):\n", " data.append({'tld': values[i].strip('.'), 'type': values[i + 1], 'organisation': values[i + 2]})" ] }, { "cell_type": "code", "execution_count": 9, "metadata": {}, "outputs": [], "source": [ "df = pd.DataFrame(data)\n", "\n", "with open('data/tld_type', 'w') as f:\n", " f.write(json.dumps(data))" ] }, { "cell_type": "code", "execution_count": null, "metadata": { "collapsed": true }, "outputs": [], "source": [ "for tld in data:\n", " data[tld]['nameservers'] = SortedDict()\n", " html = urlopen(url + tld)\n", " soup = BeautifulSoup(html, 'html5lib')\n", "\n", " for br in soup.find_all('br'):\n", " br.replace_with('\\t')\n", "\n", " for item in soup.find_all(attrs={'class': 'iana-table'}):\n", " for table in soup.find_all(attrs={'class': 'iana-table'}):\n", " values = [td.get_text(strip=False) for td in table.find_all('td')]\n", " \n", " print(values[1].split('\\t')[:-1])\n", "\n", " for i in range(0, len(values), 2):\n", " ips = values[i + 1].split('\\t')\n", " \n", " data[tld]['nameservers'][values[i]] = SortedDict()\n", " \n", " if '.' in ips[0]:\n", " data[tld]['nameservers'][values[i]]['ipv4'] = ips[0]\n", " elif '.' in ips[1]:\n", " data[tld]['nameservers'][values[i]]['ipv4'] = ips[1]\n", " \n", " if 'nameservers' in data[tld]:\n", " data[tld]['nameservers'].append((values[i], values[i + 1].split('\\t')))\n", " else:\n", " data[tld]['nameservers'] = [(values[i], values[i + 1].split('\\t'))]" ] }, { "cell_type": "code", "execution_count": 5, "metadata": { "collapsed": true }, "outputs": [], "source": [ "df = pd.DataFrame(data)" ] }, { "cell_type": "code", "execution_count": 72, "metadata": {}, "outputs": [ { "data": { "text/html": [ "
| \n", " | count | \n", "
|---|---|
| organisation | \n", "\n", " |
| Amazon Registry Services, Inc. | \n", "51 | \n", "
| Charleston Road Registry Inc. | \n", "43 | \n", "
| Uniregistry, Corp. | \n", "22 | \n", "
| United TLD Holdco Ltd. | \n", "22 | \n", "
| Top Level Domain Holdings Limited | \n", "18 | \n", "
| Afilias plc | \n", "14 | \n", "
| Not assigned | \n", "14 | \n", "
| Internet Assigned Numbers Authority | \n", "12 | \n", "
| Lifestyle Domain Holdings, Inc. | \n", "11 | \n", "
| Dish DBS Corporation | \n", "11 | \n", "