Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
41 changes: 34 additions & 7 deletions get_stats.sh
Original file line number Diff line number Diff line change
@@ -1,26 +1,53 @@
#!/bin/bash

# Download crawl statistics and top-domains csv files into ./stats/
#
# Usage: ./get_stats.sh
# Test: Override TOP_DOMAINS_S3/URL, eg.
# TOP_DOMAINS_S3=s3://test-bucket/test-prefix/crawl-analysis ./get_stats.sh


set -o pipefail

if aws s3 ls s3://commoncrawl/crawl-analysis/ | sed -E 's@.* @@; s@/$@@' >./stats/crawls.txt; then
TOP_DOMAINS_FILE=domains-top-1000-extended.csv.gz
TOP_DOMAINS_FOLDER=./stats/top-domains

CRAWL_ANALYSIS_S3=s3://commoncrawl/crawl-analysis
CRAWL_ANALYSIS_URL=https://data.commoncrawl.org/crawl-analysis

# Override to test, see above.
TOP_DOMAINS_S3=${TOP_DOMAINS_S3:-$CRAWL_ANALYSIS_S3}
TOP_DOMAINS_URL=${TOP_DOMAINS_URL:-$CRAWL_ANALYSIS_URL}


if aws s3 ls ${CRAWL_ANALYSIS_S3}/ | sed -E 's@.* @@; s@/$@@' >./stats/crawls.txt; then
ON_AWS=true;
echo "Running on AWS (AWS CLI configured for authenticated access)"
else
echo "Downloading from https://data.commoncrawl.org/ using curl"
echo "Downloading from ${CRAWL_ANALYSIS_URL} using curl"
# list of crawls enumerated in crawlstats.py
python3 -c 'from crawlstats import MonthlyCrawl; [print(c) for c in sorted(MonthlyCrawl.by_name.keys())]' >./stats/crawls.txt
ON_AWS=false
fi

mkdir -p ${TOP_DOMAINS_FOLDER}

while read crawl; do
echo $crawl
if [ -e stats/$crawl.gz ]; then
echo " ... exists"
continue
echo " ... stats exist"
elif $ON_AWS; then
aws s3 cp ${CRAWL_ANALYSIS_S3}/$crawl/stats/part-00000.gz ./stats/$crawl.gz
else
curl --silent ${CRAWL_ANALYSIS_URL}/$crawl/stats/part-00000.gz >./stats/$crawl.gz
fi
if $ON_AWS; then
aws s3 cp s3://commoncrawl/crawl-analysis/$crawl/stats/part-00000.gz ./stats/$crawl.gz
# top-domains csv might be missing for older crawls, in that case continue without failing
TOP_DOMAINS_TARGET=${TOP_DOMAINS_FOLDER}/${crawl}.${TOP_DOMAINS_FILE}
if [ -e ${TOP_DOMAINS_TARGET} ]; then
echo " ... top-domains exist"
elif $ON_AWS; then
aws s3 cp ${TOP_DOMAINS_S3}/$crawl/stats/${TOP_DOMAINS_FILE} ${TOP_DOMAINS_TARGET}
else
curl --silent https://data.commoncrawl.org/crawl-analysis/$crawl/stats/part-00000.gz >./stats/$crawl.gz
curl --silent --fail ${TOP_DOMAINS_URL}/$crawl/stats/${TOP_DOMAINS_FILE} -o ${TOP_DOMAINS_TARGET} || { rm -f ${TOP_DOMAINS_TARGET}; false; }
fi
done <./stats/crawls.txt
2 changes: 1 addition & 1 deletion index.md
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,7 @@ Statistics of [Common Crawl](https://commoncrawl.org/)'s [web archives](https://

* [size of the crawls](plots/crawlsize) - number of pages, unique URLs, hosts, domains, top-level domains (public suffixes), cumulative growth of crawled data over time
* [top-level domains](plots/tlds) - distribution and comparison
* [top-500 registered domains](plots/domains.md)
* [top-1000 registered domains](plots/domains.md)
* [crawler-related metrics](plots/crawlermetrics) - fetch status, etc.
* [overlaps between monthly crawls](plots/crawloverlap)
* distribution of
Expand Down
4 changes: 2 additions & 2 deletions plot.sh
Original file line number Diff line number Diff line change
Expand Up @@ -96,7 +96,7 @@ zcat stats/excerpt/charset.json.gz \
zcat stats/excerpt/language.json.gz \
| python3 plot/language.py

zcat stats/excerpt/domain.json.gz \
zcat stats/excerpt/size.json.gz \
| python3 plot/domain.py

echo -e "\n\nAll crawl statistics plotted\n"
echo -e "\n\nAll crawl statistics plotted\n"
47 changes: 24 additions & 23 deletions plot/domain.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,49 +2,48 @@

import pandas

from crawlstats import CST, MonthlyCrawl, MultiCount
from crawlstats import CST, MonthlyCrawl
from plot.table import TabularStats


class DomainStats(TabularStats):

# defined via crawlstats command-line option --max-top-hosts-domains
MAX_TOP_DOMAINS = 500
# extended top domains csv fetched by get_stats.sh
MAX_TOP_DOMAINS = 1000
TOP_DOMAINS_FILE = 'stats/top-domains/{}.domains-top-{}-extended.csv.gz'

def __init__(self, crawl):
super().__init__()
self.crawl = crawl
self.N = 0

def add(self, key, val):
"""Collect the crawl size records read from stdin."""
cst = CST[key[0]]
if cst not in (CST.size, CST.domain):
if cst != CST.size:
return
typeval = key[1]
crawl = key[2]
if crawl != self.crawl:
if key[2] != self.crawl:
return
if cst == CST.size:
self.size[typeval] = val
return
self.type_stats['domain'][self.N] = typeval
self.type_stats['pages'][self.N] = MultiCount.get_count(0, val)
self.type_stats['urls'][self.N] = MultiCount.get_count(1, val)
self.type_stats['hosts'][self.N] = MultiCount.get_count(2, val)
# self.type_stats['crawl'][self.N] = crawl
self.N += 1
self.size[key[1]] = val

def read_top_domains(self):
"""Read the downloaded top domains csv, handles gzipped or plain versions."""
path = self.TOP_DOMAINS_FILE.format(self.crawl, self.MAX_TOP_DOMAINS)
with open(path, 'rb') as instream:
gzipped = instream.read(2) == b'\x1f\x8b'
self.type_stats = pandas.read_csv(path, compression='gzip' if gzipped else None)

def transform_data(self):
data = pandas.DataFrame(self.type_stats)
"""Add the percentage columns, counts are crawl totals."""
data = self.type_stats
for cnt in ['pages', 'urls']:
total = self.size[cnt[:-1]]
data['%' + cnt] = 100.0 * data[cnt] / total
data.sort_values(ascending=False, inplace=True, by='pages')
print(data)
self.type_stats = data

def save_data(self, name, dir_name='data/'):
self.type_stats.to_csv('{}/{}-top-{}.csv'.format(self.PLOTDIR, name, self.MAX_TOP_DOMAINS),
def save_data(self, name):
"""Write the top domains csv next to the html table."""
self.type_stats.to_csv('{}/{}-top-{}.csv'.format(
self.PLOTDIR, name, self.MAX_TOP_DOMAINS),
float_format='%.6f', index=None)

def plot(self, name):
Expand All @@ -58,6 +57,7 @@ def plot(self, name):
float_format='%.6f',
classes=css_classes, index='domain'))


if __name__ == '__main__':
plot_crawls = sys.argv[1:]
if len(plot_crawls) == 0:
Expand All @@ -67,6 +67,7 @@ def plot(self, name):
plot_name = 'domains'
plot = DomainStats(latest_crawl)
plot.read_from_stdin_or_file()
plot.read_top_domains()
plot.transform_data()
plot.save_data(plot_name, dir_name=plot.PLOTDIR)
plot.save_data(plot_name)
plot.plot(plot_name)
8 changes: 4 additions & 4 deletions plots/domains.md
Original file line number Diff line number Diff line change
@@ -1,15 +1,15 @@
---
layout: table
table_include: domains-top-500.html
table_include: domains-top-1000.html
table_sortlist: "{sortList: [[1,1]]}"
table_searcher: "Filter for domain names"
---

Top-500 Registered Domains of the Latest Main Crawl
Top-1000 Registered Domains of the Latest Main Crawl
===================================================

The table below shows the top 500 registered domains (in terms of page captures) of the last main/monthly crawl
({{ site.latest_crawl }}). The underlying data is also provided in CSV format, see [domains-top-500.csv](./domains-top-500.csv).
The table below shows the top 1000 registered domains (in terms of page captures) of the last main/monthly crawl
({{ site.latest_crawl }}). The underlying data is also provided in CSV format, see [domains-top-1000.csv](./domains-top-1000.csv).

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I am leaving this file (repo-only) uncompressed, so its easier for the github.io users to open.


Note that the ranking by page captures only partially corresponds to the importance of domains, as the
crawler respects the robots.txt and tries hard not to overload web servers. Highly ranked domains tend to be
Expand Down
Loading