-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathwarcstat.py
More file actions
129 lines (108 loc) · 5.18 KB
/
Copy pathwarcstat.py
File metadata and controls
129 lines (108 loc) · 5.18 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
from collections import defaultdict
from urllib.parse import urlparse
from warcio.archiveiterator import ArchiveIterator
from warcio.exceptions import ArchiveLoadFailed
import argparse
import os
import json
import logging
def setup_logging(log_dir='logs'):
# create dir if it does not exist
os.makedirs(log_dir, exist_ok=True)
# configure logging
log_file = os.path.join(log_dir, 'warcstat.log')
logging.basicConfig(
level = logging.INFO,
format="%(asctime)s [%(levelname)s] %(message)s",
handlers=[
logging.FileHandler(log_file),
logging.StreamHandler()
]
)
return logging.getLogger(__name__)
def collect_stats(stream):
# initialize dictionary
stats = {
'total_records': 0, # counts total number of records in WARC
'total_bytes': 0, # tracks totals size of all records in bytes
'errors': [], # stores list of 4xx/5xx HTTP responses
'redirects': [], # stores list of 3xx HTTP responses
'hosts': defaultdict(int), # count of requests per host
'http_status_codes': defaultdict(int), # count of HTTP status codes
'mime_types': defaultdict(int), # count of MIME types
'record_types': defaultdict(int) # count of WARC records types
}
for record in ArchiveIterator(stream):
stats['total_records'] += 1
stats['record_types'][record.rec_type] += 1
if record.length:
stats['total_bytes'] += record.length
if record.rec_type == 'response' and record.http_headers: # only records that are HTTP responses & have HTTP headers
status = record.http_headers.get_statuscode()
uri = record.rec_headers.get_header('WARC-Target-URI')
if uri:
host = urlparse(uri).netloc # extract domain
if host:
stats['hosts'][host] += 1 # count number of times a host is seen
is_success = False
if status:
stats['http_status_codes'][status] += 1 # count status code
# get_statuscode() returns a string, and a malformed record
# can put anything in the status line
if status.isdigit():
code = int(status)
if code >= 400:
stats['errors'].append({'url': uri, 'status': status})
elif code >= 300:
stats['redirects'].append({'url': uri, 'status': status})
else:
is_success = 200 <= code
# only count MIME types on 2xx: a redirect stub still carries a
# Content-Type but no real content, which inflates the counts
if is_success:
content_type = record.http_headers.get_header('Content-Type') # "text/html", etc
if content_type:
content_type = content_type.split(';')[0].strip()
stats['mime_types'][content_type] += 1 # count of number of MIME type
# convert defaultdict objects back to Python dicts
stats['record_types'] = dict(stats['record_types'])
stats['mime_types'] = dict(stats['mime_types'])
stats['http_status_codes'] = dict(stats['http_status_codes'])
stats['hosts'] = dict(stats['hosts'])
return stats
def main():
# set up cli argument parsing
parser = argparse.ArgumentParser(description='Get WARC info & stats') # create argument parser object
parser.add_argument('warc_file', help='Path to the WARC file') # required argument
parser.add_argument('-o', '--output', help='Output JSON file path') # optional argument
parser.add_argument('--log-dir', default='logs', help='Log directory path') # optional argument
parser.add_argument('-q', '--quiet', action='store_true',
help='Omit the errors/redirects lists, keeping only their counts')
args = parser.parse_args()
# set up logging
logger = setup_logging(args.log_dir) # return logger object
logger.info(f"Processing: {args.warc_file}")
try:
with open(args.warc_file, 'rb') as stream:
stats = collect_stats(stream)
except OSError as error:
parser.error(f"can't open '{args.warc_file}': {error.strerror}")
except ArchiveLoadFailed:
# the raw message quotes the offending bytes, which is unreadable
# binary for anything that isn't a WARC at all (a zip, a PDF, ...)
parser.error(f"'{args.warc_file}' is not a WARC file, or is corrupt")
if args.quiet: # swap the lists for their lengths
stats['errors'] = len(stats['errors'])
stats['redirects'] = len(stats['redirects'])
output = json.dumps(stats, indent=2) # json output
if args.output: # if user specified output
output_path = args.output
if os.path.isdir(output_path): # check if output is a dir
output_path = os.path.join(output_path, "warcstat_output.json") # create file inside that dir
with open(output_path, 'w') as f:
f.write(output) # write json output
logger.info(f"Output written to: {output_path}")
else:
print(output) # print json output to terminal if no output is defined
if __name__ == "__main__":
main()