Mercurial > hg > cc > cirrus_work
view lib/python/cc/lmh/warc2cdb.py @ 311:2ecf29fd1cee trim
CC has moved
| author | Henry S. Thompson <ht@inf.ed.ac.uk> |
|---|---|
| date | Mon, 22 Dec 2025 22:19:54 +0000 |
| parents | 83c7ecd61ecf |
| children | c68714dee9f2 |
line wrap: on
line source
#!/usr/bin/env python3 '''Produce cdb_input-style files from warc responses with lmh header value Usage: warc2cdb.py CC-date segment output-dir warc_file_type warc_file_range lm_shrink_prefix cdate_shrink_prefix ''' import re, warc, sys, glob, codecs, os.path import cython, typing import email.utils from urllib.parse import quote import subprocess TUPAT: typing.Pattern[bytes] = re.compile(b'^WARC-Target-URI: (.*?)\r?$',re.MULTILINE) DPAT: typing.Pattern[bytes] = re.compile(b'^WARC-Date: (.*?)\r?$',re.MULTILINE) LMPAT: typing.Pattern[bytes] = re.compile(b'^Last-Modified: (.*?)\r?$',re.MULTILINE) FFPAT: typing.Pattern[bytes] = re.compile(b'([^ ])GMT$') DTAB: bytearray = bytearray(range(256)) DDEL: bytes = b'TZ-:' OUT: typing.BinaryIO SEG: bytes R_T: bool = False URI: bytes DATE: bytes WIN: int = 0 LOSE: int = 0 N: int = 0 UERRS: int = 0 NON_HTTP: int = 0 NON_MONTH: int = 0 NON_ERA: int = 0 def _u_esc(c): if c<65536: return '\\u%04X'%c else: return '\\U%08X'%c def java_unicode_encode(ude): '''like backslashreplace but use uppercase and \\ u00NN instead of \\ xnn''' return (''.join(_u_esc(ord(c)) for c in ude.object[ude.start:ude.end]), ude.end) codecs.register_error('java_unicode',java_unicode_encode) def LMHline(wtype: int, buf: memoryview, part: int) -> None: global TUPAT, DPAT, LMPAT, FFPAT, DTAB, DDEL, OUT, WIN, LOSE, NON_HTTP, NON_MONTH global N, UERRS, SEG, R_T, LM_ERA, LM_ERA_L, C_MONTH, C_MONTH_L, NON_ERA global DATE, URI m: typing.Match[cython.bytes] | None mm: typing.Match[cython.bytes] | None lmi: cython.bytes if part==1: if (m:=TUPAT.search(buf)): URI=m[1] else: raise ValueError(b"No target URI in %s ??"%buf) if (md:=DPAT.search(buf)): DATE=md[1] else: raise ValueError(b"No date in %s ??"%buf) else: mm=LMPAT.search(buf) if mm: N += 1 dateTime=mm[1] if dateTime.endswith(b'GMT'): if not dateTime.endswith(b' GMT'): dateTime = dateTime[:-3]+b' GMT' # FFPAT.sub(b'\\1 GMT',dateTime) try: try: lmi = b'%d'%int(email.utils.parsedate_to_datetime(dateTime.decode('utf8')).timestamp()) if LM_ERA: if lmi.startswith(LM_ERA): # save 2 bytes in ~80% of cases lmi=b'0'+lmi[LM_ERA_L:] else: NON_ERA += 1 except OverflowError: lmi = b'32535215999' except (TypeError,IndexError,ValueError) as e: print(dateTime.rstrip(),e,sep='\t',file=sys.stderr) LOSE += 1 return DATE=(DATE.translate(DTAB,DDEL)) if C_MONTH: if DATE.startswith(C_MONTH): DATE=DATE[C_MONTH_L:] else: NON_MONTH += 1 WIN += 1 try: URI.decode('ascii') except UnicodeDecodeError: UERRS += 1 # Try just fixing the non-ASCII: URI = URI.decode('utf-8').encode('ascii', errors='java_unicode') # Could just assume http, but let's check if URI.startswith(b'http'): URI=URI[4:] else: NON_HTTP += 1 l: int = len(lmi) kl: int = (len(DATE)+len(URI)+(len(SEG) if R_T else 0)) OUT.write(b'+') OUT.write(b'%d'%kl) OUT.write(b',') OUT.write(b'%d'%l) OUT.write(b':') OUT.write(DATE) if R_T: OUT.write(SEG) OUT.write(URI) OUT.write(b'->') OUT.write(lmi) OUT.write(b'\n') def main(CCdate, segment, outdir, subdir = 'warc', fpat = None, era = '', month = '' ): global OUT, N, WIN, LOSE, UERRS, SEG, R_T global NON_HTTP, C_MONTH, C_MONTH_L, NON_MONTH, LM_ERA, LM_ERA_L, NON_ERA SEG = segment.encode('utf8') R_T = (subdir == 'robotstxt') LM_ERA = era.encode('utf8') LM_ERA_L = len(era) C_MONTH = month.encode('utf8') C_MONTH_L = len(month) infile_pat='bash -c "ls $CCC/CC-MAIN-%s/*.%s/orig/%s/*00%s.warc.gz | sort -k8"'%( CCdate, segment, subdir, ("???" if fpat is None else ( (("{%s..%s}"%tuple(fpat.split(','))) if ',' in fpat else fpat)))) with open((outfile_name:="%s/%s/%s/lmh.cdb_in"%(outdir, segment, subdir)),'wb') as OUT: for infile_name in subprocess.run(infile_pat, shell=True, stdout=subprocess.PIPE).stdout.decode('utf8').split(): print(infile_name,file=sys.stderr) WIN = LOSE = N = UERRS = NON_HTTP = NON_MONTH = NON_ERA = 0 if subdir in ['warc','robotstxt']: warc.warc(infile_name,LMHline,[warc.RESP],parts=3) elif subdir == 'crawldiagnostics': warc.warc(infile_name,LMHline,[warc.RESP, warc.REVISIT],parts=3) else: print('bogus type %s'%subdir,file=sys.stderr) exit(1) print('%d LM headers, %d win, %d lose, %d non-ASCII URIs, %d non-close, %d dodgy schemes, %d dodgy WARC dates'%(N,WIN,LOSE,UERRS,NON_ERA,NON_HTTP,NON_MONTH), file=sys.stderr) OUT.write(b'\n') print(outfile_name) if __name__ == '__main__': sys.exit(main(*sys.argv[1:]))
