Mercurial > hg > cc > cirrus_work
view lib/python/cc/lmh/warc2cdb.py @ 359:3a7604314cc0 plus
towards new fixed-length key and safer mapping of oob values
| author | Henry S. Thompson <ht@inf.ed.ac.uk> |
|---|---|
| date | Mon, 16 Mar 2026 19:07:10 +0000 |
| parents | af5ee299cbbe |
| children | 97e049b06f69 |
line wrap: on
line source
#!/usr/bin/env python3 # cython: profile=False, language_level=3str '''Produce cdb_input-style files from warc responses with lmh header value Usage: warc2cdb.py CC-date segment output-dir warc_file_type warc_file_range cdate_base warc_file_type is warc|robotstxt|crawldiagnostics warc_file_range is (literally) '???' or from,to cdate_base is the most likely complete (14-digit) warc-crawl-date''' import re, warc, sys, glob, codecs, os.path import cython, typing import email.utils from urllib.parse import quote import subprocess import shrink_key LMPAT: typing.Pattern[bytes] = re.compile(b'^Last-Modified: (.*?)\r?$',re.MULTILINE) FFPAT: typing.Pattern[bytes] = re.compile(b'([^ ])GMT$') OUT: typing.BinaryIO SEG: bytes R_T: bool = False WIN: int = 0 LOSE: int = 0 N: int = 0 UERRS: int = 0 NON_HTTP: int = 0 C_BASE: bytes C_BASE_L: int def _u_esc(c: int) -> str: if c<65536: return '\\u%04X'%c else: return '\\U%08X'%c def java_unicode_encode(ude: Type[UnicodeDecodeError]) -> tuple[str,int]: '''like backslashreplace but use uppercase and \\ u00NN instead of \\ xnn''' return (''.join(_u_esc(ord(c)) for c in ude.object[ude.start:ude.end]), ude.end) codecs.register_error('java_unicode',java_unicode_encode) def LMHline(wtype: int, buf: memoryview, part: int, brange: tuple[int, int]) -> None: global LMPAT, FFPAT, OUT, WIN, LOSE, NON_MONTH global N, UERRS, SEG, R_T, C_BASE, C_BASE_L global DATE, URI lmi: int offset: int m: typing.Match[cython.bytes] | None mm: typing.Match[cython.bytes] | None key: cython.bytes val: cython.bytes # Response if part == 1: offset = (brange[0] else: # HTTP headers mm=LMPAT.search(buf) if mm: N += 1 dateTime=mm[1] if dateTime.endswith(b'GMT'): if not dateTime.endswith(b' GMT'): dateTime = dateTime[:-3]+b' GMT' # FFPAT.sub(b'\\1 GMT',dateTime) try: try: lmi = int(email.utils.parsedate_to_datetime(dateTime.decode('utf8')).timestamp()) except OverflowError: lmi = 2147483647 except (TypeError,IndexError,ValueError) as e: print(dateTime.rstrip(),e,sep='\t',file=sys.stderr) LOSE += 1 return (key,val) = shrink_key.shrink_key(SEG, FILENO, offset, lmi) WIN += 1 l: int = len(LM) kl: int = (len(DATE)+len(URI)+(len(SEG) if R_T else 0)) OUT.write(b'+') OUT.write(b'%d'%kl) OUT.write(b',') OUT.write(b'%d'%l) OUT.write(b':') OUT.write(DATE) if R_T: OUT.write(SEG) OUT.write(URI) OUT.write(b'->') OUT.write(LM) OUT.write(b'\n') def main(segment: str, outdir: str, subdir: str, fpat: str ): global OUT, N, WIN, LOSE, SEG, R_T SEG = segment.encode('utf8') R_T = (subdir == 'robotstxt') if fpat != '???': fpat = ("{%s..%s}"%tuple(fpat.split(','))) if ',' in fpat else fpat infile_pat='bash -c "ls $CCC/CC-MAIN-%s/*.%s/orig/%s/*00%s.warc.gz | sort -k8"'%(CCdate, segment, subdir, fpat) with open((outfile_name:="%s/%s/%s/lmh.cdb_in"%(outdir, segment, subdir)),'wb') as OUT: for infile_name in subprocess.run(infile_pat, shell=True, stdout=subprocess.PIPE).stdout.decode('utf8').split(): print(infile_name,file=sys.stderr) WIN = LOSE = N = 0 if subdir in ['warc','robotstxt']: warc.warc(infile_name, LMHline, [warc.RESP], parts = 3, block = True) elif subdir == 'crawldiagnostics': warc.warc(infile_name, LMHline, [warc.RESP, warc.REVISIT], parts = 3, block = True) else: print('bogus type %s'%subdir, file=sys.stderr) exit(1) print('%d LM headers, %d win, %d lose, %d non-ASCII URIs, %d dodgy schemes'%(N, WIN,LOSE), file=sys.stderr) print(outfile_name) if __name__ == '__main__': sys.exit(main(*sys.argv[1:]))
