view lib/python/cc/lmh/warc2cdb.py @ 311:2ecf29fd1cee trim

CC has moved
author Henry S. Thompson <ht@inf.ed.ac.uk>
date Mon, 22 Dec 2025 22:19:54 +0000
parents 83c7ecd61ecf
children c68714dee9f2
line wrap: on
line source

#!/usr/bin/env python3
'''Produce cdb_input-style files from warc responses with lmh header value

   Usage: warc2cdb.py CC-date segment output-dir warc_file_type warc_file_range lm_shrink_prefix cdate_shrink_prefix '''

import re, warc, sys, glob, codecs, os.path
import cython, typing
import email.utils
from urllib.parse import quote
import subprocess

TUPAT: typing.Pattern[bytes] = re.compile(b'^WARC-Target-URI: (.*?)\r?$',re.MULTILINE)
DPAT: typing.Pattern[bytes] = re.compile(b'^WARC-Date: (.*?)\r?$',re.MULTILINE)
LMPAT: typing.Pattern[bytes] = re.compile(b'^Last-Modified: (.*?)\r?$',re.MULTILINE)
FFPAT: typing.Pattern[bytes] = re.compile(b'([^ ])GMT$')

DTAB: bytearray = bytearray(range(256))
DDEL: bytes = b'TZ-:'

OUT: typing.BinaryIO
SEG: bytes
R_T: bool = False

URI: bytes
DATE: bytes

WIN: int = 0
LOSE: int = 0
N: int = 0
UERRS: int = 0
NON_HTTP: int = 0
NON_MONTH: int = 0
NON_ERA: int = 0

def _u_esc(c):
  if c<65536:
    return '\\u%04X'%c
  else:
    return '\\U%08X'%c

def java_unicode_encode(ude):
  '''like backslashreplace but use uppercase and \\ u00NN instead of \\ xnn'''
  return (''.join(_u_esc(ord(c)) for c in ude.object[ude.start:ude.end]),
          ude.end)

codecs.register_error('java_unicode',java_unicode_encode)

def LMHline(wtype: int, buf: memoryview, part: int) -> None:
  global TUPAT, DPAT, LMPAT, FFPAT, DTAB, DDEL, OUT, WIN, LOSE, NON_HTTP, NON_MONTH
  global N, UERRS, SEG, R_T, LM_ERA, LM_ERA_L, C_MONTH, C_MONTH_L, NON_ERA
  global DATE, URI
  m: typing.Match[cython.bytes] | None
  mm: typing.Match[cython.bytes] | None
  lmi: cython.bytes
  if part==1:
    if (m:=TUPAT.search(buf)):
      URI=m[1]
    else:
      raise ValueError(b"No target URI in %s ??"%buf)
    if (md:=DPAT.search(buf)):
      DATE=md[1]
    else:
      raise ValueError(b"No date in %s ??"%buf)
  else:
    mm=LMPAT.search(buf)
    if mm:
      N += 1
      dateTime=mm[1]
      if dateTime.endswith(b'GMT'):
        if not dateTime.endswith(b' GMT'):
          dateTime = dateTime[:-3]+b' GMT' # FFPAT.sub(b'\\1 GMT',dateTime)
      try:
        try:
          lmi = b'%d'%int(email.utils.parsedate_to_datetime(dateTime.decode('utf8')).timestamp())
          if LM_ERA:
            if lmi.startswith(LM_ERA): # save 2 bytes in ~80% of cases
              lmi=b'0'+lmi[LM_ERA_L:]
            else:
              NON_ERA += 1
        except OverflowError:
          lmi = b'32535215999'
      except (TypeError,IndexError,ValueError) as e:
        print(dateTime.rstrip(),e,sep='\t',file=sys.stderr)
        LOSE += 1
        return
      DATE=(DATE.translate(DTAB,DDEL))
      if C_MONTH:
        if DATE.startswith(C_MONTH):
          DATE=DATE[C_MONTH_L:]
        else:
          NON_MONTH += 1
      WIN += 1
      try:
        URI.decode('ascii')
      except UnicodeDecodeError:
        UERRS += 1
        # Try just fixing the non-ASCII:
        URI = URI.decode('utf-8').encode('ascii', errors='java_unicode')
      # Could just assume http, but let's check
      if URI.startswith(b'http'):
        URI=URI[4:]
      else:
        NON_HTTP += 1
      l: int = len(lmi)
      kl: int = (len(DATE)+len(URI)+(len(SEG) if R_T else 0))
      OUT.write(b'+')
      OUT.write(b'%d'%kl)
      OUT.write(b',')
      OUT.write(b'%d'%l)
      OUT.write(b':')
      OUT.write(DATE)
      if R_T:
        OUT.write(SEG)
      OUT.write(URI)
      OUT.write(b'->')
      OUT.write(lmi)
      OUT.write(b'\n')

def main(CCdate, segment, outdir, subdir = 'warc', fpat = None, era = '', month = '' ):
  global OUT, N, WIN, LOSE, UERRS, SEG, R_T
  global NON_HTTP, C_MONTH, C_MONTH_L, NON_MONTH, LM_ERA, LM_ERA_L, NON_ERA

  SEG = segment.encode('utf8')
  R_T = (subdir == 'robotstxt')
  LM_ERA = era.encode('utf8')
  LM_ERA_L = len(era)
  C_MONTH = month.encode('utf8')
  C_MONTH_L = len(month)
  infile_pat='bash -c "ls $CCC/CC-MAIN-%s/*.%s/orig/%s/*00%s.warc.gz | sort -k8"'%(
    CCdate, segment, subdir, ("???" if fpat is None else (
      (("{%s..%s}"%tuple(fpat.split(','))) if ',' in fpat else fpat))))
  
  with open((outfile_name:="%s/%s/%s/lmh.cdb_in"%(outdir, segment, subdir)),'wb') as OUT:
    for infile_name in subprocess.run(infile_pat, shell=True,
                                   stdout=subprocess.PIPE).stdout.decode('utf8').split():
      print(infile_name,file=sys.stderr)
      WIN = LOSE = N = UERRS = NON_HTTP = NON_MONTH = NON_ERA = 0
      if subdir in ['warc','robotstxt']:
        warc.warc(infile_name,LMHline,[warc.RESP],parts=3)
      elif subdir == 'crawldiagnostics':
        warc.warc(infile_name,LMHline,[warc.RESP, warc.REVISIT],parts=3)
      else:
        print('bogus type %s'%subdir,file=sys.stderr)
        exit(1)
      print('%d LM headers, %d win, %d lose, %d non-ASCII URIs, %d non-close, %d dodgy schemes, %d dodgy WARC dates'%(N,WIN,LOSE,UERRS,NON_ERA,NON_HTTP,NON_MONTH),
                                                                 file=sys.stderr)
    OUT.write(b'\n')

  print(outfile_name)

if __name__ == '__main__':
  sys.exit(main(*sys.argv[1:]))