"""Bounded local extraction. Run without network and on ephemeral /tmp storage."""
import base64, csv, io, json, os, pathlib, resource, subprocess, sys, tempfile, zipfile
import xml.etree.ElementTree as ET
resource.setrlimit(resource.RLIMIT_CPU, (100, 100))
resource.setrlimit(resource.RLIMIT_AS, (3*1024**3, 3*1024**3))
resource.setrlimit(resource.RLIMIT_FSIZE, (32*1024*1024, 32*1024*1024))
resource.setrlimit(resource.RLIMIT_NOFILE, (128, 128))
resource.setrlimit(resource.RLIMIT_CORE, (0, 0))
LIMIT=16384

def run(args, timeout=25):
    return subprocess.run(args, check=True, stdout=subprocess.PIPE, stderr=subprocess.DEVNULL, timeout=timeout).stdout

def bounded(text):
    if not text.strip() or len(text.encode('utf-8'))>LIMIT or '\x00' in text: raise ValueError('Uninspectable content')
    return text

def ocr(path):
    from PIL import Image
    Image.MAX_IMAGE_PIXELS=12000000
    with Image.open(path) as image:
        if image.width*image.height>12000000 or getattr(image,'n_frames',1)!=1: raise ValueError('Image limit')
        image.verify()
    rows=csv.DictReader(io.StringIO(run(['tesseract',str(path),'stdout','-l','eng','--psm','11','tsv']).decode('utf-8')),delimiter='\t')
    words=[]
    for row in rows:
        if row.get('text','').strip():
            if float(row['conf'])<65: raise ValueError('Uncertain OCR')
            words.append(row['text'])
    return bounded(' '.join(words))

def extract(mime,data,tmp,index):
    path=tmp/('input'+str(index));path.write_bytes(data)
    if mime in ['text/plain','text/csv','application/json']:
        return bounded(data.decode('utf-8','strict'))
    if mime in ['image/png','image/jpeg']:
        if (mime=='image/png' and not data.startswith(b'\x89PNG\r\n\x1a\n')) or (mime=='image/jpeg' and not data.startswith(b'\xff\xd8\xff')):raise ValueError('Type mismatch')
        return ocr(path)
    if mime=='application/pdf':
        if not data.startswith(b'%PDF-'):raise ValueError('Type mismatch')
        info=run(['pdfinfo',str(path)]).decode('utf-8');pages=None
        for line in info.splitlines():
            if line.startswith('Pages:'):pages=int(line.split(':')[1])
            if line.startswith('Encrypted:') and 'yes' in line:raise ValueError('Encrypted PDF')
        if pages is None or not 1<=pages<=10:raise ValueError('Page limit')
        selectable=run(['pdftotext','-enc','UTF-8',str(path),'-']).decode('utf-8')
        if len(selectable.encode())>LIMIT:raise ValueError('Text limit')
        prefix=tmp/('page'+str(index))
        run(['pdftoppm','-r','100','-scale-to','2000','-png',str(path),str(prefix)],40)
        texts=[selectable]
        for image in sorted(tmp.glob(prefix.name+'-*.png')):texts.append(ocr(image))
        return bounded('\n'.join(texts))
    if mime=='application/vnd.openxmlformats-officedocument.wordprocessingml.document':
        texts=[]
        with zipfile.ZipFile(path) as z:
            entries=z.infolist()
            if len(entries)>150 or sum(x.file_size for x in entries)>20*1024*1024 or any(x.flag_bits&1 for x in entries):raise ValueError('Archive limit')
            if 'word/document.xml' not in z.namelist():raise ValueError('Not DOCX')
            for entry in entries:
                name=entry.filename
                if any(x in name.lower() for x in ['vbaproject','embeddings/','activex/']):raise ValueError('Embedded executable')
                if name.endswith('.xml') or name.endswith('.rels'):
                    raw=z.read(entry)
                    if b'<!DOCTYPE' in raw.upper() or b'<!ENTITY' in raw.upper():raise ValueError('XML declaration')
                    tree=ET.fromstring(raw)
                    if any(el.attrib.get('TargetMode')=='External' for el in tree.iter()):raise ValueError('External content')
                    if name.startswith('word/'):texts.extend(el.text for el in tree.iter() if el.tag.endswith('}t') and el.text)
                elif name.startswith('word/media/'):
                    image=tmp/('docx-image-'+str(len(texts)));image.write_bytes(z.read(entry));texts.append(ocr(image))
                elif name.startswith('word/') and not name.endswith('/'):
                    raise ValueError('Unsupported document part')
        return bounded('\n'.join(texts))
    if mime in ['audio/wav','audio/mpeg','audio/mp4','audio/flac','audio/ogg']:
        signatures={'audio/wav':data.startswith(b'RIFF') and data[8:12]==b'WAVE', 'audio/mpeg':data.startswith(b'ID3') or (len(data)>2 and data[0]==255 and data[1]&224==224), 'audio/mp4':data[4:8]==b'ftyp', 'audio/flac':data.startswith(b'fLaC'), 'audio/ogg':data.startswith(b'OggS')}
        if not signatures[mime]:raise ValueError('Audio type mismatch')
        # ffprobe/ffmpeg may only open local files, never URL protocols or playlists.
        info=json.loads(run(['ffprobe','-v','error','-protocol_whitelist','file,pipe','-show_format','-show_streams','-of','json',str(path)]))
        formats=set(info.get('format',{}).get('format_name','').split(','))
        allowed={'audio/wav':{'wav'},'audio/mpeg':{'mp3'},'audio/mp4':{'mov','mp4','m4a','3gp','3g2','mj2'},'audio/flac':{'flac'},'audio/ogg':{'ogg'}}
        duration=float(info['format']['duration'])
        if not formats.intersection(allowed[mime]) or not 0<duration<=60 or any(s.get('codec_type')!='audio' for s in info.get('streams',[])) or len(info.get('streams',[]))!=1:raise ValueError('Audio limit')
        wav=tmp/('normalized-'+str(index)+'.wav')
        run(['ffmpeg','-v','error','-protocol_whitelist','file,pipe','-i',str(path),'-t','60','-ac','1','-ar','16000','-f','wav',str(wav)])
        from faster_whisper import WhisperModel
        model_path=os.environ.get('GUARD_WHISPER_MODEL_DIR','')
        if not model_path or not pathlib.Path(model_path).is_dir():raise ValueError('Local model unavailable')
        model=WhisperModel(model_path,device='cpu',compute_type='int8',cpu_threads=2,num_workers=1,local_files_only=True)
        segments,details=model.transcribe(str(wav),language='en',beam_size=5,condition_on_previous_text=False,vad_filter=False)
        texts=[]
        for segment in segments:
            if segment.avg_logprob < -0.7 or segment.no_speech_prob>0.5:raise ValueError('Uncertain transcript')
            texts.append(segment.text)
        return bounded(' '.join(texts))
    raise ValueError('Unsupported attachment')

try:
    raw=sys.stdin.buffer.read(6000001)
    if len(raw)>6000000:raise ValueError('Body limit')
    items=json.loads(raw)
    if not isinstance(items,list) or not 1<=len(items)<=4:raise ValueError('Attachment count')
    results=[];total=0
    with tempfile.TemporaryDirectory(prefix='guard-content-') as directory:
        for i,item in enumerate(items):
            if set(item)!= {'mimeType','data'}:raise ValueError('Attachment schema')
            data=base64.b64decode(item['data'],validate=True);total+=len(data)
            if not data or total>4*1024*1024:raise ValueError('Attachment size')
            results.append(extract(item['mimeType'],data,pathlib.Path(directory),i))
        bounded('\n'.join(results))
    print(json.dumps({'texts':results}))
except Exception:
    # Do not print parser exceptions, content, filenames or model traces.
    print(json.dumps({'error':'Content could not be inspected safely'}));sys.exit(1)
