diff --git a/clean_chunk_pol177.py b/clean_chunk_pol177.py new file mode 100644 index 0000000..4e57b6e --- /dev/null +++ b/clean_chunk_pol177.py @@ -0,0 +1,268 @@ +#!/usr/bin/env python3 +""" +Clean and chunk parsed PDF data from unstructured-api. +Filters noise, preserves structure, chunks to 1500-1800 chars. +""" + +import json +import re + +def load_parsed_data(path): + with open(path, 'r', encoding='utf-8') as f: + return json.load(f) + +def clean_text(text): + if not text: + return "" + text = re.sub(r'\s+', ' ', text.strip()) + return text.strip() + +def is_noise(text): + """Check if text is noise.""" + text = clean_text(text) + if not text: + return True + + # Company name header/footer + if text == "АО «ХК «Сибцем»": + return True + + # Page numbers: "Стр. 2 из 10" + if re.match(r'^Стр\.\s+\d+\s+из\s+\d+$', text): + return True + + # Standalone numbers: "4." + if re.match(r'^\d+\.$', text): + return True + + # Table of contents + lines = text.split('\n') + if len(lines) > 3: + page_pattern = re.compile(r'\d+\s*$') + page_lines = sum(1 for l in lines if page_pattern.search(l.strip())) + if page_lines > len(lines) * 0.5: + return True + + # Specific noise + noise_patterns = [ + r'^Оглавление$', + r'^Содержание$', + r'^Редакция \d+$', + r'^Тип документа:', + r'^Наименование процесса:', + r'^Ведущее подразделение:', + r'^Дата утверждения:', + ] + for pattern in noise_patterns: + if re.search(pattern, text): + return True + + return False + +def is_section_title(text): + """Check if text is a section title like '4. Термины, определения и сокращения'.""" + text = clean_text(text) + return bool(re.match(r'^\d+\.\s+\w', text)) + +def is_section_subtitle(text): + """Check if text is a section subtitle like '6.7. Электронные...'.""" + text = clean_text(text) + return bool(re.match(r'^\d+\.\d+\.\s+\w', text)) + +def is_section_header(text): + """Check if text starts with a section/subsection number.""" + text = clean_text(text) + return bool(re.match(r'^\d+(?:\.\d+)*[\.\s]', text)) + +def merge_elements(elements): + """ + Merge short fragments with context. + Strategy: merge consecutive non-title elements into text blocks. + Keep titles and tables as separate blocks. + """ + if not elements: + return [] + + # Filter noise first + filtered = [el for el in elements if not is_noise(el.get('text', ''))] + + # Separate into blocks: text blocks, title blocks, table blocks + blocks = [] + current_text_block = [] + + for el in filtered: + text = el.get('text', '') + el_type = el.get('type', '') + + # Skip Image (OCR noise from cover page) + if el_type == 'Image': + continue + + # If this is a title/section header, flush text block and add title + if el_type in ('Title',) and is_section_header(text): + if current_text_block: + blocks.append({ + 'text': ' '.join(clean_text(e['text']) for e in current_text_block), + 'type': 'NarrativeText', + 'metadata': current_text_block[-1].get('metadata', {}) + }) + current_text_block = [] + blocks.append({ + 'text': text, + 'type': 'Title', + 'metadata': el.get('metadata', {}) + }) + continue + + # If this is a table, flush text block and add table + if el_type == 'Table': + if current_text_block: + blocks.append({ + 'text': ' '.join(clean_text(e['text']) for e in current_text_block), + 'type': 'NarrativeText', + 'metadata': current_text_block[-1].get('metadata', {}) + }) + current_text_block = [] + blocks.append({ + 'text': text, + 'type': 'Table', + 'metadata': el.get('metadata', {}) + }) + continue + + # Otherwise, accumulate into text block + current_text_block.append(el) + + # Flush remaining text block + if current_text_block: + blocks.append({ + 'text': ' '.join(clean_text(e['text']) for e in current_text_block), + 'type': 'NarrativeText', + 'metadata': current_text_block[-1].get('metadata', {}) + }) + + # Now merge short text blocks with adjacent content + # Specifically: merge short blocks that are continuations of previous sections + result = [] + for i, block in enumerate(blocks): + text = block['text'] + el_type = block['type'] + clean = clean_text(text) + + # If this is a short text block, try to merge with previous + if len(clean) < 60 and el_type == 'NarrativeText' and result: + # Check if it looks like a continuation (section subtitle or fragment) + if is_section_subtitle(clean) or not clean[0].isdigit(): + # Merge with previous block + result[-1]['text'] = f"{result[-1]['text']} {clean}" + continue + + result.append(block) + + return result + +def chunk_text(text, max_chunk_size=1800): + """Split text into chunks of ~1500-1800 characters.""" + if len(text) <= max_chunk_size: + return [text] + + # Try to split by paragraphs first + paragraphs = re.split(r'\n+', text) + if len(paragraphs) > 1: + chunks = [] + current = "" + for para in paragraphs: + para = para.strip() + if not para: + continue + if len(current) + len(para) + 1 <= max_chunk_size: + current = f"{current}\n{para}" + else: + if current: + chunks.append(current) + current = para + if current: + chunks.append(current) + return chunks + + # Split by sentences + sentences = re.split(r'(?<=[.!?])\s+', text) + chunks = [] + current = "" + for sentence in sentences: + if len(current) + len(sentence) + 1 <= max_chunk_size: + current = f"{current} {sentence}" + else: + if current: + chunks.append(current) + current = sentence + if current: + chunks.append(current) + return chunks + +def process_elements(elements, max_chunk_size=1800): + """Process elements into chunks.""" + # Merge blocks + merged = merge_elements(elements) + + # Create chunks + chunks = [] + chunk_index = 0 + + for el in merged: + text = el.get('text', '') + el_type = el.get('type', 'Unknown') + page = el.get('metadata', {}).get('page_number', '?') + + # Skip very short chunks + if len(text.strip()) < 10: + continue + + # Split long texts + parts = chunk_text(text, max_chunk_size) + + for part in parts: + chunk_index += 1 + chunks.append({ + 'index': chunk_index, + 'type': el_type, + 'page': page, + 'text': part, + 'size': len(part) + }) + + return chunks, merged + +def main(): + parsed_path = '/tmp/pol177_parsed.json' + output_path = '/tmp/pol177_clean_chunk.json' + + print("Loading parsed data...") + elements = load_parsed_data(parsed_path) + print(f"Loaded {len(elements)} elements") + + print("Processing...") + chunks, merged = process_elements(elements) + + print(f"Generated {len(chunks)} chunks") + + # Stats + if chunks: + sizes = [c['size'] for c in chunks] + print(f"Chunk sizes: min={min(sizes)}, max={max(sizes)}, avg={sum(sizes)/len(sizes):.0f}") + + # Type distribution + types = {} + for c in chunks: + types[c['type']] = types.get(c['type'], 0) + 1 + print(f"Chunk types: {types}") + + # Show all chunks + print("\n=== All chunks ===") + for c in chunks: + print(f"[{c['index']}] {c['type']} (page {c['page']}, {c['size']} chars)") + print(f" {c['text'][:120]}") + print() + +if __name__ == '__main__': + main()