Add clean_chunk_pol177.py - Clean and chunk parsed PDF data
This commit is contained in:
@@ -0,0 +1,268 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""
|
||||||
|
Clean and chunk parsed PDF data from unstructured-api.
|
||||||
|
Filters noise, preserves structure, chunks to 1500-1800 chars.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import json
|
||||||
|
import re
|
||||||
|
|
||||||
|
def load_parsed_data(path):
|
||||||
|
with open(path, 'r', encoding='utf-8') as f:
|
||||||
|
return json.load(f)
|
||||||
|
|
||||||
|
def clean_text(text):
|
||||||
|
if not text:
|
||||||
|
return ""
|
||||||
|
text = re.sub(r'\s+', ' ', text.strip())
|
||||||
|
return text.strip()
|
||||||
|
|
||||||
|
def is_noise(text):
|
||||||
|
"""Check if text is noise."""
|
||||||
|
text = clean_text(text)
|
||||||
|
if not text:
|
||||||
|
return True
|
||||||
|
|
||||||
|
# Company name header/footer
|
||||||
|
if text == "АО «ХК «Сибцем»":
|
||||||
|
return True
|
||||||
|
|
||||||
|
# Page numbers: "Стр. 2 из 10"
|
||||||
|
if re.match(r'^Стр\.\s+\d+\s+из\s+\d+$', text):
|
||||||
|
return True
|
||||||
|
|
||||||
|
# Standalone numbers: "4."
|
||||||
|
if re.match(r'^\d+\.$', text):
|
||||||
|
return True
|
||||||
|
|
||||||
|
# Table of contents
|
||||||
|
lines = text.split('\n')
|
||||||
|
if len(lines) > 3:
|
||||||
|
page_pattern = re.compile(r'\d+\s*$')
|
||||||
|
page_lines = sum(1 for l in lines if page_pattern.search(l.strip()))
|
||||||
|
if page_lines > len(lines) * 0.5:
|
||||||
|
return True
|
||||||
|
|
||||||
|
# Specific noise
|
||||||
|
noise_patterns = [
|
||||||
|
r'^Оглавление$',
|
||||||
|
r'^Содержание$',
|
||||||
|
r'^Редакция \d+$',
|
||||||
|
r'^Тип документа:',
|
||||||
|
r'^Наименование процесса:',
|
||||||
|
r'^Ведущее подразделение:',
|
||||||
|
r'^Дата утверждения:',
|
||||||
|
]
|
||||||
|
for pattern in noise_patterns:
|
||||||
|
if re.search(pattern, text):
|
||||||
|
return True
|
||||||
|
|
||||||
|
return False
|
||||||
|
|
||||||
|
def is_section_title(text):
|
||||||
|
"""Check if text is a section title like '4. Термины, определения и сокращения'."""
|
||||||
|
text = clean_text(text)
|
||||||
|
return bool(re.match(r'^\d+\.\s+\w', text))
|
||||||
|
|
||||||
|
def is_section_subtitle(text):
|
||||||
|
"""Check if text is a section subtitle like '6.7. Электронные...'."""
|
||||||
|
text = clean_text(text)
|
||||||
|
return bool(re.match(r'^\d+\.\d+\.\s+\w', text))
|
||||||
|
|
||||||
|
def is_section_header(text):
|
||||||
|
"""Check if text starts with a section/subsection number."""
|
||||||
|
text = clean_text(text)
|
||||||
|
return bool(re.match(r'^\d+(?:\.\d+)*[\.\s]', text))
|
||||||
|
|
||||||
|
def merge_elements(elements):
|
||||||
|
"""
|
||||||
|
Merge short fragments with context.
|
||||||
|
Strategy: merge consecutive non-title elements into text blocks.
|
||||||
|
Keep titles and tables as separate blocks.
|
||||||
|
"""
|
||||||
|
if not elements:
|
||||||
|
return []
|
||||||
|
|
||||||
|
# Filter noise first
|
||||||
|
filtered = [el for el in elements if not is_noise(el.get('text', ''))]
|
||||||
|
|
||||||
|
# Separate into blocks: text blocks, title blocks, table blocks
|
||||||
|
blocks = []
|
||||||
|
current_text_block = []
|
||||||
|
|
||||||
|
for el in filtered:
|
||||||
|
text = el.get('text', '')
|
||||||
|
el_type = el.get('type', '')
|
||||||
|
|
||||||
|
# Skip Image (OCR noise from cover page)
|
||||||
|
if el_type == 'Image':
|
||||||
|
continue
|
||||||
|
|
||||||
|
# If this is a title/section header, flush text block and add title
|
||||||
|
if el_type in ('Title',) and is_section_header(text):
|
||||||
|
if current_text_block:
|
||||||
|
blocks.append({
|
||||||
|
'text': ' '.join(clean_text(e['text']) for e in current_text_block),
|
||||||
|
'type': 'NarrativeText',
|
||||||
|
'metadata': current_text_block[-1].get('metadata', {})
|
||||||
|
})
|
||||||
|
current_text_block = []
|
||||||
|
blocks.append({
|
||||||
|
'text': text,
|
||||||
|
'type': 'Title',
|
||||||
|
'metadata': el.get('metadata', {})
|
||||||
|
})
|
||||||
|
continue
|
||||||
|
|
||||||
|
# If this is a table, flush text block and add table
|
||||||
|
if el_type == 'Table':
|
||||||
|
if current_text_block:
|
||||||
|
blocks.append({
|
||||||
|
'text': ' '.join(clean_text(e['text']) for e in current_text_block),
|
||||||
|
'type': 'NarrativeText',
|
||||||
|
'metadata': current_text_block[-1].get('metadata', {})
|
||||||
|
})
|
||||||
|
current_text_block = []
|
||||||
|
blocks.append({
|
||||||
|
'text': text,
|
||||||
|
'type': 'Table',
|
||||||
|
'metadata': el.get('metadata', {})
|
||||||
|
})
|
||||||
|
continue
|
||||||
|
|
||||||
|
# Otherwise, accumulate into text block
|
||||||
|
current_text_block.append(el)
|
||||||
|
|
||||||
|
# Flush remaining text block
|
||||||
|
if current_text_block:
|
||||||
|
blocks.append({
|
||||||
|
'text': ' '.join(clean_text(e['text']) for e in current_text_block),
|
||||||
|
'type': 'NarrativeText',
|
||||||
|
'metadata': current_text_block[-1].get('metadata', {})
|
||||||
|
})
|
||||||
|
|
||||||
|
# Now merge short text blocks with adjacent content
|
||||||
|
# Specifically: merge short blocks that are continuations of previous sections
|
||||||
|
result = []
|
||||||
|
for i, block in enumerate(blocks):
|
||||||
|
text = block['text']
|
||||||
|
el_type = block['type']
|
||||||
|
clean = clean_text(text)
|
||||||
|
|
||||||
|
# If this is a short text block, try to merge with previous
|
||||||
|
if len(clean) < 60 and el_type == 'NarrativeText' and result:
|
||||||
|
# Check if it looks like a continuation (section subtitle or fragment)
|
||||||
|
if is_section_subtitle(clean) or not clean[0].isdigit():
|
||||||
|
# Merge with previous block
|
||||||
|
result[-1]['text'] = f"{result[-1]['text']} {clean}"
|
||||||
|
continue
|
||||||
|
|
||||||
|
result.append(block)
|
||||||
|
|
||||||
|
return result
|
||||||
|
|
||||||
|
def chunk_text(text, max_chunk_size=1800):
|
||||||
|
"""Split text into chunks of ~1500-1800 characters."""
|
||||||
|
if len(text) <= max_chunk_size:
|
||||||
|
return [text]
|
||||||
|
|
||||||
|
# Try to split by paragraphs first
|
||||||
|
paragraphs = re.split(r'\n+', text)
|
||||||
|
if len(paragraphs) > 1:
|
||||||
|
chunks = []
|
||||||
|
current = ""
|
||||||
|
for para in paragraphs:
|
||||||
|
para = para.strip()
|
||||||
|
if not para:
|
||||||
|
continue
|
||||||
|
if len(current) + len(para) + 1 <= max_chunk_size:
|
||||||
|
current = f"{current}\n{para}"
|
||||||
|
else:
|
||||||
|
if current:
|
||||||
|
chunks.append(current)
|
||||||
|
current = para
|
||||||
|
if current:
|
||||||
|
chunks.append(current)
|
||||||
|
return chunks
|
||||||
|
|
||||||
|
# Split by sentences
|
||||||
|
sentences = re.split(r'(?<=[.!?])\s+', text)
|
||||||
|
chunks = []
|
||||||
|
current = ""
|
||||||
|
for sentence in sentences:
|
||||||
|
if len(current) + len(sentence) + 1 <= max_chunk_size:
|
||||||
|
current = f"{current} {sentence}"
|
||||||
|
else:
|
||||||
|
if current:
|
||||||
|
chunks.append(current)
|
||||||
|
current = sentence
|
||||||
|
if current:
|
||||||
|
chunks.append(current)
|
||||||
|
return chunks
|
||||||
|
|
||||||
|
def process_elements(elements, max_chunk_size=1800):
|
||||||
|
"""Process elements into chunks."""
|
||||||
|
# Merge blocks
|
||||||
|
merged = merge_elements(elements)
|
||||||
|
|
||||||
|
# Create chunks
|
||||||
|
chunks = []
|
||||||
|
chunk_index = 0
|
||||||
|
|
||||||
|
for el in merged:
|
||||||
|
text = el.get('text', '')
|
||||||
|
el_type = el.get('type', 'Unknown')
|
||||||
|
page = el.get('metadata', {}).get('page_number', '?')
|
||||||
|
|
||||||
|
# Skip very short chunks
|
||||||
|
if len(text.strip()) < 10:
|
||||||
|
continue
|
||||||
|
|
||||||
|
# Split long texts
|
||||||
|
parts = chunk_text(text, max_chunk_size)
|
||||||
|
|
||||||
|
for part in parts:
|
||||||
|
chunk_index += 1
|
||||||
|
chunks.append({
|
||||||
|
'index': chunk_index,
|
||||||
|
'type': el_type,
|
||||||
|
'page': page,
|
||||||
|
'text': part,
|
||||||
|
'size': len(part)
|
||||||
|
})
|
||||||
|
|
||||||
|
return chunks, merged
|
||||||
|
|
||||||
|
def main():
|
||||||
|
parsed_path = '/tmp/pol177_parsed.json'
|
||||||
|
output_path = '/tmp/pol177_clean_chunk.json'
|
||||||
|
|
||||||
|
print("Loading parsed data...")
|
||||||
|
elements = load_parsed_data(parsed_path)
|
||||||
|
print(f"Loaded {len(elements)} elements")
|
||||||
|
|
||||||
|
print("Processing...")
|
||||||
|
chunks, merged = process_elements(elements)
|
||||||
|
|
||||||
|
print(f"Generated {len(chunks)} chunks")
|
||||||
|
|
||||||
|
# Stats
|
||||||
|
if chunks:
|
||||||
|
sizes = [c['size'] for c in chunks]
|
||||||
|
print(f"Chunk sizes: min={min(sizes)}, max={max(sizes)}, avg={sum(sizes)/len(sizes):.0f}")
|
||||||
|
|
||||||
|
# Type distribution
|
||||||
|
types = {}
|
||||||
|
for c in chunks:
|
||||||
|
types[c['type']] = types.get(c['type'], 0) + 1
|
||||||
|
print(f"Chunk types: {types}")
|
||||||
|
|
||||||
|
# Show all chunks
|
||||||
|
print("\n=== All chunks ===")
|
||||||
|
for c in chunks:
|
||||||
|
print(f"[{c['index']}] {c['type']} (page {c['page']}, {c['size']} chars)")
|
||||||
|
print(f" {c['text'][:120]}")
|
||||||
|
print()
|
||||||
|
|
||||||
|
if __name__ == '__main__':
|
||||||
|
main()
|
||||||
Reference in New Issue
Block a user