-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathtest_sample_contract.py
More file actions
91 lines (73 loc) · 3.36 KB
/
Copy pathtest_sample_contract.py
File metadata and controls
91 lines (73 loc) · 3.36 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
#!/usr/bin/env python3
"""
Quick test script to verify extraction patterns work with sample contract or not...
testing the core extraction functionality before starting the full system.....
"""
import sys
import os
sys.path.append(os.path.join(os.path.dirname(__file__), 'backend'))
from backend.app.services.pdf_processor import PDFProcessor
from backend.app.services.extraction_patterns import ContractPatterns
def test_sample_contract():
"""Test extraction on the sample contract"""
print("Testing Contract Intelligence Parser with sample contract...")
# Initialize processors
pdf_processor = PDFProcessor()
patterns = ContractPatterns.compile_patterns()
# Process the sample contract
contract_path = "sample_contract.pdf"
if not os.path.exists(contract_path):
print(f"Error: Sample contract not found at {contract_path}")
return
try:
# Extract text
print(f"\n1. Extracting text from {contract_path}...")
pages_data, ocr_used = pdf_processor.extract_text_from_pdf(contract_path)
print(f" - Pages extracted: {len(pages_data)}")
print(f" - OCR used: {ocr_used}")
# Combine text
full_text = "\n".join([page["content"] for page in pages_data])
print(f" - Total text length: {len(full_text)} characters")
# Test extraction patterns
print(f"\n2. Testing extraction patterns...")
extractions = {}
for field_name, pattern in patterns.items():
value, confidence, snippet = ContractPatterns.extract_field(full_text, pattern)
if value is not None:
extractions[field_name] = {
'value': value,
'confidence': confidence,
'snippet': snippet[:100] + '...' if len(snippet) > 100 else snippet
}
# Display results
print(f"\n3. Extraction Results ({len(extractions)} fields found):")
print("=" * 80)
for field_name, data in extractions.items():
print(f"\n{field_name.replace('_', ' ').title()}:")
print(f" Value: {data['value']}")
print(f" Confidence: {data['confidence']:.2f}")
print(f" Evidence: {data['snippet']}")
if not extractions:
print("No fields extracted. This might indicate:")
print("- The sample contract format is not recognized by current patterns")
print("- The PDF text extraction failed")
print("- The regex patterns need adjustment")
# Show first 500 characters of extracted text for debugging
print(f"\nFirst 500 characters of extracted text:")
print("-" * 50)
print(full_text[:500])
print("-" * 50)
print(f"\n4. Summary:")
print(f" - Contract processed successfully: {len(pages_data) > 0}")
print(f" - Fields extracted: {len(extractions)}")
print(f" - Average confidence: {sum(e['confidence'] for e in extractions.values()) / len(extractions):.2f if extractions else 0}")
return len(extractions) > 0
except Exception as e:
print(f"Error processing contract: {e}")
import traceback
traceback.print_exc()
return False
if __name__ == "__main__":
success = test_sample_contract()
print(f"\nTest {'PASSED' if success else 'FAILED'}")
sys.exit(0 if success else 1)