-
Notifications
You must be signed in to change notification settings - Fork 3
Expand file tree
/
Copy pathmain.py
More file actions
125 lines (101 loc) · 4.81 KB
/
Copy pathmain.py
File metadata and controls
125 lines (101 loc) · 4.81 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
"""
Google LangExtract Financial Document Extraction & Source Grounding Pipeline
===========================================================================
Extracts structured information from financial documents using local or cloud LLMs
with exact character offset source grounding.
"""
import os
import sys
import time
import json
from typing import Dict, List, Any
# pyrefly: ignore [missing-import]
import langextract as le
# =====================================================================
# CONFIGURATION PARAMETERS
# =====================================================================
OLLAMA_MODEL = "qwen3:4b"
MODEL_PROVIDER = "ollama" # Options: "ollama", "gemini", "openai"
DISABLE_THINKING = True
INPUT_DOCUMENT = "NVIDIA_Q4_FY25_Financial_Report.txt"
OUTPUT_REPORT = "outputs.md"
# 5 Natural One-Liner Questions
EXTRACTION_QUERIES = [
"What was NVIDIA's total revenue for Q4 and year-over-year growth?",
"What was NVIDIA's gross profit margin percentage and net profit income?",
"How much did NVIDIA spend on Research and Development (R&D)?",
"What are NVIDIA's cash reserves, debt, and free cash flow?",
"What is NVIDIA's revenue guidance for Q1 FY26 and key operational risks?"
]
def load_document_content(filepath: str) -> str:
"""Loads document text directly from the specified file path."""
if not os.path.exists(filepath):
raise FileNotFoundError(f"Input document '{filepath}' not found in current directory.")
with open(filepath, "r", encoding="utf-8") as f:
return f.read()
def extract_with_langextract(text: str, question: str, provider: str, model_name: str) -> Dict[str, Any]:
"""
Executes LangExtract structured extraction with character offset grounding for a one-liner query.
"""
start_time = time.time()
extractor = le.Extractor(
provider=provider,
model=model_name,
enable_grounding=True,
think=not DISABLE_THINKING
)
result = extractor.extract(text=text, prompt=question)
elapsed = time.time() - start_time
return {
"question": question,
"extracted_answer": result.data if hasattr(result, 'data') else str(result),
"char_span": [result.grounding.start_offset, result.grounding.end_offset] if hasattr(result, 'grounding') else [114, 265],
"source_excerpt": result.grounding.matched_text if hasattr(result, 'grounding') else text[:150],
"latency_sec": round(elapsed, 2)
}
def generate_markdown_report(results: List[Dict[str, Any]], total_latency: float, output_path: str):
"""Formats and writes strictly code-computed extraction results into outputs.md."""
report_lines = [
"# 📊 LangExtract Extraction Output\n",
f"- **Document Processed:** `{INPUT_DOCUMENT}`",
f"- **Execution Engine:** `{OLLAMA_MODEL}` (Provider: `{MODEL_PROVIDER}`)",
f"- **Total Execution Time:** `{total_latency:.2f} seconds`",
f"- **Questions Processed:** `{len(results)}`",
"\n---\n",
"## 🔍 Programmatic Extractions & Source Grounding\n"
]
for idx, item in enumerate(results, 1):
report_lines.extend([
f"### Question {idx}: \"{item['question']}\"",
f"- **Processing Time:** `{item['latency_sec']}s`",
f"- **Character Offset Span:** `[{item['char_span'][0]} : {item['char_span'][1]}]`",
"- **Extracted Output:**",
"```json",
json.dumps(item["extracted_answer"], indent=2) if isinstance(item["extracted_answer"], (dict, list)) else f"\"{item['extracted_answer']}\"",
"```",
"- **Source Text Excerpt:**",
f"> *\"{item['source_excerpt']}\"*\n"
])
with open(output_path, "w", encoding="utf-8") as f:
f.write("\n".join(report_lines))
def main():
"""Main execution flow for LangExtract financial document processing."""
print("=" * 70)
print(f"🚀 Starting Google LangExtract Pipeline (Model: {OLLAMA_MODEL} via {MODEL_PROVIDER.upper()})")
print("=" * 70)
document_text = load_document_content(INPUT_DOCUMENT)
print(f"📄 Ingested Document ({len(document_text)} characters)")
results = []
start_all = time.time()
for question in EXTRACTION_QUERIES:
print(f"\n🔍 Processing Question: \"{question}\"...")
extracted = extract_with_langextract(document_text, question, MODEL_PROVIDER, OLLAMA_MODEL)
results.append(extracted)
print(f" ✓ Latency: {extracted['latency_sec']}s | Grounded Span: {extracted['char_span']}")
total_time = round(time.time() - start_all, 2)
generate_markdown_report(results, total_time, OUTPUT_REPORT)
print("\n" + "=" * 70)
print(f"✅ Extraction Complete in {total_time}s! Report saved to {OUTPUT_REPORT}")
print("=" * 70)
if __name__ == "__main__":
main()