File size: 10,384 Bytes
bcfd9c8
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
import os
import json
import re
from typing import List, Dict, Any, Tuple
import pandas as pd

class DocumentProcessor:
    def __init__(self, chunk_size: int = 600, chunk_overlap: int = 80):
        self.chunk_size = chunk_size
        self.chunk_overlap = chunk_overlap

    def process_file(self, file_path: str, filename: str) -> Tuple[List[str], List[Dict[str, Any]]]:
        """
        Process a file based on its extension and return chunk texts and metadatas.
        """
        ext = os.path.splitext(filename)[1].lower()

        if ext in ['.csv', '.xlsx', '.xls']:
            return self._process_tabular(file_path, filename, ext)
        elif ext == '.pdf':
            return self._process_pdf(file_path, filename)
        elif ext in ['.docx', '.doc']:
            return self._process_docx(file_path, filename)
        elif ext in ['.json']:
            return self._process_json(file_path, filename)
        elif ext in ['.html', '.htm']:
            return self._process_html(file_path, filename)
        elif ext in ['.txt', '.md']:
            return self._process_text(file_path, filename, file_type=ext[1:].upper())
        else:
            # Fallback to plain text
            return self._process_text(file_path, filename, file_type="TXT")

    def _process_tabular(self, file_path: str, filename: str, ext: str) -> Tuple[List[str], List[Dict[str, Any]]]:
        """Convert tabular rows into structured key-value text blocks with sheet and row metadata."""
        chunks = []
        metadatas = []

        try:
            if ext == '.csv':
                sheets_dict = {"Sheet1": pd.read_csv(file_path)}
            else:
                sheets_dict = pd.read_excel(file_path, sheet_name=None)
        except Exception as e:
            print(f"Error reading tabular file {filename}: {e}")
            return [], []

        for sheet_name, df in sheets_dict.items():
            if df.empty:
                continue
            
            # Fill NA values cleanly
            df = df.fillna("N/A")
            
            for row_idx, row in df.iterrows():
                row_num = row_idx + 1
                row_str_list = []
                for col_name in df.columns:
                    val = str(row[col_name]).strip()
                    row_str_list.append(f"{col_name}: {val}")
                
                chunk_text = f"Source Table: {filename} | Sheet: {sheet_name} | Record #{row_num}\n" + "\n".join(row_str_list)
                
                chunks.append(chunk_text)
                metadatas.append({
                    "filename": filename,
                    "sheet": str(sheet_name),
                    "row_number": row_num,
                    "file_type": ext[1:].upper(),
                    "source": f"{filename} (Sheet: {sheet_name}, Row: {row_num})"
                })

        return chunks, metadatas

    def _process_pdf(self, file_path: str, filename: str) -> Tuple[List[str], List[Dict[str, Any]]]:
        """Extract pages & paragraphs from PDF."""
        chunks = []
        metadatas = []
        try:
            from pypdf import PdfReader
            reader = PdfReader(file_path)
            for page_num, page in enumerate(reader.pages, start=1):
                text = page.extract_text() or ""
                page_chunks = self._chunk_text_string(text)
                for chunk_idx, text_chunk in enumerate(page_chunks):
                    chunks.append(text_chunk)
                    metadatas.append({
                        "filename": filename,
                        "page_number": page_num,
                        "chunk_index": chunk_idx + 1,
                        "file_type": "PDF",
                        "source": f"{filename} (Page {page_num})"
                    })
        except Exception as e:
            print(f"Error extracting PDF {filename}: {e}")
            # Fallback plain text read
            with open(file_path, "r", encoding="utf-8", errors="ignore") as f:
                content = f.read()
            return self._chunk_generic_text(content, filename, "PDF")

        return chunks, metadatas

    def _process_docx(self, file_path: str, filename: str) -> Tuple[List[str], List[Dict[str, Any]]]:
        """Extract headings & paragraphs from DOCX."""
        chunks = []
        metadatas = []
        try:
            import docx
            doc = docx.Document(file_path)
            full_text = []
            for p in doc.paragraphs:
                if p.text.strip():
                    full_text.append(p.text.strip())
            combined = "\n\n".join(full_text)
            return self._chunk_generic_text(combined, filename, "DOCX")
        except Exception as e:
            print(f"Error processing DOCX {filename}: {e}")
            return [], []

    def _process_json(self, file_path: str, filename: str) -> Tuple[List[str], List[Dict[str, Any]]]:
        """Process JSON records or generic structure."""
        try:
            with open(file_path, "r", encoding="utf-8") as f:
                data = json.load(f)

            if isinstance(data, list):
                chunks = []
                metadatas = []
                for idx, item in enumerate(data, start=1):
                    item_str = json.dumps(item, indent=2)
                    chunks.append(f"JSON Record #{idx}:\n{item_str}")
                    metadatas.append({
                        "filename": filename,
                        "record_number": idx,
                        "file_type": "JSON",
                        "source": f"{filename} (Record #{idx})"
                    })
                return chunks, metadatas
            else:
                formatted = json.dumps(data, indent=2)
                return self._chunk_generic_text(formatted, filename, "JSON")
        except Exception as e:
            print(f"Error reading JSON {filename}: {e}")
            return [], []

    def _process_html(self, file_path: str, filename: str) -> Tuple[List[str], List[Dict[str, Any]]]:
        """Extract text content from HTML."""
        try:
            with open(file_path, "r", encoding="utf-8", errors="ignore") as f:
                html_content = f.read()
            # Strip tags using regex
            clean_text = re.sub(r'<style.*?>.*?</style>', '', html_content, flags=re.DOTALL)
            clean_text = re.sub(r'<script.*?>.*?</script>', '', clean_text, flags=re.DOTALL)
            clean_text = re.sub(r'<[^>]+>', ' ', clean_text)
            clean_text = re.sub(r'\s+', ' ', clean_text).strip()
            return self._chunk_generic_text(clean_text, filename, "HTML")
        except Exception as e:
            print(f"Error reading HTML {filename}: {e}")
            return [], []

    def _process_text(self, file_path: str, filename: str, file_type: str = "TXT") -> Tuple[List[str], List[Dict[str, Any]]]:
        """Read text/markdown file."""
        try:
            with open(file_path, "r", encoding="utf-8", errors="ignore") as f:
                text = f.read()
            return self._chunk_generic_text(text, filename, file_type)
        except Exception as e:
            print(f"Error reading text file {filename}: {e}")
            return [], []

    def _chunk_generic_text(self, text: str, filename: str, file_type: str) -> Tuple[List[str], List[Dict[str, Any]]]:
        chunks_str = self._chunk_text_string(text, header_prefix=f"Document: {filename} | Type: {file_type}\n")
        chunks = []
        metadatas = []
        for idx, c in enumerate(chunks_str, start=1):
            chunks.append(c)
            metadatas.append({
                "filename": filename,
                "chunk_index": idx,
                "file_type": file_type,
                "source": f"{filename} (Chunk #{idx})"
            })
        return chunks, metadatas

    def _chunk_text_string(self, text: str, header_prefix: str = "") -> List[str]:
        """
        Smart recursive sentence & paragraph aware chunking.
        Prevents breaking words, numbers, or sentences mid-way.
        """
        text = text.strip()
        if not text:
            return []

        # If text fits inside chunk size
        if len(text) <= self.chunk_size:
            return [f"{header_prefix}{text}" if header_prefix else text]

        # Recursive separators: paragraphs, lines, sentences, clauses
        separators = ["\n\n", "\n", ". ", "; ", "? ", "! ", " "]
        
        def split_text_by_separators(txt: str, sep_idx: int = 0) -> List[str]:
            if sep_idx >= len(separators) or len(txt) <= self.chunk_size:
                return [txt] if txt.strip() else []

            sep = separators[sep_idx]
            parts = txt.split(sep)
            
            result_chunks = []
            current_chunk = []
            current_length = 0

            for part in parts:
                part_str = part + (sep if sep != " " else " ")
                part_len = len(part_str)

                if current_length + part_len > self.chunk_size:
                    if current_chunk:
                        chunk_text = "".join(current_chunk).strip()
                        if chunk_text:
                            result_chunks.append(chunk_text)
                        current_chunk = []
                        current_length = 0

                    if part_len > self.chunk_size:
                        # Sub-split long parts using next separator
                        sub_parts = split_text_by_separators(part, sep_idx + 1)
                        result_chunks.extend(sub_parts)
                    else:
                        current_chunk.append(part_str)
                        current_length += part_len
                else:
                    current_chunk.append(part_str)
                    current_length += part_len

            if current_chunk:
                final_text = "".join(current_chunk).strip()
                if final_text:
                    result_chunks.append(final_text)

            return result_chunks

        raw_chunks = split_text_by_separators(text, 0)
        
        # Add overlap and optional header prefix
        final_chunks = []
        for i, chunk in enumerate(raw_chunks):
            chunk_with_header = f"{header_prefix}{chunk}" if header_prefix else chunk
            final_chunks.append(chunk_with_header)

        return final_chunks