sinan7 commited on
Commit
f68c5f2
·
verified ·
1 Parent(s): 0748b16

Upload main.py

Browse files
Files changed (1) hide show
  1. main.py +7 -5
main.py CHANGED
@@ -46,13 +46,15 @@ def extract_text_from_pdf(pdf_path: str) -> str:
46
  raise HTTPException(status_code=400, detail="PDF contains no extractable text.")
47
  return text
48
 
49
- def chunk_text(text: str, max_tokens: int = 400) -> list:
50
- """Chunks text into manageable pieces."""
51
- tokens = tokenizer.encode(text)
52
- return [
53
- tokenizer.decode(tokens[i : i + max_tokens])
54
  for i in range(0, len(tokens), max_tokens)
55
  ]
 
 
56
 
57
  def generate_chunk_output(chunk: str) -> dict:
58
  """Processes a chunk of text with flan-t5-base."""
 
46
  raise HTTPException(status_code=400, detail="PDF contains no extractable text.")
47
  return text
48
 
49
+ def chunk_text(text: str, max_tokens: int = 512) -> list:
50
+ """Chunks text into pieces within the model's token limit."""
51
+ tokens = tokenizer.encode(text, truncation=False) # No truncation yet
52
+ chunks = [
53
+ tokens[i: i + max_tokens]
54
  for i in range(0, len(tokens), max_tokens)
55
  ]
56
+ return [tokenizer.decode(chunk) for chunk in chunks]
57
+
58
 
59
  def generate_chunk_output(chunk: str) -> dict:
60
  """Processes a chunk of text with flan-t5-base."""