sinan7 commited on
Commit
fc9d1ac
·
verified ·
1 Parent(s): 6a74c0e

Upload main.py

Browse files
Files changed (1) hide show
  1. main.py +77 -54
main.py CHANGED
@@ -5,7 +5,7 @@ import os
5
  import fitz # PyMuPDF for PDF handling
6
  import torch
7
  import json
8
- from transformers import pipeline, GPT2Tokenizer, GPT2LMHeadModel
9
  import logging
10
 
11
  # Initialize FastAPI app
@@ -15,11 +15,15 @@ app = FastAPI()
15
  logging.basicConfig(level=logging.INFO)
16
  logger = logging.getLogger(__name__)
17
 
18
- # Initialize the GPT2-medium pipeline
 
 
 
 
19
  qa_pipeline = pipeline(
20
  "text-generation",
21
- model="gpt2-medium", # Switching to GPT2-medium
22
- tokenizer="gpt2-medium",
23
  device=-1 # Use CPU
24
  )
25
 
@@ -39,62 +43,81 @@ class ExtractedInfo(BaseModel):
39
 
40
  def extract_text_from_pdf(pdf_path: str) -> str:
41
  """Extracts text from a PDF file."""
42
- with fitz.open(pdf_path) as doc:
43
- text = "".join([page.get_text() for page in doc])
44
- if not text.strip():
45
- raise HTTPException(status_code=400, detail="PDF contains no extractable text.")
46
- return text
47
-
48
- def chunk_text(text: str, max_tokens: int = 900) -> list:
49
- """Splits the text into chunks that fit the model token limit."""
50
- tokens = qa_pipeline.tokenizer.encode(text)
51
- chunks = [
52
- qa_pipeline.tokenizer.decode(tokens[i:i + max_tokens])
 
 
 
 
53
  for i in range(0, len(tokens), max_tokens)
54
  ]
55
- return chunks
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
56
 
57
  def generate_structured_output(text: str) -> dict:
58
- """Generates structured output from the text using GPT2-medium."""
59
  chunks = chunk_text(text)
60
-
61
- # Collect results from each chunk
62
- generated_text = ""
 
 
 
 
 
 
 
 
63
  for chunk in chunks:
64
- prompt = f"""
65
- Extract the following information from the resume in JSON format:
66
- {{
67
- "work_experience": "<Summarized single work experience>",
68
- "education": {{
69
- "degree": "<Degree obtained>",
70
- "university": "<University attended>",
71
- "graduation_year": "<Year of graduation>"
72
- }},
73
- "professional_course_detail": "<Details of professional courses completed>",
74
- "software_usage": "<List of software tools used>",
75
- "safety_course_detail": "<Safety courses completed>",
76
- "hse_description": "<HSE (Health, Safety, Environment) practices>",
77
- "good_conduct_certificate": "<Details of good conduct certificate>"
78
- }}
79
- Resume text:
80
- {chunk}
81
- """
82
-
83
- response = qa_pipeline(prompt, max_new_tokens=300, temperature=0.7)
84
- generated_text += response[0]["generated_text"]
85
-
86
- # Extract JSON from the generated text
87
- try:
88
- json_start = generated_text.find("{")
89
- json_end = generated_text.rfind("}") + 1
90
- if json_start != -1 and json_end != -1:
91
- json_str = generated_text[json_start:json_end]
92
- return json.loads(json_str)
93
- else:
94
- raise ValueError("No valid JSON found in the model output")
95
- except Exception as e:
96
- logger.error(f"Error generating structured output: {e}")
97
- raise HTTPException(status_code=500, detail="Failed to generate structured output.")
98
 
99
  @app.post("/process_cv/", response_model=ExtractedInfo)
100
  async def process_cv(background_tasks: BackgroundTasks, file: UploadFile = File(...)):
 
5
  import fitz # PyMuPDF for PDF handling
6
  import torch
7
  import json
8
+ from transformers import pipeline, GPT2Tokenizer, AutoModelForCausalLM
9
  import logging
10
 
11
  # Initialize FastAPI app
 
15
  logging.basicConfig(level=logging.INFO)
16
  logger = logging.getLogger(__name__)
17
 
18
+ # Load GPT-Neo-2.7B tokenizer and model
19
+ tokenizer = GPT2Tokenizer.from_pretrained("EleutherAI/gpt-neo-2.7B")
20
+ model = AutoModelForCausalLM.from_pretrained("EleutherAI/gpt-neo-2.7B")
21
+
22
+ # Initialize the text-generation pipeline
23
  qa_pipeline = pipeline(
24
  "text-generation",
25
+ model=model,
26
+ tokenizer=tokenizer,
27
  device=-1 # Use CPU
28
  )
29
 
 
43
 
44
  def extract_text_from_pdf(pdf_path: str) -> str:
45
  """Extracts text from a PDF file."""
46
+ try:
47
+ with fitz.open(pdf_path) as doc:
48
+ text = "".join([page.get_text() for page in doc])
49
+ if not text.strip():
50
+ raise ValueError("PDF contains no extractable text.")
51
+ return text
52
+ except Exception as e:
53
+ logger.error(f"Error extracting text from PDF: {e}")
54
+ raise HTTPException(status_code=400, detail="Failed to extract text from the PDF.")
55
+
56
+ def chunk_text(text: str, max_tokens: int = 1800) -> list:
57
+ """Chunks text using the tokenizer to fit within GPT-Neo's 2048-token limit."""
58
+ tokens = tokenizer.encode(text)
59
+ return [
60
+ tokenizer.decode(tokens[i : i + max_tokens])
61
  for i in range(0, len(tokens), max_tokens)
62
  ]
63
+
64
+ def generate_chunk_output(chunk: str) -> dict:
65
+ """Generates structured output for a single chunk."""
66
+ prompt = f"""
67
+ Extract the following information from the resume in JSON format:
68
+ {{
69
+ "work_experience": "<Summarized single work experience>",
70
+ "education": {{
71
+ "degree": "<Degree obtained>",
72
+ "university": "<University attended>",
73
+ "graduation_year": "<Year of graduation>"
74
+ }},
75
+ "professional_course_detail": "<Details of professional courses completed>",
76
+ "software_usage": "<List of software tools used>",
77
+ "safety_course_detail": "<Safety courses completed>",
78
+ "hse_description": "<HSE (Health, Safety, Environment) practices>",
79
+ "good_conduct_certificate": "<Details of good conduct certificate>"
80
+ }}
81
+ Resume text:
82
+ {chunk}
83
+ """
84
+
85
+ response = qa_pipeline(prompt, max_new_tokens=500, temperature=0.7)
86
+ generated_text = response[0]["generated_text"]
87
+
88
+ # Extract JSON from the generated text
89
+ json_start = generated_text.find("{")
90
+ json_end = generated_text.rfind("}") + 1
91
+ if json_start != -1 and json_end != -1:
92
+ json_str = generated_text[json_start:json_end]
93
+ return json.loads(json_str)
94
+ else:
95
+ logger.warning("Invalid JSON output from model.")
96
+ return {} # Return empty dict if parsing fails
97
 
98
  def generate_structured_output(text: str) -> dict:
99
+ """Generates structured output from multiple chunks."""
100
  chunks = chunk_text(text)
101
+ merged_output = {
102
+ "work_experience": "",
103
+ "education": {"degree": "", "university": "", "graduation_year": ""},
104
+ "professional_course_detail": "",
105
+ "software_usage": "",
106
+ "safety_course_detail": "",
107
+ "hse_description": "",
108
+ "good_conduct_certificate": ""
109
+ }
110
+
111
+ # Process each chunk and merge results
112
  for chunk in chunks:
113
+ chunk_output = generate_chunk_output(chunk)
114
+ for key, value in chunk_output.items():
115
+ if isinstance(value, dict):
116
+ merged_output[key].update(value)
117
+ elif not merged_output[key]: # Only fill if not already populated
118
+ merged_output[key] = value
119
+
120
+ return merged_output
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
121
 
122
  @app.post("/process_cv/", response_model=ExtractedInfo)
123
  async def process_cv(background_tasks: BackgroundTasks, file: UploadFile = File(...)):