Image-to-Text
PEFT
Safetensors
Portuguese
English
vision-language
table-extraction
scientific-figures
markdown-table
qwen2.5-vl
lora
icdar-metric-loss
Instructions to use lucasoc/sci-image-models with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- PEFT
How to use lucasoc/sci-image-models with PEFT:
from peft import PeftModel from transformers import AutoModelForCausalLM base_model = AutoModelForCausalLM.from_pretrained("Qwen/Qwen2.5-VL-3B-Instruct") model = PeftModel.from_pretrained(base_model, "lucasoc/sci-image-models") - Notebooks
- Google Colab
- Kaggle
Download src/metrics/table_parser.py from lucasoc/sci-image-models: direct link, hf CLI and curl.
- Browser
- Download file 3.48 kB
-
https://huggingface.co/lucasoc/sci-image-models/resolve/main/src/metrics/table_parser.py
- Command line
-
hf download hf://lucasoc/sci-image-models/src/metrics/table_parser.py
-
curl -L -o table_parser.py https://huggingface.co/lucasoc/sci-image-models/resolve/main/src/metrics/table_parser.py
3.48 kB
| """ | |
| Robust Markdown table parsing and DataFrame extraction. | |
| """ | |
| import io | |
| import re | |
| from typing import Optional, Tuple, List | |
| import pandas as pd | |
| def extract_markdown_table_block(text: str) -> Optional[str]: | |
| """Extracts the markdown table section from generated model text. | |
| Handles markdown code blocks (```markdown ... ``` or ``` ... ```) and | |
| raw pipe-delimited table patterns. | |
| """ | |
| if not text or not isinstance(text, str): | |
| return None | |
| # Check for markdown code fences first | |
| code_block_match = re.search(r"```(?:markdown|table)?\s*([\s\S]*?)\s*```", text, re.IGNORECASE) | |
| candidate_text = code_block_match.group(1) if code_block_match else text | |
| # Look for pipe-separated table lines (| ... | ... |) | |
| lines = candidate_text.strip().split("\n") | |
| table_lines: List[str] = [] | |
| in_table = False | |
| for line in lines: | |
| stripped = line.strip() | |
| if stripped.startswith("|") and stripped.endswith("|") and "|" in stripped[1:-1]: | |
| in_table = True | |
| table_lines.append(stripped) | |
| elif in_table: | |
| # End of table encountered | |
| break | |
| if len(table_lines) >= 2: | |
| return "\n".join(table_lines) | |
| return None | |
| def parse_markdown_to_dataframe(text: str) -> Tuple[Optional[pd.DataFrame], bool]: | |
| """Parses a markdown table string into a pandas DataFrame. | |
| Returns: | |
| Tuple of (DataFrame or None, is_valid_boolean) | |
| """ | |
| table_str = extract_markdown_table_block(text) | |
| if not table_str: | |
| return None, False | |
| try: | |
| lines = [line.strip() for line in table_str.strip().split("\n") if line.strip()] | |
| if len(lines) < 2: | |
| return None, False | |
| # Header line | |
| header_line = lines[0] | |
| header_cells = [c.strip() for c in header_line.split("|")[1:-1]] | |
| if not header_cells: | |
| return None, False | |
| # Check separator line (e.g. |---|---| or |:---|---:|) | |
| separator_line = lines[1] | |
| if not re.match(r"^\|?(\s*:?-+:?\s*\|?)+\s*$", separator_line): | |
| # Not a standard separator, but let's check if rows start immediately | |
| data_start_idx = 1 | |
| else: | |
| data_start_idx = 2 | |
| data_rows = [] | |
| for line in lines[data_start_idx:]: | |
| row_cells = [c.strip() for c in line.split("|")[1:-1]] | |
| if not row_cells: | |
| continue | |
| # Pad or truncate to match header length | |
| if len(row_cells) < len(header_cells): | |
| row_cells.extend([""] * (len(header_cells) - len(row_cells))) | |
| elif len(row_cells) > len(header_cells): | |
| row_cells = row_cells[:len(header_cells)] | |
| data_rows.append(row_cells) | |
| if not data_rows: | |
| return None, False | |
| df = pd.DataFrame(data_rows, columns=header_cells) | |
| # Attempt to convert numeric columns to float | |
| for col in df.columns: | |
| try: | |
| # Remove common formatting like commas or units if applicable | |
| cleaned_col = df[col].astype(str).str.replace(",", "", regex=False) | |
| df[col] = pd.to_numeric(cleaned_col) | |
| except Exception: | |
| pass | |
| return df, True | |
| except Exception: | |
| return None, False | |
| def dataframe_to_markdown(df: pd.DataFrame) -> str: | |
| """Converts a pandas DataFrame into a clean Markdown table.""" | |
| return df.to_markdown(index=False) | |