Lab1_app / app.py
agnesgoransson's picture
Update app.py
8d9d418 verified
Raw History Blame Contribute Delete
2.09 kB
import gradio as gr
import json
from huggingface_hub import hf_hub_download
import spaces
import regex as re
FOLDER = "Selma"
master_idx = {}
print("Loading index...")
try:
file_path = hf_hub_download(
repo_id="pnugues/selma_idx",
filename="master.json",
repo_type="dataset"
)
with open(file_path, "r", encoding="utf-8") as f:
master_idx = json.load(f)
print("Loading complete!")
except Exception as e:
print(f"Error loading index: {e}")
@spaces.GPU
def concordance(word, master_index, window):
# Check if word exists in index
if word not in master_index:
print(f"The word '{word}' was not found.")
return
# Iterate files and positions
for file_name, positions in master_index[word].items():
print(file_name)
# Open file and read
with open('Selma/' + file_name, 'r', encoding='utf-8') as f:
text = f.read().lower()
# extract the surrounding text
for pos in positions:
# Calculate start and end indices, no negative indices
start_idx = max(0, pos - window)
end_idx = min(len(text), pos + len(word) + window)
# Extract snippet, replace newlines with spaces
snippet = text[start_idx:end_idx].replace('\n', ' ')
print(f"\t{snippet}")
with gr.Blocks(title="Selma explorer") as demo:
gr.Markdown("# Selma explorer")
with gr.Row():
with gr.Column():
in_text = gr.Textbox(
label="Input text",
placeholder="Write your word..."
)
with gr.Column():
out_novel = gr.Textbox(
label="Selma's novels",
placeholder="Selma's concordances will show here...",
lines=20
)
btn = gr.Button("Show me the concordances!")
btn.click(fn=concordance, inputs=in_text, outputs=out_novel)
if __name__ == "__main__":
demo.launch(ssr_mode=False)