import gradio as gr import json from huggingface_hub import hf_hub_download import spaces import regex as re FOLDER = "Selma" master_idx = {} print("Loading index...") try: file_path = hf_hub_download( repo_id="pnugues/selma_idx", filename="master.json", repo_type="dataset" ) with open(file_path, "r", encoding="utf-8") as f: master_idx = json.load(f) print("Loading complete!") except Exception as e: print(f"Error loading index: {e}") @spaces.GPU def concordance(word, master_index, window): # Check if word exists in index if word not in master_index: print(f"The word '{word}' was not found.") return # Iterate files and positions for file_name, positions in master_index[word].items(): print(file_name) # Open file and read with open('Selma/' + file_name, 'r', encoding='utf-8') as f: text = f.read().lower() # extract the surrounding text for pos in positions: # Calculate start and end indices, no negative indices start_idx = max(0, pos - window) end_idx = min(len(text), pos + len(word) + window) # Extract snippet, replace newlines with spaces snippet = text[start_idx:end_idx].replace('\n', ' ') print(f"\t{snippet}") with gr.Blocks(title="Selma explorer") as demo: gr.Markdown("# Selma explorer") with gr.Row(): with gr.Column(): in_text = gr.Textbox( label="Input text", placeholder="Write your word..." ) with gr.Column(): out_novel = gr.Textbox( label="Selma's novels", placeholder="Selma's concordances will show here...", lines=20 ) btn = gr.Button("Show me the concordances!") btn.click(fn=concordance, inputs=in_text, outputs=out_novel) if __name__ == "__main__": demo.launch(ssr_mode=False)