"""Hebrew text summarization Gradio application. Summarises Hebrew-language passages using the DictaIL `dictalm2.0` instruction-tuned seq2seq model served through a Hugging Face ``text2text-generation`` pipeline. The model is loaded lazily on first inference rather than at import time. On the free ``cpu-basic`` Hugging Face Space tier this matters: the checkpoint is several hundred megabytes, and eager loading delays the appearance of the Gradio UI past the Space startup timeout. Example: >>> from app import summarize_text >>> summarize_text("זהו טקסט לדוגמה ארוך מספיק לסיכום.") '...' """ from __future__ import annotations import logging import threading from typing import Final import gradio as gr import torch from transformers import pipeline #: Hugging Face repository id of the Hebrew summarisation model. MODEL_NAME: Final[str] = "dicta-il/dictalm2.0" #: Hebrew instruction prefix used to steer the instruction-tuned model. SUMMARIZATION_PROMPT_TEMPLATE: Final[str] = "סכם את הטקסט הבא: {text}" #: Minimum number of tokens the model should generate for a usable summary. MIN_SUMMARY_TOKENS: Final[int] = 30 #: Maximum number of tokens the model may generate. MAX_SUMMARY_TOKENS: Final[int] = 512 #: Longest input accepted, in characters. Guards against CPU timeouts. MAX_INPUT_CHARS: Final[int] = 20_000 logger = logging.getLogger(__name__) # Guards lazy initialisation: concurrent Gradio requests must not each build a # separate copy of the model. _pipeline_lock = threading.Lock() _summarizer = None class ModelLoadError(RuntimeError): """Raised when the summarisation model cannot be loaded or executed.""" def _build_pipeline(): """Construct the Hugging Face text2text-generation pipeline. Uses ``float32`` rather than ``bfloat16``. The ``cpu-basic`` Space tier has no CUDA device, where bfloat16 is either unsupported or falls back through slow software emulation paths. Returns: pipeline: A configured Hugging Face text2text-generation pipeline. Raises: ModelLoadError: If the pipeline cannot be constructed. """ try: return pipeline( "text2text-generation", model=MODEL_NAME, # float32 is the only reliable dtype for CPU inference. torch_dtype=torch.float32, ) except Exception as exc: # noqa: BLE001 - surfaced to the user below. raise ModelLoadError( f"Could not load summarisation model '{MODEL_NAME}'. " "The Space may be rate-limited or the model may be unavailable." ) from exc def get_summarizer(): """Return the shared summarisation pipeline, loading it on first use. Thread-safe: concurrent callers block on a lock while the first caller downloads and builds the model, then all share the same instance. Returns: pipeline: The shared text2text-generation pipeline. Raises: ModelLoadError: If the model cannot be loaded on first use. """ global _summarizer if _summarizer is None: with _pipeline_lock: if _summarizer is None: logger.info("Loading %s ...", MODEL_NAME) _summarizer = _build_pipeline() return _summarizer def summarize_text(text: str) -> str: """Summarise a Hebrew passage. Args: text: The Hebrew-language passage to summarise. Leading and trailing whitespace is ignored. May be empty. Returns: str: The generated Hebrew summary, or an empty string when ``text`` is blank. Raises: ValueError: If ``text`` exceeds :data:`MAX_INPUT_CHARS`. ModelLoadError: If the model cannot be loaded. RuntimeError: If inference itself fails. Example: >>> summarize_text(" ") '' >>> summarize_text("חברה מקבוצת טלה מדווחת על רכישה ...") 'חברה מקבוצת טלה מדווחת על רכישה ...' """ cleaned = (text or "").strip() if not cleaned: return "" if len(cleaned) > MAX_INPUT_CHARS: raise ValueError( f"Input is too long ({len(cleaned)} characters). " f"Please paste at most {MAX_INPUT_CHARS} characters." ) prompt = SUMMARIZATION_PROMPT_TEMPLATE.format(text=cleaned) summarizer = get_summarizer() try: output = summarizer( prompt, max_new_tokens=MAX_SUMMARY_TOKENS, min_length=MIN_SUMMARY_TOKENS, do_sample=False, ) except Exception as exc: # noqa: BLE001 - converted into a readable message. raise RuntimeError(f"Summarisation failed: {exc}") from exc return output[0]["generated_text"] def summarize_with_error_handling(text: str) -> str: """Summarise text and convert failures into user-readable messages. Wraps :func:`summarize_text` for direct use as a Gradio callback so the user sees a plain explanation rather than a stack trace. Args: text: The Hebrew-language passage to summarise. Returns: str: The generated summary, or a message describing what went wrong. Example: >>> isinstance(summarize_with_error_handling("שלום"), str) True """ try: return summarize_text(text) except ValueError as exc: return f"Input problem: {exc}" except ModelLoadError as exc: return f"Model unavailable: {exc}" except RuntimeError as exc: return f"Summarisation error: {exc}" def build_demo() -> gr.Blocks: """Assemble the Gradio interface. Returns: gr.Blocks: The configured demo, ready to launch. Example: >>> demo = build_demo() >>> isinstance(demo, gr.Blocks) True """ with gr.Blocks(title="Hebrew Text Summarizer") as demo: gr.Markdown( "# Hebrew Text Summarizer\n" "Summarises Hebrew-language passages using the DictaIL `dictalm2.0` model. " "Inference runs on CPU, so short passages return fastest." ) with gr.Row(): with gr.Column(): source = gr.Textbox( label="Input text to summarize", lines=8, placeholder="Paste Hebrew text here ...", ) submit = gr.Button("Summarize", variant="primary") with gr.Column(): output = gr.Textbox(label="Summarized text", lines=8) gr.Examples( examples=[ ["חברה מקבוצת טלה מדווחת על החתמת הסכם עסקה ..."], ], inputs=[source], label="Examples", ) submit.click( fn=summarize_with_error_handling, inputs=[source], outputs=[output], ) return demo def main() -> None: """Launch the Gradio application.""" demo = build_demo() demo.queue().launch() if __name__ == "__main__": main()