alibayram commited on
Commit
1a04cf0
·
1 Parent(s): da42085

tokenizer updated

Browse files
Files changed (2) hide show
  1. app.py +42 -22
  2. requirements.txt +1 -1
app.py CHANGED
@@ -1,6 +1,6 @@
1
  import gradio as gr
2
  import pkg_resources
3
- from turkish_tokenizer import TokenType, TurkishTokenizer
4
 
5
  # Get the version from the installed package
6
  try:
@@ -12,11 +12,12 @@ tokenizer = TurkishTokenizer()
12
 
13
  # Define colors for each token type
14
  color_map = {
15
- TokenType.ROOT.name: "#FF6B6B", # Red
16
- TokenType.SUFFIX.name: "#4ECDC4", # Teal
17
- TokenType.BPE.name: "#FFE66D", # Yellow
18
  }
19
 
 
20
  def tokenize_and_display(text):
21
  """
22
  Tokenizes the input text and prepares it for display in Gradio's HighlightedText component.
@@ -25,7 +26,7 @@ def tokenize_and_display(text):
25
  # Return a structure that matches all outputs to avoid errors
26
  return [], "", "", ""
27
 
28
- tokens, _ = tokenizer.tokenize_text(text)
29
 
30
  # Create the list of (token, label) for HighlightedText
31
  highlighted_tokens = []
@@ -33,7 +34,7 @@ def tokenize_and_display(text):
33
 
34
  for t in tokens:
35
  token_text = t["token"]
36
- token_type = t["type"].name
37
 
38
  # Count token types for statistics
39
  token_stats[token_type] = token_stats.get(token_type, 0) + 1
@@ -49,7 +50,12 @@ def tokenize_and_display(text):
49
  compression_ratio = (1 - total_tokens / total_chars) * 100 if total_chars > 0 else 0
50
 
51
  # Define colors for the stats block
52
- bg_col, text_col, card_col, border_col = ('#f8f9fa', '#2d3748', '#ffffff', '#e2e8f0')
 
 
 
 
 
53
 
54
  # Create statistics HTML
55
  stats_html = f"""
@@ -71,6 +77,7 @@ def tokenize_and_display(text):
71
  </div>"""
72
  return highlighted_tokens, str(encoded_ids), decoded_text, stats_html
73
 
 
74
  # Custom CSS for better styling
75
  custom_css = """
76
  .gradio-container{font-family:'Inter',-apple-system,BlinkMacSystemFont,sans-serif;}
@@ -81,30 +88,38 @@ custom_css = """
81
  """
82
 
83
  # Create the Gradio Interface
84
- with gr.Blocks(theme=gr.themes.Soft(), title="Turkish Tokenizer", css=custom_css) as demo:
 
 
85
  with gr.Row():
86
  with gr.Column(scale=3):
87
- gr.Markdown(f"""
 
88
  # Turkish Tokenizer
89
  ### Advanced Turkish Text Tokenization with Visual Analysis
90
  Enter text to see how it's tokenized. Tokens are color-coded by type.
91
- """)
 
92
 
93
  input_text = gr.Textbox(
94
  label="📝 Input Text",
95
  placeholder="Merhaba Dünya, kitapları okumak güzeldir.",
96
  lines=4,
97
- elem_classes=["input-textbox"]
98
  )
99
 
100
  with gr.Row():
101
- process_button = gr.Button("🚀 Tokenize", variant="primary", elem_classes=["custom-button"], size="lg")
 
 
102
  clear_button = gr.Button("🗑️ Clear", variant="secondary", size="lg")
103
 
104
  gr.Markdown("---")
105
  gr.Markdown("### 🔄 Encoded & Decoded Output")
106
  with gr.Row():
107
- encoded_output = gr.Textbox(label="🔢 Encoded Token IDs", interactive=False, lines=2)
 
 
108
  decoded_output = gr.Textbox(label="📝 Decoded Text", interactive=False, lines=2)
109
 
110
  gr.Markdown("### 💡 Example Texts")
@@ -117,23 +132,22 @@ with gr.Blocks(theme=gr.themes.Soft(), title="Turkish Tokenizer", css=custom_css
117
  ["Yapay zeka ve makine öğrenmesi teknolojileri gelişiyor."],
118
  ],
119
  inputs=input_text,
120
- label="Try these examples:"
121
  )
122
 
123
  gr.Markdown("---")
124
  gr.Markdown("### 🎨 Tokenization Output")
125
  highlighted_output = gr.HighlightedText(
126
- label="Colorized Tokens",
127
- color_map=color_map,
128
- show_legend=True
129
  )
130
 
131
  gr.Markdown("---")
132
  gr.Markdown("### 📊 Statistics")
133
  stats_output = gr.HTML(label="")
134
 
135
-
136
- gr.Markdown(f"--- \n **Turkish Tokenizer v{VERSION}** - Advanced tokenization for Turkish text.")
 
137
 
138
  # --- Event Handlers ---
139
  def process_with_theme(text):
@@ -146,18 +160,24 @@ with gr.Blocks(theme=gr.themes.Soft(), title="Turkish Tokenizer", css=custom_css
146
  process_button.click(
147
  fn=process_with_theme,
148
  inputs=[input_text],
149
- outputs=[highlighted_output, encoded_output, decoded_output, stats_output]
150
  )
151
 
152
  clear_button.click(
153
  fn=clear_all,
154
- outputs=[input_text, highlighted_output, encoded_output, decoded_output, stats_output]
 
 
 
 
 
 
155
  )
156
 
157
  # Auto-process on load with a default example
158
  demo.load(
159
  fn=lambda: tokenize_and_display("Merhaba Dünya!"),
160
- outputs=[highlighted_output, encoded_output, decoded_output, stats_output]
161
  )
162
 
163
  if __name__ == "__main__":
 
1
  import gradio as gr
2
  import pkg_resources
3
+ from turkish_tokenizer import TurkishTokenizer
4
 
5
  # Get the version from the installed package
6
  try:
 
12
 
13
  # Define colors for each token type
14
  color_map = {
15
+ "ROOT": "#FF6B6B", # Red
16
+ "SUFFIX": "#4ECDC4", # Teal
17
+ "BPE": "#FFE66D", # Yellow
18
  }
19
 
20
+
21
  def tokenize_and_display(text):
22
  """
23
  Tokenizes the input text and prepares it for display in Gradio's HighlightedText component.
 
26
  # Return a structure that matches all outputs to avoid errors
27
  return [], "", "", ""
28
 
29
+ tokens = tokenizer.tokenize_text(text)
30
 
31
  # Create the list of (token, label) for HighlightedText
32
  highlighted_tokens = []
 
34
 
35
  for t in tokens:
36
  token_text = t["token"]
37
+ token_type = t["type"]
38
 
39
  # Count token types for statistics
40
  token_stats[token_type] = token_stats.get(token_type, 0) + 1
 
50
  compression_ratio = (1 - total_tokens / total_chars) * 100 if total_chars > 0 else 0
51
 
52
  # Define colors for the stats block
53
+ bg_col, text_col, card_col, border_col = (
54
+ "#f8f9fa",
55
+ "#2d3748",
56
+ "#ffffff",
57
+ "#e2e8f0",
58
+ )
59
 
60
  # Create statistics HTML
61
  stats_html = f"""
 
77
  </div>"""
78
  return highlighted_tokens, str(encoded_ids), decoded_text, stats_html
79
 
80
+
81
  # Custom CSS for better styling
82
  custom_css = """
83
  .gradio-container{font-family:'Inter',-apple-system,BlinkMacSystemFont,sans-serif;}
 
88
  """
89
 
90
  # Create the Gradio Interface
91
+ with gr.Blocks(
92
+ theme=gr.themes.Soft(), title="Turkish Tokenizer", css=custom_css
93
+ ) as demo:
94
  with gr.Row():
95
  with gr.Column(scale=3):
96
+ gr.Markdown(
97
+ f"""
98
  # Turkish Tokenizer
99
  ### Advanced Turkish Text Tokenization with Visual Analysis
100
  Enter text to see how it's tokenized. Tokens are color-coded by type.
101
+ """
102
+ )
103
 
104
  input_text = gr.Textbox(
105
  label="📝 Input Text",
106
  placeholder="Merhaba Dünya, kitapları okumak güzeldir.",
107
  lines=4,
108
+ elem_classes=["input-textbox"],
109
  )
110
 
111
  with gr.Row():
112
+ process_button = gr.Button(
113
+ "🚀 Tokenize", variant="primary", elem_classes=["custom-button"], size="lg"
114
+ )
115
  clear_button = gr.Button("🗑️ Clear", variant="secondary", size="lg")
116
 
117
  gr.Markdown("---")
118
  gr.Markdown("### 🔄 Encoded & Decoded Output")
119
  with gr.Row():
120
+ encoded_output = gr.Textbox(
121
+ label="🔢 Encoded Token IDs", interactive=False, lines=2
122
+ )
123
  decoded_output = gr.Textbox(label="📝 Decoded Text", interactive=False, lines=2)
124
 
125
  gr.Markdown("### 💡 Example Texts")
 
132
  ["Yapay zeka ve makine öğrenmesi teknolojileri gelişiyor."],
133
  ],
134
  inputs=input_text,
135
+ label="Try these examples:",
136
  )
137
 
138
  gr.Markdown("---")
139
  gr.Markdown("### 🎨 Tokenization Output")
140
  highlighted_output = gr.HighlightedText(
141
+ label="Colorized Tokens", color_map=color_map, show_legend=True
 
 
142
  )
143
 
144
  gr.Markdown("---")
145
  gr.Markdown("### 📊 Statistics")
146
  stats_output = gr.HTML(label="")
147
 
148
+ gr.Markdown(
149
+ f"--- \n **Turkish Tokenizer v{VERSION}** - Advanced tokenization for Turkish text."
150
+ )
151
 
152
  # --- Event Handlers ---
153
  def process_with_theme(text):
 
160
  process_button.click(
161
  fn=process_with_theme,
162
  inputs=[input_text],
163
+ outputs=[highlighted_output, encoded_output, decoded_output, stats_output],
164
  )
165
 
166
  clear_button.click(
167
  fn=clear_all,
168
+ outputs=[
169
+ input_text,
170
+ highlighted_output,
171
+ encoded_output,
172
+ decoded_output,
173
+ stats_output,
174
+ ],
175
  )
176
 
177
  # Auto-process on load with a default example
178
  demo.load(
179
  fn=lambda: tokenize_and_display("Merhaba Dünya!"),
180
+ outputs=[highlighted_output, encoded_output, decoded_output, stats_output],
181
  )
182
 
183
  if __name__ == "__main__":
requirements.txt CHANGED
@@ -1,3 +1,3 @@
1
  gradio
2
 
3
- turkish-tokenizer==0.2.26
 
1
  gradio
2
 
3
+ turkish-tokenizer>=1.0.4