Artemyr commited on
Commit
aa5d4b8
·
verified ·
1 Parent(s): 954c229

Update app.py

Browse files
Files changed (1) hide show
  1. app.py +93 -81
app.py CHANGED
@@ -4,54 +4,59 @@ import pandas as pd
4
  from transformers import pipeline
5
 
6
  MAX_CHARS = 1500
7
- HISTORY_SIZE = 5
8
 
9
  MODELS = {
10
  "Русский (rubert-tiny)": "cointegrated/rubert-tiny-sentiment-balanced",
11
- "Английский (roberta)": "cardiffnlp/twitter-roberta-base-sentiment"
 
12
  }
13
 
14
  _pipes = {}
15
 
 
16
  def get_pipe(key):
17
- if key not in _pipes:
18
- _pipes[key] = pipeline("sentiment-analysis", model=MODELS[key])
19
- return _pipes[key]
20
-
21
- def analyze_text(text, model_key, history):
22
- start = time.time()
23
-
24
- if text is None or not text.strip():
25
- return "Ошибка: пустой ввод", "", 0.0, history
26
-
27
- text = text.strip()
28
- if len(text) > MAX_CHARS:
29
- return "Ошибка: текст слишком длинный", "", 0.0, history
30
-
31
- try:
32
- clf = get_pipe(model_key)
33
- res = clf(text)[0]
34
- label = res["label"]
35
- score = round(float(res["score"]), 4)
36
- except Exception as e:
37
- return f"Ошибка модели: {e}", "", 0.0, history
38
-
39
- latency = round(time.time() - start, 3)
40
-
41
- record = f"[{model_key}] {text[:80]}... → {label} ({score}), {latency}s"
42
- history = (history or [])[-(HISTORY_SIZE-1):] + [record]
43
-
44
- return label, score, latency, history
45
-
46
- def batch_process(file_obj, model_key):
47
- if file_obj is None:
48
- return pd.DataFrame({"error": ["Файл не загружен"]})
49
-
50
- path = file_obj.name.lower()
51
- if not (path.endswith(".txt") or path.endswith(".csv")):
52
- return pd.DataFrame({"error": ["Поддерживаются только .txt и .csv"]})
53
-
54
- try:
 
 
 
55
  if path.endswith(".txt"):
56
  with open(file_obj.name, "r", encoding="utf-8", errors="ignore") as f:
57
  texts = [x.strip() for x in f.read().splitlines() if x.strip()]
@@ -59,66 +64,73 @@ def batch_process(file_obj, model_key):
59
  df = pd.read_csv(file_obj.name)
60
  col = "text" if "text" in df.columns else df.columns[0]
61
  texts = df[col].astype(str).tolist()
62
- except Exception as e:
63
  return pd.DataFrame({"error": [f"Ошибка чтения файла: {e}"]})
64
-
65
- clf = get_pipe(model_key)
66
- rows = []
67
- for t in texts:
68
- t = t[:MAX_CHARS]
69
- res = clf(t)[0]
70
- rows.append({
71
- "text": t,
72
- "label": res["label"],
73
- "score": round(float(res["score"]), 4)
74
- })
75
- return pd.DataFrame(rows)
76
-
77
 
78
  with gr.Blocks() as demo:
79
- gr.Markdown("# 🧠 Sentiment Analysis (RU/EN) — Hugging Face Spaces (Gradio)")
80
-
81
- with gr.Row():
82
- text_input = gr.Textbox(label="Введите текст", lines=5)
83
- model_choice = gr.Dropdown(list(MODELS.keys()), value="Русский (rubert-tiny)", label="Модель")
84
 
85
- run_btn = gr.Button("Обработать")
 
 
86
 
87
- with gr.Row():
88
- out_label = gr.Textbox(label="Тональность")
89
- out_score = gr.Textbox(label="Уверенн��сть")
90
- out_latency = gr.Textbox(label="Время ответа (сек)")
91
 
92
- history_state = gr.State([])
 
 
 
93
 
94
- history_box = gr.Textbox(label="История запросов (последние 5)", lines=6)
 
95
 
96
- run_btn.click(
97
  analyze_text,
98
- inputs=[text_input, model_choice, history_state],
99
- outputs=[out_label, out_score, out_latency, history_state]
100
  ).then(
101
  lambda h: "\n".join(h),
102
- inputs=history_state,
103
- outputs=history_box
104
  )
105
 
106
- gr.Markdown("## 📦 Пакетная обработка (TXT/CSV)")
107
- file_input = gr.File(label="Загрузите файл (.txt или .csv с колонкой text)")
108
- batch_btn = gr.Button("Обработать файл")
109
- batch_out = gr.Dataframe(label="Результаты")
110
-
111
- batch_btn.click(batch_process, inputs=[file_input, model_choice], outputs=batch_out)
112
 
 
113
  gr.Examples(
114
  examples=[
115
- ["Мне очень понравился сервис, всё отлично!", "Русский (rubert-tiny)"],
116
  ["Это худший опыт в моей жизни.", "Русский (rubert-tiny)"],
117
- ["The app is amazing!", "Английский (roberta)"],
118
- ["This is terrible and buggy.", "Английский (roberta)"]
119
  ],
120
  inputs=[text_input, model_choice],
121
  label="Примеры"
122
  )
123
 
124
- demo.launch()
 
 
 
 
 
 
 
 
 
 
4
  from transformers import pipeline
5
 
6
  MAX_CHARS = 1500
7
+ History_size = 5
8
 
9
  MODELS = {
10
  "Русский (rubert-tiny)": "cointegrated/rubert-tiny-sentiment-balanced",
11
+ "Twitter-roBERTa": "cardiffnlp/twitter-roberta-base-sentiment-latest"
12
+
13
  }
14
 
15
  _pipes = {}
16
 
17
+ #фукнция выбора конкретной модели
18
  def get_pipe(key):
19
+ if key not in _pipes:
20
+ _pipes[key] = pipeline("sentiment-analysis", model=MODELS[key])
21
+ #анализ текста с историей
22
+ def analyze_text(text,model_key, history):
23
+ start = time.time()
24
+
25
+ if text is None or not text.strip():
26
+ return "Ошибка пустой ввод", "", 0.0, history
27
+
28
+ text = text.strip()
29
+ if len(text) > MAX_CHARS:
30
+ return "Ошибка: текст слишком длинный", "", 0.0, history
31
+
32
+ try:
33
+ a = get_pipe(model_key)
34
+ res = a(text)[0]
35
+ label = res["label"]
36
+ score = round(float(res["score"]),3)
37
+ except Exception as e:
38
+ return f"Ошибка модели: {e}", "", 0.0, history
39
+
40
+ latency = round(time.time()- start,3)
41
+ record = f"[{model_key}] {text[:80]}... → {label} ({score}), {latency}s"
42
+ if not history:
43
+ history = []
44
+ history = history[-(History_size-1):]
45
+ history.append(record)
46
+
47
+ return label, score, latency, history
48
+
49
+ #функция для обработки файла
50
+ def obr_file(file_obj, model_key):
51
+ if file_obj is None:
52
+ return pd.DataFrame({"error": ["Файл не загружен"]})
53
+ #проверяем расширение файла
54
+ path = file_obj.name.lower()
55
+ if not (path.endswith(".txt") or path.endswith(".csv")):
56
+ return pd.DataFrame({"error": ["Поддерживаются только .txt и .csv"]})
57
+
58
+ #чтение файла
59
+ try:
60
  if path.endswith(".txt"):
61
  with open(file_obj.name, "r", encoding="utf-8", errors="ignore") as f:
62
  texts = [x.strip() for x in f.read().splitlines() if x.strip()]
 
64
  df = pd.read_csv(file_obj.name)
65
  col = "text" if "text" in df.columns else df.columns[0]
66
  texts = df[col].astype(str).tolist()
67
+ except Exception as e:
68
  return pd.DataFrame({"error": [f"Ошибка чтения файла: {e}"]})
69
+
70
+ a = get_pipe(model_key)
71
+ rows = []
72
+
73
+ for t in texts:
74
+ t = t[:MAX_CHARS]
75
+ res = a(t)[0]
76
+ rows.append({
77
+ "text": t,
78
+ "label": res["label"],
79
+ "score": round(float(res["score"]), 3)
80
+ })
81
+ return pd.DataFrame(rows)
82
 
83
  with gr.Blocks() as demo:
84
+ gr.Markdown('🧠 Sentiment Analysis')
 
 
 
 
85
 
86
+ with gr.Row():
87
+ text_input = gr.Textbox(label='Введите текст', lines =5)
88
+ model_choice = gr.Dropdown(list(MODELS.keys()), value="Русский (rubert-tiny)", label="Модель")
89
 
90
+ btn = gr.Button("Обработать")
 
 
 
91
 
92
+ with gr.Row():
93
+ lab = gr.Textbox(label='Тональность')
94
+ scr = gr.Textbox(label = 'Уверенность')
95
+ lat = gr.Textbox(label = 'Время ответа (сек)')
96
 
97
+ state = gr.State([])
98
+ box = gr.Textbox(label="История запросов (последние 5)", lines=5)
99
 
100
+ btn.click(
101
  analyze_text,
102
+ inputs = [text_input, model_choice, state],
103
+ outputs = [lab, scr, lat, state]
104
  ).then(
105
  lambda h: "\n".join(h),
106
+ inputs = state,
107
+ outputs = box
108
  )
109
 
110
+ gr.Markdown("# Пакетная обработка (TXT/CSV)")
111
+ file_in = gr.File(label = "Загрузите файл (.txt или .csv)")
112
+ bbtn = gr.Button("Обработать файл")
113
+ obtn = gr.DataFrame(label = 'Результаты')
 
 
114
 
115
+ bbtn.click(obr_file, inputs=[file_in, model_choice], outputs=obtn)
116
  gr.Examples(
117
  examples=[
118
+ ["Мне очень понравился этот фильм, всё отлично!", "Русский (rubert-tiny)"],
119
  ["Это худший опыт в моей жизни.", "Русский (rubert-tiny)"],
120
+ ["The app is amazing!", "Английский (Twitter-roBERTa)"],
121
+ ["This is terrible and buggy.", "Английский (Twitter-roBERTa)"]
122
  ],
123
  inputs=[text_input, model_choice],
124
  label="Примеры"
125
  )
126
 
127
+ demo.launch()
128
+
129
+
130
+
131
+
132
+
133
+
134
+
135
+
136
+