ollibolli commited on
Commit
4b9c5bf
·
1 Parent(s): 0035c98
Files changed (4) hide show
  1. README.md +32 -5
  2. app.log +0 -0
  3. app.py +270 -0
  4. requirements.txt +6 -0
README.md CHANGED
@@ -1,13 +1,40 @@
1
  ---
2
- title: Promptvisualization
3
- emoji: 🌖
4
  colorFrom: blue
5
- colorTo: yellow
6
  sdk: gradio
7
- sdk_version: 5.44.1
8
  app_file: app.py
9
  pinned: false
10
- short_description: api to visualize prompt engineering
11
  ---
12
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
13
  Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference
 
1
  ---
2
+ title: Token Probability Visualization API
3
+ emoji: 📊
4
  colorFrom: blue
5
+ colorTo: purple
6
  sdk: gradio
7
+ sdk_version: 4.44.0
8
  app_file: app.py
9
  pinned: false
10
+ short_description: API for visualizing LLM token probabilities and embeddings
11
  ---
12
 
13
+ # Token Probability Visualization API
14
+
15
+ This Space provides API endpoints for visualizing token probabilities and embeddings from language models.
16
+
17
+ ## Features
18
+
19
+ - **Token Probabilities**: Get next-token probability distributions for any input text
20
+ - **Embeddings**: Access token embeddings with PCA projections for visualization
21
+ - **Comparative Analysis**: Compare conditional vs unconditional distributions
22
+ - **Multiple Contexts**: Analyze probability shifts across different text contexts
23
+
24
+ ## API Endpoints
25
+
26
+ ### `/predict` - Main Prediction Endpoint
27
+ Returns comprehensive token probabilities, embeddings, and PCA projections.
28
+
29
+ ### `/embeddings` - Token Embeddings
30
+ Get embeddings for specific token IDs.
31
+
32
+ ## Usage
33
+
34
+ Use the API to build interactive visualizations showing:
35
+ - Token probability landscapes
36
+ - Embedding space visualizations
37
+ - Probability distribution comparisons
38
+ - Token movement analysis
39
+
40
  Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference
app.log ADDED
File without changes
app.py ADDED
@@ -0,0 +1,270 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import gradio as gr
2
+ import torch
3
+ import numpy as np
4
+ from transformers import AutoTokenizer, AutoModelForCausalLM
5
+ from sklearn.decomposition import PCA
6
+ import json
7
+
8
+ # Model configuration
9
+ MODEL_ID = "openai-community/gpt2"
10
+ device = "cuda" if torch.cuda.is_available() else "cpu"
11
+
12
+ # Load model and tokenizer
13
+ tokenizer = AutoTokenizer.from_pretrained(MODEL_ID)
14
+ model = AutoModelForCausalLM.from_pretrained(MODEL_ID, dtype=torch.float32).to(device).eval()
15
+
16
+ # Get vocabulary info
17
+ vocab_size = tokenizer.vocab_size if tokenizer.vocab_size is not None else len(tokenizer)
18
+ vocab_tokens = tokenizer.convert_ids_to_tokens(list(range(vocab_size)))
19
+
20
+ # Cache for embeddings and PCA (computed once at startup)
21
+ embeddings_cache = None
22
+ pca_cache = None
23
+ pca_projections_cache = None
24
+
25
+ def initialize_embeddings():
26
+ """Initialize embeddings and PCA projections once at startup"""
27
+ global embeddings_cache, pca_cache, pca_projections_cache
28
+
29
+ # Get embeddings
30
+ embeddings_cache = model.get_input_embeddings().weight.detach().cpu().numpy() # [V, d]
31
+
32
+ # Compute PCA
33
+ pca_cache = PCA(n_components=2, random_state=0)
34
+ pca_projections_cache = pca_cache.fit_transform(embeddings_cache) # [V, 2]
35
+
36
+ return embeddings_cache, pca_projections_cache
37
+
38
+ # Initialize embeddings at startup
39
+ initialize_embeddings()
40
+
41
+ def nice_tok(tok: str) -> str:
42
+ """Clean up token display"""
43
+ return tok.replace("Ġ", " ").replace("▁", " ")
44
+
45
+ @torch.no_grad()
46
+ def get_next_token_probs(text: str | None):
47
+ """Get next token probabilities for given text context"""
48
+ if text and len(text) > 0:
49
+ enc = tokenizer(text, return_tensors="pt").to(device)
50
+ out = model(**enc, return_dict=True)
51
+ logits = out.logits[0, -1, :]
52
+ else:
53
+ # Unconditional: use BOS or EOS token
54
+ bos_id = tokenizer.bos_token_id or tokenizer.eos_token_id
55
+ if bos_id is None:
56
+ raise ValueError("Tokenizer has neither BOS nor EOS.")
57
+ input_ids = torch.tensor([[bos_id]], device=device)
58
+ out = model(input_ids=input_ids, return_dict=True)
59
+ logits = out.logits[0, -1, :]
60
+
61
+ return logits
62
+
63
+ @torch.no_grad()
64
+ def predict_comprehensive(
65
+ text: str,
66
+ text2: str = "",
67
+ top_k: int = 0,
68
+ include_embeddings: bool = False,
69
+ include_pca: bool = False,
70
+ include_unconditional: bool = False,
71
+ use_logprobs: bool = True
72
+ ):
73
+ """
74
+ Comprehensive prediction endpoint that returns token probabilities,
75
+ embeddings, and PCA projections for visualization
76
+ """
77
+
78
+ result = {
79
+ "model": MODEL_ID,
80
+ "vocab_size": vocab_size,
81
+ "device": device
82
+ }
83
+
84
+ # Get primary text probabilities
85
+ logits = get_next_token_probs(text)
86
+ probs = torch.softmax(logits, dim=-1).cpu().numpy()
87
+
88
+ if use_logprobs:
89
+ log_probs = torch.log_softmax(logits, dim=-1).cpu().numpy()
90
+ else:
91
+ log_probs = None
92
+
93
+ # Always include vocabulary tokens (cleaned)
94
+ result["vocab"] = {
95
+ "tokens": [nice_tok(t) for t in vocab_tokens],
96
+ "raw_tokens": vocab_tokens,
97
+ "size": vocab_size
98
+ }
99
+
100
+ # Primary context probabilities
101
+ result["probs"] = probs.tolist()
102
+ if log_probs is not None:
103
+ result["logprobs"] = log_probs.tolist()
104
+
105
+ # Top-k tokens if requested
106
+ if top_k and top_k > 0:
107
+ vals, idxs = torch.topk(torch.from_numpy(probs), k=min(top_k, len(probs)))
108
+ idxs = idxs.tolist()
109
+ vals = vals.tolist()
110
+ result["topk"] = {
111
+ "ids": idxs,
112
+ "tokens": [nice_tok(vocab_tokens[i]) for i in idxs],
113
+ "probs": vals
114
+ }
115
+ if log_probs is not None:
116
+ result["topk"]["logprobs"] = [float(log_probs[i]) for i in idxs]
117
+
118
+ # Second text context if provided
119
+ if text2 and len(text2) > 0:
120
+ logits2 = get_next_token_probs(text2)
121
+ probs2 = torch.softmax(logits2, dim=-1).cpu().numpy()
122
+ result["probs2"] = probs2.tolist()
123
+
124
+ if use_logprobs:
125
+ log_probs2 = torch.log_softmax(logits2, dim=-1).cpu().numpy()
126
+ result["logprobs2"] = log_probs2.tolist()
127
+
128
+ # Unconditional probabilities
129
+ if include_unconditional:
130
+ logits_uncond = get_next_token_probs(None)
131
+ probs_uncond = torch.softmax(logits_uncond, dim=-1).cpu().numpy()
132
+ result["unconditional_probs"] = probs_uncond.tolist()
133
+
134
+ if use_logprobs:
135
+ log_probs_uncond = torch.log_softmax(logits_uncond, dim=-1).cpu().numpy()
136
+ result["unconditional_logprobs"] = log_probs_uncond.tolist()
137
+
138
+ # Embeddings if requested
139
+ if include_embeddings:
140
+ # Return first 10 dimensions as sample (full embeddings would be too large)
141
+ result["embeddings_sample"] = {
142
+ "shape": list(embeddings_cache.shape),
143
+ "first_10_tokens_sample": embeddings_cache[:10, :10].tolist() if embeddings_cache is not None else None
144
+ }
145
+
146
+ # PCA projections if requested
147
+ if include_pca:
148
+ if pca_projections_cache is not None:
149
+ result["pca"] = {
150
+ "projections": pca_projections_cache.tolist(),
151
+ "explained_variance_ratio": pca_cache.explained_variance_ratio_.tolist()
152
+ }
153
+
154
+ return result
155
+
156
+ @torch.no_grad()
157
+ def get_embeddings_endpoint(token_ids: str = ""):
158
+ """
159
+ Get embeddings for specific token IDs
160
+ """
161
+ if token_ids:
162
+ try:
163
+ ids = [int(x.strip()) for x in token_ids.split(",")]
164
+ ids = [i for i in ids if 0 <= i < vocab_size]
165
+ except:
166
+ ids = list(range(min(100, vocab_size))) # Default to first 100
167
+ else:
168
+ ids = list(range(min(100, vocab_size))) # Default to first 100
169
+
170
+ embeddings_subset = embeddings_cache[ids]
171
+ tokens_subset = [nice_tok(vocab_tokens[i]) for i in ids]
172
+
173
+ return {
174
+ "token_ids": ids,
175
+ "tokens": tokens_subset,
176
+ "embeddings": embeddings_subset.tolist(),
177
+ "embedding_dim": embeddings_cache.shape[1],
178
+ "total_vocab_size": vocab_size
179
+ }
180
+
181
+ # Create Gradio interface
182
+ with gr.Blocks(title="Token Probability Visualization API") as demo:
183
+ gr.Markdown("""
184
+ # Token Probability Visualization API
185
+
186
+ This API provides token probabilities, embeddings, and PCA projections for visualization.
187
+ Model: `openai-community/gpt2`
188
+ """)
189
+
190
+ with gr.Tab("Comprehensive API"):
191
+ gr.Markdown("### Get token probabilities with optional embeddings and PCA")
192
+
193
+ with gr.Row():
194
+ with gr.Column():
195
+ text_input = gr.Textbox(
196
+ label="Primary Context",
197
+ value="You are an expert in medieval history.",
198
+ lines=3
199
+ )
200
+ text2_input = gr.Textbox(
201
+ label="Secondary Context (optional)",
202
+ placeholder="Enter second text for comparison",
203
+ lines=3
204
+ )
205
+
206
+ with gr.Row():
207
+ top_k_slider = gr.Slider(0, 200, step=1, value=20, label="Top-K tokens")
208
+ include_embeddings = gr.Checkbox(False, label="Include Embeddings Sample")
209
+ include_pca = gr.Checkbox(True, label="Include PCA Projections")
210
+ include_unconditional = gr.Checkbox(True, label="Include Unconditional")
211
+ use_logprobs = gr.Checkbox(True, label="Include Log Probabilities")
212
+
213
+ predict_btn = gr.Button("Get Predictions", variant="primary")
214
+
215
+ with gr.Column():
216
+ output_json = gr.JSON(label="API Response")
217
+
218
+ predict_btn.click(
219
+ fn=predict_comprehensive,
220
+ inputs=[text_input, text2_input, top_k_slider, include_embeddings,
221
+ include_pca, include_unconditional, use_logprobs],
222
+ outputs=output_json,
223
+ api_name="predict"
224
+ )
225
+
226
+ with gr.Tab("Embeddings API"):
227
+ gr.Markdown("### Get embeddings for specific tokens")
228
+
229
+ with gr.Row():
230
+ with gr.Column():
231
+ token_ids_input = gr.Textbox(
232
+ label="Token IDs (comma-separated)",
233
+ placeholder="e.g., 0,1,2,3,4 or leave empty for first 100",
234
+ value="0,1,2,3,4,5,6,7,8,9"
235
+ )
236
+ get_embeddings_btn = gr.Button("Get Embeddings", variant="primary")
237
+
238
+ with gr.Column():
239
+ embeddings_output = gr.JSON(label="Embeddings Response")
240
+
241
+ get_embeddings_btn.click(
242
+ fn=get_embeddings_endpoint,
243
+ inputs=token_ids_input,
244
+ outputs=embeddings_output,
245
+ api_name="embeddings"
246
+ )
247
+
248
+ gr.Markdown("""
249
+ ## API Usage
250
+
251
+ ### Endpoints:
252
+ - `/predict`: Main endpoint for token probabilities with optional embeddings and PCA
253
+ - `/embeddings`: Get embeddings for specific token IDs
254
+
255
+ ### Response includes:
256
+ - Token probabilities (conditional and unconditional)
257
+ - Log probabilities
258
+ - PCA projections for all vocabulary tokens
259
+ - Token embeddings (sample or specific)
260
+ - Vocabulary mappings
261
+
262
+ ### Use this data to:
263
+ - Visualize token probability landscapes
264
+ - Compare conditional vs unconditional distributions
265
+ - Create embedding visualizations
266
+ - Build interactive token explorers
267
+ """)
268
+
269
+ if __name__ == "__main__":
270
+ demo.launch(server_name="0.0.0.0", server_port=7860, share=False)
requirements.txt ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ gradio==4.44.0
2
+ torch>=2.0.0
3
+ transformers>=4.30.0
4
+ scikit-learn>=1.3.0
5
+ numpy>=1.24.0
6
+ spaces