the-data-nerd commited on
Commit
163ee0b
·
verified ·
1 Parent(s): 741ecb4

chore: upload app.py

Browse files
Files changed (1) hide show
  1. app.py +342 -0
app.py ADDED
@@ -0,0 +1,342 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """VC Deal Flow Signal — Interactive Explorer.
2
+
3
+ Hugging Face Space that loads the live HF dataset
4
+ (huggingface.co/datasets/the-data-nerd/vc-deal-flow-signal) and renders
5
+ an interactive Gradio dashboard.
6
+
7
+ 5 tabs:
8
+ - Overview : KPI strip + signal-type composition + top movers (latest quarter)
9
+ - Sector heatmap : sector x quarter avg commit velocity
10
+ - Top movers : filterable ranking of accelerating startups
11
+ - Startup drilldown : per-startup four-quarter trajectory
12
+ - Methodology & cite : SSRN, Zenodo, classifier code, MCP server, citation BibTeX
13
+
14
+ The dataset is public (CC-BY-4.0); no auth required at runtime.
15
+ """
16
+ from __future__ import annotations
17
+
18
+ import gradio as gr
19
+ import pandas as pd
20
+ import plotly.express as px
21
+ import plotly.graph_objects as go
22
+ from huggingface_hub import hf_hub_download
23
+
24
+ REPO_ID = "the-data-nerd/vc-deal-flow-signal"
25
+ PERIOD_ORDER = ["q3-2025", "q4-2025", "q1-2026", "q2-2026"]
26
+ PERIOD_LABEL = {p: p.upper().replace("-", " ") for p in PERIOD_ORDER}
27
+
28
+
29
+ def load_csv(name: str) -> pd.DataFrame:
30
+ path = hf_hub_download(repo_id=REPO_ID, filename=name, repo_type="dataset")
31
+ return pd.read_csv(path)
32
+
33
+
34
+ SIGNALS = load_csv("startup_signals.csv")
35
+ SECTORS = load_csv("sector_aggregates.csv")
36
+ TIMESERIES = load_csv("signal_type_timeseries.csv")
37
+
38
+ for _df in (SIGNALS, SECTORS, TIMESERIES):
39
+ _df["period"] = pd.Categorical(_df["period"], categories=PERIOD_ORDER, ordered=True)
40
+
41
+ LATEST = str(SIGNALS["period"].max())
42
+ N_STARTUPS = SIGNALS["startup_name"].nunique()
43
+ N_SECTORS = SIGNALS["sector_name"].nunique()
44
+ N_QUARTERS = SIGNALS["period"].nunique()
45
+ N_OBSERVATIONS = len(SIGNALS)
46
+ SECTORS_SORTED = sorted(SIGNALS["sector_name"].unique())
47
+ STAGES_SORTED = sorted(s for s in SIGNALS["stage"].unique() if isinstance(s, str))
48
+ SIGNAL_TYPES = sorted(s for s in SIGNALS["signal_type"].unique() if isinstance(s, str))
49
+ STARTUPS_SORTED = sorted(SIGNALS["startup_name"].unique())
50
+
51
+
52
+ def overview_kpis() -> str:
53
+ latest_df = SIGNALS[SIGNALS["period"] == LATEST]
54
+ top_mover = latest_df.sort_values("commit_velocity_change_pct", ascending=False).iloc[0]
55
+ return (
56
+ f"**{N_STARTUPS}** startups · **{N_SECTORS}** sectors · **{N_QUARTERS}** quarters · **{N_OBSERVATIONS}** observations \n"
57
+ f"**Top mover {PERIOD_LABEL[LATEST]}**: `{top_mover['startup_name']}` "
58
+ f"({top_mover['sector_name']}) — `{top_mover['commit_velocity_change_pct']:+.0f}%` Δ commit velocity"
59
+ )
60
+
61
+
62
+ def overview_signal_share_fig() -> go.Figure:
63
+ ts = TIMESERIES.copy()
64
+ ts["period"] = ts["period"].astype(str)
65
+ fig = px.bar(
66
+ ts,
67
+ x="period",
68
+ y="share_of_total",
69
+ color="signal_type",
70
+ title="Signal Type Share by Quarter",
71
+ labels={"share_of_total": "Share of total", "period": "Quarter", "signal_type": "Signal"},
72
+ category_orders={"period": PERIOD_ORDER},
73
+ )
74
+ fig.update_layout(margin=dict(l=20, r=20, t=50, b=20), height=380, legend_title_text="")
75
+ return fig
76
+
77
+
78
+ def overview_top10() -> pd.DataFrame:
79
+ latest_df = SIGNALS[SIGNALS["period"] == LATEST]
80
+ return (
81
+ latest_df.sort_values("commit_velocity_change_pct", ascending=False)
82
+ .head(10)[
83
+ [
84
+ "startup_name",
85
+ "sector_name",
86
+ "stage",
87
+ "commit_velocity_14d",
88
+ "commit_velocity_change_pct",
89
+ "signal_type",
90
+ "github_url",
91
+ ]
92
+ ]
93
+ .rename(
94
+ columns={
95
+ "startup_name": "Startup",
96
+ "sector_name": "Sector",
97
+ "stage": "Stage",
98
+ "commit_velocity_14d": "Velocity (14d)",
99
+ "commit_velocity_change_pct": "Change %",
100
+ "signal_type": "Signal",
101
+ "github_url": "GitHub",
102
+ }
103
+ )
104
+ .reset_index(drop=True)
105
+ )
106
+
107
+
108
+ def sector_heatmap_fig() -> go.Figure:
109
+ pivot = (
110
+ SECTORS.pivot_table(
111
+ index="sector_name",
112
+ columns="period",
113
+ values="avg_commit_velocity_14d",
114
+ aggfunc="mean",
115
+ observed=True,
116
+ )
117
+ .reindex(columns=PERIOD_ORDER)
118
+ .sort_index()
119
+ )
120
+ fig = px.imshow(
121
+ pivot,
122
+ labels=dict(x="Quarter", y="Sector", color="Avg Commit Velocity (14d)"),
123
+ aspect="auto",
124
+ color_continuous_scale="Viridis",
125
+ title="Average Commit Velocity by Sector × Quarter",
126
+ text_auto=".0f",
127
+ )
128
+ fig.update_layout(margin=dict(l=20, r=20, t=50, b=20), height=620)
129
+ return fig
130
+
131
+
132
+ def filter_movers(period: str, sector: str, stage: str, signal: str, top_n: int):
133
+ df = SIGNALS.copy()
134
+ if period != "All":
135
+ df = df[df["period"] == period]
136
+ if sector != "All":
137
+ df = df[df["sector_name"] == sector]
138
+ if stage != "All":
139
+ df = df[df["stage"] == stage]
140
+ if signal != "All":
141
+ df = df[df["signal_type"] == signal]
142
+
143
+ df = df.sort_values("commit_velocity_change_pct", ascending=False).head(int(top_n))
144
+
145
+ if df.empty:
146
+ empty_fig = go.Figure()
147
+ empty_fig.add_annotation(text="No rows match the selected filters", x=0.5, y=0.5, showarrow=False)
148
+ empty_fig.update_layout(height=380, margin=dict(l=20, r=20, t=50, b=20))
149
+ return df, empty_fig
150
+
151
+ fig = px.bar(
152
+ df,
153
+ y="startup_name",
154
+ x="commit_velocity_change_pct",
155
+ color="signal_type",
156
+ orientation="h",
157
+ title=f"Top {len(df)} Movers — Δ Commit Velocity",
158
+ labels={"commit_velocity_change_pct": "Velocity Change %", "startup_name": "Startup", "signal_type": "Signal"},
159
+ hover_data=["sector_name", "stage", "commit_velocity_14d"],
160
+ )
161
+ fig.update_layout(
162
+ yaxis={"categoryorder": "total ascending"},
163
+ height=max(380, 28 * len(df)),
164
+ margin=dict(l=20, r=20, t=50, b=20),
165
+ legend_title_text="",
166
+ )
167
+
168
+ table = (
169
+ df[
170
+ [
171
+ "startup_name",
172
+ "sector_name",
173
+ "stage",
174
+ "commit_velocity_14d",
175
+ "commit_velocity_change_pct",
176
+ "signal_type",
177
+ "github_url",
178
+ ]
179
+ ]
180
+ .rename(
181
+ columns={
182
+ "startup_name": "Startup",
183
+ "sector_name": "Sector",
184
+ "stage": "Stage",
185
+ "commit_velocity_14d": "Velocity (14d)",
186
+ "commit_velocity_change_pct": "Change %",
187
+ "signal_type": "Signal",
188
+ "github_url": "GitHub",
189
+ }
190
+ )
191
+ .reset_index(drop=True)
192
+ )
193
+ return table, fig
194
+
195
+
196
+ def drilldown(startup: str):
197
+ if not startup:
198
+ return "_Pick a startup above_", go.Figure()
199
+ df = SIGNALS[SIGNALS["startup_name"] == startup].copy()
200
+ if df.empty:
201
+ return f"_No rows for `{startup}`_", go.Figure()
202
+ df = df.sort_values("period")
203
+ df["period_str"] = df["period"].astype(str)
204
+
205
+ fig = go.Figure()
206
+ fig.add_trace(
207
+ go.Scatter(
208
+ x=df["period_str"],
209
+ y=df["commit_velocity_14d"],
210
+ mode="lines+markers+text",
211
+ name="Commit velocity (14d)",
212
+ text=df["commit_velocity_14d"].astype(str),
213
+ textposition="top center",
214
+ )
215
+ )
216
+ fig.update_layout(
217
+ title=f"{startup} — commit velocity trajectory",
218
+ xaxis_title="Quarter",
219
+ yaxis_title="Commit velocity (14d)",
220
+ height=380,
221
+ margin=dict(l=20, r=20, t=50, b=20),
222
+ )
223
+
224
+ latest = df.iloc[-1]
225
+ md = (
226
+ f"**Sector:** {latest['sector_name']} \n"
227
+ f"**Stage:** {latest['stage']} \n"
228
+ f"**Geography:** {latest['geography']} \n"
229
+ f"**Latest signal ({PERIOD_LABEL[str(latest['period'])]}):** `{latest['signal_type']}` \n"
230
+ f"**Commit velocity (14d):** {latest['commit_velocity_14d']} (Δ {latest['commit_velocity_change_pct']:+.0f}%) \n"
231
+ f"**Contributors:** {latest['contributors']} (growth {latest['contributor_growth_pct']:+.0f}%) \n"
232
+ f"**GitHub:** [{latest['github_url']}]({latest['github_url']})"
233
+ )
234
+ return md, fig
235
+
236
+
237
+ with gr.Blocks(title="VC Deal Flow Signal — Interactive Explorer", theme=gr.themes.Soft()) as demo:
238
+ gr.Markdown(
239
+ f"""
240
+ # 📊 VC Deal Flow Signal — Interactive Explorer
241
+
242
+ Live engineering-velocity panel across **{N_STARTUPS}** venture-backed startups in **{N_SECTORS}** sectors over **{N_QUARTERS}** quarters of GitHub data.
243
+
244
+ Source: [`the-data-nerd/vc-deal-flow-signal`](https://huggingface.co/datasets/the-data-nerd/vc-deal-flow-signal) · CC-BY-4.0 · methodology on [SSRN 6606558](https://ssrn.com/abstract=6606558) · companion chat agent: [`vc-deal-flow-deepseek`](https://huggingface.co/spaces/the-data-nerd/vc-deal-flow-deepseek)
245
+ """
246
+ )
247
+
248
+ with gr.Tabs():
249
+ with gr.Tab("Overview"):
250
+ gr.Markdown(overview_kpis())
251
+ gr.Plot(value=overview_signal_share_fig(), label="Signal-type composition over quarters")
252
+ gr.Markdown(f"### Top 10 movers — {PERIOD_LABEL[LATEST]}")
253
+ gr.Dataframe(value=overview_top10(), interactive=False, wrap=True)
254
+
255
+ with gr.Tab("Sector heatmap"):
256
+ gr.Plot(value=sector_heatmap_fig(), label="Sector × Quarter heatmap")
257
+ gr.Markdown(
258
+ "_Each cell shows the **average 14-day commit velocity** of the startups tracked in that sector "
259
+ "for that quarter. Brighter = more engineering throughput. Use it to spot sector-level rotations._"
260
+ )
261
+
262
+ with gr.Tab("Top movers"):
263
+ with gr.Row():
264
+ period_dd = gr.Dropdown(["All"] + PERIOD_ORDER, value=LATEST, label="Quarter")
265
+ sector_dd = gr.Dropdown(["All"] + SECTORS_SORTED, value="All", label="Sector")
266
+ stage_dd = gr.Dropdown(["All"] + STAGES_SORTED, value="All", label="Stage")
267
+ signal_dd = gr.Dropdown(["All"] + SIGNAL_TYPES, value="All", label="Signal type")
268
+ topn_slider = gr.Slider(5, 50, value=15, step=5, label="Top N")
269
+
270
+ init_table, init_fig = filter_movers(LATEST, "All", "All", "All", 15)
271
+ movers_table = gr.Dataframe(value=init_table, label="Filtered movers", interactive=False, wrap=True)
272
+ movers_fig = gr.Plot(value=init_fig, label="Velocity-change ranking")
273
+
274
+ for control in (period_dd, sector_dd, stage_dd, signal_dd, topn_slider):
275
+ control.change(
276
+ filter_movers,
277
+ inputs=[period_dd, sector_dd, stage_dd, signal_dd, topn_slider],
278
+ outputs=[movers_table, movers_fig],
279
+ )
280
+
281
+ with gr.Tab("Startup drilldown"):
282
+ startup_dd = gr.Dropdown(STARTUPS_SORTED, value=STARTUPS_SORTED[0], label="Pick a startup")
283
+ init_md, init_drill_fig = drilldown(STARTUPS_SORTED[0])
284
+ drill_md = gr.Markdown(value=init_md)
285
+ drill_fig = gr.Plot(value=init_drill_fig, label="Commit velocity over time")
286
+ startup_dd.change(drilldown, inputs=[startup_dd], outputs=[drill_md, drill_fig])
287
+
288
+ with gr.Tab("Methodology & cite"):
289
+ gr.Markdown(
290
+ """
291
+ ## How signals are computed
292
+
293
+ The dataset is derived live from the [GitHub REST API v3](https://docs.github.com/en/rest). For each tracked startup we sample its most active public organisation repository on a 14-day rolling window, four times per quarter.
294
+
295
+ **Working hypothesis (testable, falsifiable):** sustained engineering acceleration — commit velocity rising significantly above a startup's own baseline — tends to precede fundraise announcements by roughly 6–12 weeks.
296
+
297
+ The classifier maps every (startup, quarter) observation onto one of four signal types:
298
+
299
+ | Signal | Definition |
300
+ |---|---|
301
+ | `Engineering hiring burst` | Unique-contributor count spikes vs. trailing 90-day baseline |
302
+ | `Infrastructure buildout` | Multiple new public repos created in the last 30 days |
303
+ | `Deploy frequency spike` | Commit velocity ≥ 2× the trailing 90-day baseline |
304
+ | `Framework migration` | High commit volume with low contributor growth and zero new repos |
305
+
306
+ Full classifier source (MIT): [github.com/kindrat86/gitdealflow-signal-classifier](https://github.com/kindrat86/gitdealflow-signal-classifier)
307
+
308
+ ## Cite this dataset
309
+
310
+ ```bibtex
311
+ @dataset{vc_deal_flow_signal_2026,
312
+ author = {The Data Nerd},
313
+ title = {Startup GitHub Engineering Velocity Panel},
314
+ year = {2026},
315
+ publisher = {Zenodo},
316
+ doi = {10.5281/zenodo.19650920},
317
+ url = {https://huggingface.co/datasets/the-data-nerd/vc-deal-flow-signal}
318
+ }
319
+ ```
320
+
321
+ ## Live mirrors and related artefacts
322
+
323
+ - **HF dataset (this Space's source):** https://huggingface.co/datasets/the-data-nerd/vc-deal-flow-signal
324
+ - **Zenodo (DOI'd version):** https://zenodo.org/records/19650920 — concept DOI [10.5281/zenodo.19650919](https://doi.org/10.5281/zenodo.19650919)
325
+ - **Kaggle mirror:** https://www.kaggle.com/datasets/thedatanerd2026/vc-deal-flow-signal
326
+ - **Data.world mirror:** https://data.world/thedatanerd2026/vc-deal-flow-signal-startup-engineering-acceleration
327
+ - **SSRN preprint (methodology):** https://ssrn.com/abstract=6606558
328
+ - **Live MCP server (read-only, public):** https://signals.gitdealflow.com/api/mcp/rpc
329
+ - **Companion chat agent (Space):** https://huggingface.co/spaces/the-data-nerd/vc-deal-flow-deepseek
330
+ - **Production web app:** https://signals.gitdealflow.com
331
+ """
332
+ )
333
+
334
+ gr.Markdown(
335
+ """---
336
+ Built with [Gradio](https://gradio.app) on top of [Hugging Face Datasets](https://huggingface.co/docs/datasets). Code: MIT. Data: CC-BY-4.0. _Past acceleration does not guarantee future outcomes — this is alternative-data research, not investment advice._
337
+ """
338
+ )
339
+
340
+
341
+ if __name__ == "__main__":
342
+ demo.launch()