ysharma HF Staff commited on
Commit
c5e2f5b
Β·
verified Β·
1 Parent(s): a328e08

Create app.py

Browse files
Files changed (1) hide show
  1. app.py +99 -0
app.py ADDED
@@ -0,0 +1,99 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ 03 Β· Data Detective (LOCAL β€” no token) Β· JSON + bind, mixed output types
3
+ =============================================================================
4
+
5
+ Upload a CSV; a single **file** reference fans out to four analysts that render
6
+ different port types on the canvas β€” a preview table, summary statistics, a
7
+ missing-value report, and a correlation heatmap.
8
+
9
+ Graph:
10
+ β”Œβ”€β–Ά preview ─▢ [Preview] (dataframe)
11
+ [CSV file] ─▢──────┼─▢ summary_stats ─▢ [Statistics] (dataframe)
12
+ β”œβ”€β–Ά missing_report ─▢ [Missing values] (json)
13
+ └─▢ correlation ─▢ [Heatmap] (image)
14
+
15
+ Because functions exchange values as JSON, each analyst reads the CSV path
16
+ itself and returns a JSON-safe payload: a `{headers, data}` table, a dict, or a
17
+ base64 image. Port types (`dataframe` / `json` / `image`) tell the canvas how to
18
+ render each result.
19
+
20
+ Try it with the bundled ../../assets/sample_data.csv
21
+
22
+ Run it:
23
+ python apps/03_data_detective/app.py
24
+ """
25
+
26
+ import base64
27
+ import io
28
+ import os
29
+
30
+ import gradio as gr
31
+ import matplotlib
32
+ matplotlib.use("Agg")
33
+ import matplotlib.pyplot as plt
34
+ import pandas as pd
35
+
36
+
37
+ def _read(file) -> pd.DataFrame:
38
+ if isinstance(file, dict): # {path|url} from a file component
39
+ file = file.get("path") or file.get("url")
40
+ return pd.read_csv(file)
41
+
42
+
43
+ def _table(df: pd.DataFrame) -> dict:
44
+ """A JSON-safe {headers, data} payload for a `dataframe` port."""
45
+ df = df.astype(object).where(pd.notna(df), None)
46
+ data = [[(x.item() if hasattr(x, "item") else x) for x in row] for row in df.values.tolist()]
47
+ return {"headers": [str(c) for c in df.columns], "data": data}
48
+
49
+
50
+ def preview(file: str) -> dict:
51
+ return _table(_read(file).head(25))
52
+
53
+
54
+ def summary_stats(file: str) -> dict:
55
+ desc = _read(file).describe(include="all").transpose().round(3)
56
+ desc.insert(0, "column", desc.index)
57
+ return _table(desc.reset_index(drop=True))
58
+
59
+
60
+ def missing_report(file: str) -> dict:
61
+ df = _read(file)
62
+ na = df.isna().sum()
63
+ return {
64
+ "rows": int(len(df)),
65
+ "columns": int(df.shape[1]),
66
+ "total_missing": int(na.sum()),
67
+ "missing_by_column": {c: int(v) for c, v in na.items() if v > 0} or "none πŸŽ‰",
68
+ "dtypes": {c: str(t) for c, t in df.dtypes.items()},
69
+ }
70
+
71
+
72
+ def correlation(file: str) -> str:
73
+ """A correlation heatmap as a base64 `data:image/png` string."""
74
+ corr = _read(file).corr(numeric_only=True)
75
+ fig, ax = plt.subplots(figsize=(1.1 * len(corr) + 2, 1.1 * len(corr) + 1.5))
76
+ im = ax.imshow(corr.values, cmap="RdBu", vmin=-1, vmax=1)
77
+ ax.set_xticks(range(len(corr)), corr.columns, rotation=45, ha="right")
78
+ ax.set_yticks(range(len(corr)), corr.columns)
79
+ for i in range(len(corr)):
80
+ for j in range(len(corr)):
81
+ v = corr.values[i, j]
82
+ ax.text(j, i, f"{v:.2f}", ha="center", va="center",
83
+ color="white" if abs(v) > 0.5 else "black", fontsize=8)
84
+ ax.set_title("Correlation matrix")
85
+ fig.colorbar(im, ax=ax, shrink=0.8)
86
+ buf = io.BytesIO()
87
+ fig.savefig(buf, format="png", bbox_inches="tight", dpi=110)
88
+ plt.close(fig)
89
+ return "data:image/png;base64," + base64.b64encode(buf.getvalue()).decode()
90
+
91
+
92
+ BIND = {"preview": preview, "summary_stats": summary_stats,
93
+ "missing_report": missing_report, "correlation": correlation}
94
+
95
+ WORKFLOW = os.path.join(os.path.dirname(os.path.abspath(__file__)), "workflow.json")
96
+ demo = gr.Workflow(WORKFLOW, bind=BIND)
97
+
98
+ if __name__ == "__main__":
99
+ demo.launch()