dedemerve commited on
Commit
e0f27d3
·
verified ·
1 Parent(s): b7bbd27

Add survey dataset viewer app

Browse files
Files changed (1) hide show
  1. app.py +171 -0
app.py ADDED
@@ -0,0 +1,171 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ ILSA-Survey-Dataset — Clickable Source Viewer
3
+ """
4
+
5
+ import gradio as gr
6
+ import pandas as pd
7
+ from huggingface_hub import hf_hub_download
8
+
9
+ REPO_ID = "dedemerve/ILSA-Survey-Dataset"
10
+ MAX_CELL_CHARS = 300
11
+
12
+ _CACHE = {}
13
+
14
+ SHEETS = {
15
+ "Articles (130 studies)": "data/articles_master.csv",
16
+ "Main Findings (202 outcomes)": "data/main_findings.csv",
17
+ "Confounders (1907 predictors)": "data/confounders.csv",
18
+ }
19
+
20
+
21
+ def _is_blank(val) -> bool:
22
+ if val is None:
23
+ return True
24
+ try:
25
+ if pd.isna(val):
26
+ return True
27
+ except (TypeError, ValueError):
28
+ pass
29
+ return str(val).strip().lower() in ("", "none", "null", "nan", "n/a", "<na>")
30
+
31
+
32
+ def _truncate(val) -> str:
33
+ if _is_blank(val):
34
+ return ""
35
+ s = str(val)
36
+ return s if len(s) <= MAX_CELL_CHARS else s[:MAX_CELL_CHARS] + "…"
37
+
38
+
39
+ _LONGTEXT_HINTS = ("interpretation", "summary", "description", "definition", "finding",
40
+ "confounder", "notes", "abstract", "text", "criteria", "explanation",
41
+ "outcome", "primary", "technique", "filter")
42
+ _SHORT_HINTS = ("year", "doi", "type", "category", "used", "id", "url", "label", "size")
43
+
44
+
45
+ def _column_width(col: str) -> str:
46
+ c = col.lower()
47
+ if col == "Source":
48
+ return "200px"
49
+ if "title" in c or c in ("name",):
50
+ return "280px"
51
+ if "journal" in c or "venue" in c or "authors" in c:
52
+ return "220px"
53
+ if any(h in c for h in _LONGTEXT_HINTS):
54
+ return "300px"
55
+ if any(h in c for h in _SHORT_HINTS):
56
+ return "110px"
57
+ return "160px"
58
+
59
+
60
+ def _load_raw(sheet_key: str) -> pd.DataFrame:
61
+ if sheet_key not in _CACHE:
62
+ csv_path = SHEETS[sheet_key]
63
+ local = hf_hub_download(repo_id=REPO_ID, filename=csv_path, repo_type="dataset")
64
+ _CACHE[sheet_key] = pd.read_csv(local, dtype=str)
65
+ return _CACHE[sheet_key]
66
+
67
+
68
+ def _make_source_link(row) -> str:
69
+ for col in ("source_url", "paper_url"):
70
+ val = row.get(col)
71
+ if not _is_blank(val):
72
+ url = str(val).strip()
73
+ return f"[Open paper]({url})"
74
+ doi = row.get("doi")
75
+ if not _is_blank(doi):
76
+ doi = str(doi).strip()
77
+ url = doi if doi.startswith("http") else f"https://doi.org/{doi}"
78
+ return f"[Open via DOI]({url})"
79
+ import urllib.parse
80
+ title = row.get("title", "")
81
+ if not _is_blank(title):
82
+ q = urllib.parse.quote_plus(str(title).strip())
83
+ return f"[Search Google Scholar](https://scholar.google.com/scholar?q={q})"
84
+ return ""
85
+
86
+
87
+ def build_table(sheet_key: str, search_text: str, max_rows: int):
88
+ if not sheet_key:
89
+ return gr.update(), "Select a table to get started."
90
+ try:
91
+ df = _load_raw(sheet_key).copy()
92
+ except Exception as e:
93
+ return gr.update(), f"Could not load data: {e}"
94
+
95
+ total_rows = len(df)
96
+
97
+ if search_text:
98
+ title_col = next((c for c in df.columns if "title" in c.lower()), None)
99
+ if title_col:
100
+ df = df[df[title_col].astype(str).str.contains(search_text, case=False, na=False)]
101
+
102
+ filtered_rows = len(df)
103
+ df = df.head(max_rows)
104
+
105
+ # Build Source link column if link-related columns exist
106
+ link_cols = [c for c in df.columns if c in ("source_url", "paper_url", "doi")]
107
+ has_links = bool(link_cols)
108
+
109
+ if has_links:
110
+ source_col = df.apply(_make_source_link, axis=1)
111
+ cols_to_drop = [c for c in ("source_url", "paper_url") if c in df.columns]
112
+ df = df.drop(columns=cols_to_drop)
113
+ df.insert(0, "Source", source_col)
114
+
115
+ for col in df.columns:
116
+ if col == "Source":
117
+ continue
118
+ df[col] = df[col].apply(_truncate)
119
+
120
+ datatype = ["markdown" if col == "Source" else "str" for col in df.columns]
121
+ column_widths = [_column_width(col) for col in df.columns]
122
+
123
+ info = (
124
+ f"**{sheet_key}** — {total_rows} rows total, "
125
+ f"{filtered_rows} after filtering, showing {len(df)} rows."
126
+ )
127
+ if has_links:
128
+ info += " Click **Source** to open the paper."
129
+
130
+ return gr.update(value=df, datatype=datatype, column_widths=column_widths), info
131
+
132
+
133
+ with gr.Blocks(title="ILSA Survey Dataset Viewer") as demo:
134
+ gr.Markdown(
135
+ f"""
136
+ # ILSA Survey Dataset — Clickable Source Viewer
137
+
138
+ Browse the three relational tables from the survey paper
139
+ *"Artificial Intelligence Applications in International Large-Scale Assessments:
140
+ A Survey with LLM-Assisted Evidence Synthesis"* (Dede & Çetinkaya, 2026).
141
+
142
+ Each row in the **Articles** table links directly to the paper via DOI.
143
+
144
+ Dataset: [`{REPO_ID}`](https://huggingface.co/datasets/{REPO_ID}) &nbsp;|&nbsp;
145
+ Website: [dedemerve.github.io/ILSA-Survey-Extractor](https://dedemerve.github.io/ILSA-Survey-Extractor/)
146
+ """
147
+ )
148
+ with gr.Row():
149
+ sheet_dd = gr.Dropdown(
150
+ choices=list(SHEETS.keys()),
151
+ value=list(SHEETS.keys())[0],
152
+ label="Table",
153
+ )
154
+ with gr.Row():
155
+ search_box = gr.Textbox(
156
+ label="Search by title (Articles table only)",
157
+ placeholder="e.g. PISA, reading, ICCS…",
158
+ )
159
+ max_rows_box = gr.Slider(minimum=20, maximum=2000, value=200, step=20, label="Max rows")
160
+ load_btn = gr.Button("Load / Filter", variant="primary")
161
+
162
+ status_md = gr.Markdown("Loading…")
163
+ table = gr.Dataframe(label="Results", wrap=True, datatype="str", max_height=650)
164
+
165
+ demo.load(build_table, inputs=[sheet_dd, search_box, max_rows_box], outputs=[table, status_md])
166
+ sheet_dd.change(build_table, inputs=[sheet_dd, search_box, max_rows_box], outputs=[table, status_md])
167
+ load_btn.click(build_table, inputs=[sheet_dd, search_box, max_rows_box], outputs=[table, status_md])
168
+ search_box.submit(build_table, inputs=[sheet_dd, search_box, max_rows_box], outputs=[table, status_md])
169
+
170
+ if __name__ == "__main__":
171
+ demo.launch()