Repository navigation
Expand file tree
/
Copy pathprep_submissions.py
More file actions
279 lines (231 loc) · 10.3 KB
/
Copy pathprep_submissions.py
File metadata and controls
279 lines (231 loc) · 10.3 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
#!/usr/bin/env python3
"""
prep_submissions.py — normalize an EasyChair submissions TSV into the
4-column input table that /bosc-pre-review expects.
Usage:
python3 prep_submissions.py <input.tsv> [--full] [--section SECTION]
Output (stdout):
Clean tab-delimited table with columns: # Title License Source
(plus a header row)
Use --full to preserve extra columns (Authors, Submission Type, etc.) in
addition to the four required ones — helpful for humans reading the
intermediate file.
Use --section to extract only rows from a specific contiguous section of
the file. The test file ships with two sections separated by blank rows
(2026 submissions and 2025 test cases). --section 2026 keeps the first
section, --section 2025 keeps the second.
Validation summary goes to stderr (count of rows, any row-level
warnings, unresolved column-drift cases, etc.).
The script is defensive about two known EasyChair quirks:
1. Column drift: when the chair exports "submissions with action links",
EasyChair interleaves extra columns per row containing strings like
"information on submission N", "review assignment for submission N",
"update submission N". These push the real columns to later positions
and break naive index-based parsing. We detect and strip them.
2. Mixed-section files: a single TSV may contain both a submissions
section and a test/reference section, separated by blank rows and
its own re-header. We detect section boundaries by blank rows.
"""
from __future__ import annotations
import argparse
import csv
import re
import sys
from pathlib import Path
from typing import Optional
EASYCHAIR_ACTION_PREFIXES = (
"information on submission",
"review assignment for submission",
"update submission",
)
REQUIRED_COLS = ("#", "title", "license", "source")
ABSTRACT_COL = "abstract"
def is_action_cell(cell: str) -> bool:
cell_lower = cell.strip().lower()
return any(cell_lower.startswith(p) for p in EASYCHAIR_ACTION_PREFIXES)
def strip_action_columns(row):
"""Remove EasyChair action cells from a row. Returns (clean_row, stripped_count)."""
clean = [c for c in row if not is_action_cell(c)]
return clean, len(row) - len(clean)
def find_col(header, name):
"""Case-insensitive, space/underscore-insensitive header lookup."""
norm_name = name.lower().replace(" ", "").replace("_", "")
for i, h in enumerate(header):
if h.lower().replace(" ", "").replace("_", "") == norm_name:
return i
return None
def split_into_sections(rows):
"""Split rows at blank-row boundaries. Blank = all cells empty/whitespace."""
sections = []
current = []
for row in rows:
if all(c.strip() == "" for c in row):
if current:
sections.append(current)
current = []
else:
current.append(row)
if current:
sections.append(current)
return sections
def parse_section(section):
"""
Parse a section: first non-empty row is header, rest are data rows.
Returns (header, clean_rows, warnings).
"""
warnings = []
if not section:
return [], [], warnings
header = [c.strip() for c in section[0]]
# Strip action cells from header too (belt-and-suspenders)
header, stripped = strip_action_columns(header)
if stripped:
warnings.append(f"header: stripped {stripped} EasyChair action column(s)")
clean_rows = []
for idx, raw in enumerate(section[1:], start=1):
# Strip EasyChair action cells first
row, stripped = strip_action_columns(raw)
if stripped:
warnings.append(f"row {idx}: stripped {stripped} EasyChair action cell(s)")
# Trim trailing empty cells (common artifact of spreadsheet export)
while row and row[-1].strip() == "":
row.pop()
# Pad to header length if short (some exports drop trailing empties)
while len(row) < len(header):
row.append("")
# If still longer than header, warn
if len(row) > len(header):
warnings.append(
f"row {idx}: {len(row)} cells vs header {len(header)}; truncating to header width"
)
row = row[: len(header)]
clean_rows.append(row)
return header, clean_rows, warnings
def extract_four_col(header, rows, with_abstract=False):
"""Return rows reduced to (#, Title, License, Source [, Abstract]). Warn about any row-level data issues."""
warnings = []
cols_to_find = list(REQUIRED_COLS) + ([ABSTRACT_COL] if with_abstract else [])
col_idx = {c: find_col(header, c) for c in cols_to_find}
missing_required = [c for c in REQUIRED_COLS if col_idx.get(c) is None]
if missing_required:
raise SystemExit(
f"ERROR: could not find required column(s) in header: {missing_required}\n"
f"Header seen: {header}"
)
if with_abstract and col_idx.get(ABSTRACT_COL) is None:
print("WARN: --with-abstract requested but no 'Abstract' column found; omitting", file=sys.stderr)
with_abstract = False
out = []
for idx, row in enumerate(rows, start=1):
cells = {c: (row[col_idx[c]] if col_idx[c] < len(row) else "") for c in REQUIRED_COLS}
# Data-quality warnings (not errors — the skill expects to see these)
if not cells["#"].strip():
warnings.append(f"row {idx}: empty '#'")
if not cells["title"].strip():
warnings.append(f"row {idx}: empty Title")
if cells["source"].strip().lower() in ("", "na", "n/a", "none", "-"):
warnings.append(f"row {idx} ({cells['#']}): empty or NA Source")
if re.match(r"https?://", cells["license"].strip()):
warnings.append(
f"row {idx} ({cells['#']}): License looks like a URL — possible data-entry error"
)
# Multi-URL source detection: more than one http(s):// in the source field
source_urls = re.findall(r"https?://\S+", cells["source"])
if len(source_urls) > 1:
warnings.append(
f"row {idx} ({cells['#']}): Source contains {len(source_urls)} URLs — "
f"skill will need to choose which one to evaluate"
)
# Source that isn't a URL at all (orcid, doi, bare domain without scheme, etc.)
elif (
cells["source"].strip()
and not re.match(r"https?://", cells["source"].strip())
and cells["source"].strip().lower() not in ("", "na", "n/a", "none", "-")
):
warnings.append(
f"row {idx} ({cells['#']}): Source is not an http(s) URL "
f"({cells['source'].strip()[:60]!r})"
)
row_out = [cells[c] for c in REQUIRED_COLS]
if with_abstract:
abs_idx = col_idx.get(ABSTRACT_COL)
abs_text = (row[abs_idx] if abs_idx is not None and abs_idx < len(row) else "")
row_out.append(abs_text)
out.append(row_out)
return out, warnings
def main():
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument("input", type=Path, help="EasyChair submissions TSV")
ap.add_argument(
"--full",
action="store_true",
help="Preserve all columns instead of reducing to 4",
)
ap.add_argument(
"--with-abstract",
action="store_true",
help="Include Abstract column (5th column) for better Fresh checking",
)
ap.add_argument(
"--section",
type=int,
default=None,
help="1-indexed section number (if the file has multiple sections)",
)
args = ap.parse_args()
if not args.input.exists():
sys.exit(f"ERROR: input file not found: {args.input}")
with args.input.open(newline="", encoding="utf-8") as f:
# xlsx2csv emits CSV-style double-quote escaping even with -d tab, so
# accept quoted fields here. csv.QUOTE_MINIMAL on read means: if a cell
# is wrapped in quotes, parse it as a quoted cell (unescape "" -> ").
reader = csv.reader(f, delimiter="\t", quotechar='"', quoting=csv.QUOTE_MINIMAL)
all_rows = list(reader)
sections = split_into_sections(all_rows)
print(f"INFO: found {len(sections)} section(s) in {args.input.name}", file=sys.stderr)
for i, sec in enumerate(sections, start=1):
print(f"INFO: section {i}: {len(sec)} row(s) (incl. header)", file=sys.stderr)
if args.section is not None:
if not (1 <= args.section <= len(sections)):
sys.exit(f"ERROR: --section {args.section} out of range (have {len(sections)})")
sections = [sections[args.section - 1]]
def emit(row):
# Write raw tab-separated values without quoting or escaping.
# TSV fields must not contain tab or newline characters; we strip
# any that sneak in from spreadsheet exports.
cleaned = [c.replace("\t", " ").replace("\n", " ").replace("\r", "") for c in row]
sys.stdout.write("\t".join(cleaned) + "\n")
header_written = False
total_data_rows = 0
total_warnings = []
for sec_idx, section in enumerate(sections, start=1):
header, rows, parse_warnings = parse_section(section)
total_warnings.extend(f"[section {sec_idx}] {w}" for w in parse_warnings)
if args.full:
if not header_written:
emit(header)
header_written = True
for r in rows:
emit(r)
total_data_rows += len(rows)
else:
four_col, row_warnings = extract_four_col(header, rows, with_abstract=args.with_abstract)
total_warnings.extend(f"[section {sec_idx}] {w}" for w in row_warnings)
if not header_written:
out_header = ["#", "Title", "License", "Source"]
if args.with_abstract:
out_header.append("Abstract")
emit(out_header)
header_written = True
for r in four_col:
emit(r)
total_data_rows += len(four_col)
print(f"INFO: wrote {total_data_rows} data row(s) to stdout", file=sys.stderr)
if total_warnings:
print(f"WARN: {len(total_warnings)} validation warning(s):", file=sys.stderr)
for w in total_warnings:
print(f" - {w}", file=sys.stderr)
else:
print("INFO: no validation warnings", file=sys.stderr)
if __name__ == "__main__":
main()