"""Uniform row reader with indent level of the label column, for xlsx (openpyxl) and xls (xlrd).""" import re def _label_col(cells): return next((i for i, v in enumerate(cells) if isinstance(v, str) and v and not re.fullmatch(r'[\d.]+', v)), 0) def read_rows(path, sheet_hint='piel', label_col=None, max_col=12): """Return list of dicts: row (1-based), cells, indent (of label cell), bold. Stops after 200 empty rows.""" out, empty = [], 0 if path.lower().endswith('.xlsx'): import openpyxl wb = openpyxl.load_workbook(path, read_only=False, data_only=True) ws = next(w for w in wb.worksheets if sheet_hint in w.title.lower().replace('.', '')) for row in ws.iter_rows(min_row=1, max_col=max_col): cells = [("" if c.value is None else (c.value.strip() if isinstance(c.value, str) else c.value)) for c in row] if not any(str(v) for v in cells): empty += 1 if empty > 200: break continue empty = 0 lc = label_col if label_col is not None else _label_col(cells) c = row[lc] if lc < len(row) else row[0] out.append({'row': row[0].row, 'cells': cells, 'indent': int(c.alignment.indent or 0), 'bold': bool(c.font and c.font.b)}) else: import xlrd bk = xlrd.open_workbook(path, formatting_info=True) sh = next(bk.sheet_by_name(n) for n in bk.sheet_names() if sheet_hint in n.lower().replace('.', '')) for r in range(sh.nrows): cells = [(c.strip() if isinstance(c, str) else c) for c in sh.row_values(r)][:max_col] if not any(str(v) for v in cells): continue lc = label_col if label_col is not None else _label_col(cells) xf = bk.xf_list[sh.cell_xf_index(r, lc)] if lc < sh.ncols else None out.append({'row': r + 1, 'cells': cells, 'indent': xf.alignment.indent_level if xf else 0, 'bold': bool(xf and bk.font_list[xf.font_index].bold)}) return out