"""Classification code lists from the regulation tables on likumi.lv: MK 27.12.2005. noteikumi Nr. 1031 (expenditure, EKK), MK noteikumi par budžetu ieņēmumu klasifikāciju (revenue), MK 22.11.2005. noteikumi Nr. 875 (financing). Input: source/klasifikacijas/.html Output: source/kodi/klasdict.json (name -> [[scheme, code], ...]) """ import re, html, json, os ROOT = os.environ.get('BUDGET_ROOT', os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) out = {} CODE = re.compile(r'(?:[A-Z]{1,2})?\d{1,2}(?:\.\d{1,2}){1,5}\.?|\d{4,5}|F\d{8}|[A-Z]\d{1,2}(?:\.\d+)*\.?') for act, kind in (('124833', 'ekk'), ('124831', 'revenue'), ('122159', 'financing')): s = open(os.path.join(ROOT, 'source', 'klasifikacijas', f'{act}.html'), encoding='utf-8', errors='replace').read() n = 0 for tr in re.findall(r'(?is)]*>(.*?)', s): cells = [re.sub(r'\s+', ' ', html.unescape(re.sub(r'<[^>]+>', ' ', c))).strip() for c in re.findall(r'(?is)]*>(.*?)', tr)] for i, c in enumerate(cells[:-1]): if CODE.fullmatch(c) and cells[i + 1] and not CODE.fullmatch(cells[i + 1]) and not cells[i + 1].startswith('Kodā'): code = c.rstrip('.') if kind == 'revenue' and '.' in code: p = code.split('.'); code = (p[0].zfill(2) + ''.join(p[1:])).ljust(5, '0') if kind == 'financing' and '.' in code: code = 'F' + code.replace('.', '') out.setdefault(cells[i + 1].strip(' .;:'), set()).add((kind, code)); n += 1 break print(act, kind, 'rows', n) for k in [k for k in out if re.search(r'Dotācija no vispār|savstarpējie|Kapitālo izdevumu transferti$|7200|uz valsts pamatbudžetu', k)][:12]: print(' ', sorted(out[k])[:3], k[:100]) json.dump({k: sorted(v) for k, v in out.items()}, open(os.path.join(ROOT, 'source', 'kodi', 'klasdict.json'), 'w'), ensure_ascii=False)