Benjamin-eecs/openrsi-commit-runtime-assets
0168
1diff --git a/casparser/parsers/__init__.py b/casparser/parsers/__init__.py2index a56ad4c..f2d50d6 1006443--- a/casparser/parsers/__init__.py4+++ b/casparser/parsers/__init__.py5@@ -90,61 +90,71 @@ def read_cas_pdf(6 # parser / investor extractor calls — every pypdfium2 open re-runs7 # the password decrypt + content-stream parse, so the savings on8 # multi-page detailed statements are significant.9+ # Open the document once and ALWAYS close it before returning.10+ # pypdfium2 tracks pages / text-pages as children of the document;11+ # leaving the document open leaks those handles and makes pdfium emit12+ # "objects still open" at interpreter / library teardown. `close()`13+ # cascades to all child handles created during parsing. The parsed14+ # `data` is plain pydantic models holding no pdfium references, so it15+ # is safe to return after the document is closed.16 doc = _open_document(filename, password)17-18- file_type = detect_file_type(filename, password, _doc=doc)19- if file_type == FileType.UNKNOWN:20- raise CASParseError(21- "Could not identify the CAS issuer. Supported issuers are "22- "CAMS, KFintech, NSDL, and CDSL."23- )24-25- if file_type in (FileType.CAMS, FileType.KFINTECH):26- cas_type = detect_cas_type(filename, password, _doc=doc)27- if cas_type == CASFileType.DETAILED:28- from . import cams_detailed29-30- data: Union[CASData, NSDLCASData] = cams_detailed.parse(31- filename,32- password,33- file_type=file_type,34- _doc=doc,35- )36- elif cas_type == CASFileType.SUMMARY:37- from . import cams_summary38-39- data = cams_summary.parse(40- filename,41- password,42- file_type=file_type,43- _doc=doc,44- )45- else:46+ try:47+ file_type = detect_file_type(filename, password, _doc=doc)48+ if file_type == FileType.UNKNOWN:49 raise CASParseError(50- "Could not identify whether this is a DETAILED or " "SUMMARY CAMS / KFin statement."51+ "Could not identify the CAS issuer. Supported issuers are "52+ "CAMS, KFintech, NSDL, and CDSL."53 )54- if sort_transactions and isinstance(data, CASData):55- data = _sort_transactions(data)56- elif file_type == FileType.NSDL:57- from . import nsdl58 59- data = nsdl.parse_nsdl(60- filename,61- password,62- file_type=FileType.NSDL,63- _doc=doc,64- )65- elif file_type == FileType.CDSL:66- from . import cdsl67+ if file_type in (FileType.CAMS, FileType.KFINTECH):68+ cas_type = detect_cas_type(filename, password, _doc=doc)69+ if cas_type == CASFileType.DETAILED:70+ from . import cams_detailed71 72- data = cdsl.parse_cdsl(73- filename,74- password,75- file_type=FileType.CDSL,76- _doc=doc,77- )78- else: # pragma: no cover — handled above79- raise CASParseError(f"Unsupported file type: {file_type}")80+ data: Union[CASData, NSDLCASData] = cams_detailed.parse(81+ filename,82+ password,83+ file_type=file_type,84+ _doc=doc,85+ )86+ elif cas_type == CASFileType.SUMMARY:87+ from . import cams_summary88+89+ data = cams_summary.parse(90+ filename,91+ password,92+ file_type=file_type,93+ _doc=doc,94+ )95+ else:96+ raise CASParseError(97+ "Could not identify whether this is a DETAILED or "98+ "SUMMARY CAMS / KFin statement."99+ )100+ if sort_transactions and isinstance(data, CASData):101+ data = _sort_transactions(data)102+ elif file_type == FileType.NSDL:103+ from . import nsdl104+105+ data = nsdl.parse_nsdl(106+ filename,107+ password,108+ file_type=FileType.NSDL,109+ _doc=doc,110+ )111+ elif file_type == FileType.CDSL:112+ from . import cdsl113+114+ data = cdsl.parse_cdsl(115+ filename,116+ password,117+ file_type=FileType.CDSL,118+ _doc=doc,119+ )120+ else: # pragma: no cover — handled above121+ raise CASParseError(f"Unsupported file type: {file_type}")122+ finally:123+ doc.close()124 125 if output == "dict":126 return data127diff --git a/casparser/parsers/cdsl.py b/casparser/parsers/cdsl.py128index 0b82d30..cf3774b 100644129--- a/casparser/parsers/cdsl.py130+++ b/casparser/parsers/cdsl.py131@@ -59,6 +59,13 @@ INE_ISIN_RE = re.compile(r"^IN[E9][0-9A-Z]{8}\d$")132 # `0.196` unit balance prints as `.196`), so the integer part is133 # optional when a decimal part is present.134 NUMERIC_RE = re.compile(r"^-?(?:[\d,]+(?:\.\d+)?|\.\d+)$")135+# A long CDSL folio wraps its tail onto the next display line, which the136+# extractor emits as a separate cell — e.g. folio "91012112582/0" splits137+# into "910121125" | "82/0". The tail is a `<digits>/<digits>` token; we138+# splice it back onto the folio head (the on-page hyphen at the wrap is139+# only a line-break indicator, absent from the authoritative140+# "Folio No :" block).141+FOLIO_TAIL_RE = re.compile(r"\d+/\d+")142 143 PERIOD_RE = re.compile(144 r"(?:for\s+the\s+period\s+from|statement\s+for\s+the\s+period\s+from)\s+"145@@ -543,14 +550,29 @@ def _parse_mf_holdings_row(146 147 # Folio = cell right after ISIN (`<digits>/<digits>` or just digits).148 folio = None149+ folio_end = isin_idx + 1150 if isin_idx + 1 < len(block.cells):151 folio = block.cells[isin_idx + 1].text.strip() or None152 153+ # A long folio wraps its tail into the NEXT cell:154+ # "910121125" | "82/0" | DIRECT | units | ...155+ # The real folio "91012112582/0" is split across two cells. Splice156+ # the `<digits>/<digits>` tail back on (no separator — the on-page157+ # hyphen is just the wrap indicator); otherwise the folio is158+ # truncated to its head and the tail is mistaken for the159+ # distribution-mode column below.160+ if folio and isin_idx + 2 < len(block.cells):161+ tail = block.cells[isin_idx + 2].text.strip()162+ if FOLIO_TAIL_RE.fullmatch(tail):163+ folio = folio + tail164+ folio_end = isin_idx + 2165+166 # Discriminate 13-cell (has ARN/DIRECT) vs 7-cell (no such column).167- has_distrib_col = isin_idx + 2 < len(block.cells) and not _looks_numeric(168- block.cells[isin_idx + 2].text169- )170- data_start = isin_idx + (3 if has_distrib_col else 2)171+ # The distribution-mode discriminator is the cell right after the172+ # folio (which may span two cells when it wrapped).173+ disc_idx = folio_end + 1174+ has_distrib_col = disc_idx < len(block.cells) and not _looks_numeric(block.cells[disc_idx].text)175+ data_start = disc_idx + (1 if has_distrib_col else 0)176 numerics = [c.text.strip() for c in block.cells[data_start:] if _looks_numeric(c.text)]177 if len(numerics) < 3:178 return None179 