Team Ai
Datasetpublic

Benjamin-eecs/openrsi-commit-runtime-assets

sourceHugging Faceupdated 16d agoView on Hugging Face
0likes168downloads
patch.diff179 linesDownload Raw Back to solution
1diff --git a/casparser/parsers/__init__.py b/casparser/parsers/__init__.py2index a56ad4c..f2d50d6 1006443--- a/casparser/parsers/__init__.py4+++ b/casparser/parsers/__init__.py5@@ -90,61 +90,71 @@ def read_cas_pdf(6     # parser / investor extractor calls — every pypdfium2 open re-runs7     # the password decrypt + content-stream parse, so the savings on8     # multi-page detailed statements are significant.9+    # Open the document once and ALWAYS close it before returning.10+    # pypdfium2 tracks pages / text-pages as children of the document;11+    # leaving the document open leaks those handles and makes pdfium emit12+    # "objects still open" at interpreter / library teardown. `close()`13+    # cascades to all child handles created during parsing. The parsed14+    # `data` is plain pydantic models holding no pdfium references, so it15+    # is safe to return after the document is closed.16     doc = _open_document(filename, password)17-18-    file_type = detect_file_type(filename, password, _doc=doc)19-    if file_type == FileType.UNKNOWN:20-        raise CASParseError(21-            "Could not identify the CAS issuer. Supported issuers are "22-            "CAMS, KFintech, NSDL, and CDSL."23-        )24-25-    if file_type in (FileType.CAMS, FileType.KFINTECH):26-        cas_type = detect_cas_type(filename, password, _doc=doc)27-        if cas_type == CASFileType.DETAILED:28-            from . import cams_detailed29-30-            data: Union[CASData, NSDLCASData] = cams_detailed.parse(31-                filename,32-                password,33-                file_type=file_type,34-                _doc=doc,35-            )36-        elif cas_type == CASFileType.SUMMARY:37-            from . import cams_summary38-39-            data = cams_summary.parse(40-                filename,41-                password,42-                file_type=file_type,43-                _doc=doc,44-            )45-        else:46+    try:47+        file_type = detect_file_type(filename, password, _doc=doc)48+        if file_type == FileType.UNKNOWN:49             raise CASParseError(50-                "Could not identify whether this is a DETAILED or " "SUMMARY CAMS / KFin statement."51+                "Could not identify the CAS issuer. Supported issuers are "52+                "CAMS, KFintech, NSDL, and CDSL."53             )54-        if sort_transactions and isinstance(data, CASData):55-            data = _sort_transactions(data)56-    elif file_type == FileType.NSDL:57-        from . import nsdl58 59-        data = nsdl.parse_nsdl(60-            filename,61-            password,62-            file_type=FileType.NSDL,63-            _doc=doc,64-        )65-    elif file_type == FileType.CDSL:66-        from . import cdsl67+        if file_type in (FileType.CAMS, FileType.KFINTECH):68+            cas_type = detect_cas_type(filename, password, _doc=doc)69+            if cas_type == CASFileType.DETAILED:70+                from . import cams_detailed71 72-        data = cdsl.parse_cdsl(73-            filename,74-            password,75-            file_type=FileType.CDSL,76-            _doc=doc,77-        )78-    else:  # pragma: no cover — handled above79-        raise CASParseError(f"Unsupported file type: {file_type}")80+                data: Union[CASData, NSDLCASData] = cams_detailed.parse(81+                    filename,82+                    password,83+                    file_type=file_type,84+                    _doc=doc,85+                )86+            elif cas_type == CASFileType.SUMMARY:87+                from . import cams_summary88+89+                data = cams_summary.parse(90+                    filename,91+                    password,92+                    file_type=file_type,93+                    _doc=doc,94+                )95+            else:96+                raise CASParseError(97+                    "Could not identify whether this is a DETAILED or "98+                    "SUMMARY CAMS / KFin statement."99+                )100+            if sort_transactions and isinstance(data, CASData):101+                data = _sort_transactions(data)102+        elif file_type == FileType.NSDL:103+            from . import nsdl104+105+            data = nsdl.parse_nsdl(106+                filename,107+                password,108+                file_type=FileType.NSDL,109+                _doc=doc,110+            )111+        elif file_type == FileType.CDSL:112+            from . import cdsl113+114+            data = cdsl.parse_cdsl(115+                filename,116+                password,117+                file_type=FileType.CDSL,118+                _doc=doc,119+            )120+        else:  # pragma: no cover — handled above121+            raise CASParseError(f"Unsupported file type: {file_type}")122+    finally:123+        doc.close()124 125     if output == "dict":126         return data127diff --git a/casparser/parsers/cdsl.py b/casparser/parsers/cdsl.py128index 0b82d30..cf3774b 100644129--- a/casparser/parsers/cdsl.py130+++ b/casparser/parsers/cdsl.py131@@ -59,6 +59,13 @@ INE_ISIN_RE = re.compile(r"^IN[E9][0-9A-Z]{8}\d$")132 # `0.196` unit balance prints as `.196`), so the integer part is133 # optional when a decimal part is present.134 NUMERIC_RE = re.compile(r"^-?(?:[\d,]+(?:\.\d+)?|\.\d+)$")135+# A long CDSL folio wraps its tail onto the next display line, which the136+# extractor emits as a separate cell — e.g. folio "91012112582/0" splits137+# into "910121125" | "82/0". The tail is a `<digits>/<digits>` token; we138+# splice it back onto the folio head (the on-page hyphen at the wrap is139+# only a line-break indicator, absent from the authoritative140+# "Folio No :" block).141+FOLIO_TAIL_RE = re.compile(r"\d+/\d+")142 143 PERIOD_RE = re.compile(144     r"(?:for\s+the\s+period\s+from|statement\s+for\s+the\s+period\s+from)\s+"145@@ -543,14 +550,29 @@ def _parse_mf_holdings_row(146 147     # Folio = cell right after ISIN (`<digits>/<digits>` or just digits).148     folio = None149+    folio_end = isin_idx + 1150     if isin_idx + 1 < len(block.cells):151         folio = block.cells[isin_idx + 1].text.strip() or None152 153+    # A long folio wraps its tail into the NEXT cell:154+    #   "910121125" | "82/0" | DIRECT | units | ...155+    # The real folio "91012112582/0" is split across two cells. Splice156+    # the `<digits>/<digits>` tail back on (no separator — the on-page157+    # hyphen is just the wrap indicator); otherwise the folio is158+    # truncated to its head and the tail is mistaken for the159+    # distribution-mode column below.160+    if folio and isin_idx + 2 < len(block.cells):161+        tail = block.cells[isin_idx + 2].text.strip()162+        if FOLIO_TAIL_RE.fullmatch(tail):163+            folio = folio + tail164+            folio_end = isin_idx + 2165+166     # Discriminate 13-cell (has ARN/DIRECT) vs 7-cell (no such column).167-    has_distrib_col = isin_idx + 2 < len(block.cells) and not _looks_numeric(168-        block.cells[isin_idx + 2].text169-    )170-    data_start = isin_idx + (3 if has_distrib_col else 2)171+    # The distribution-mode discriminator is the cell right after the172+    # folio (which may span two cells when it wrapped).173+    disc_idx = folio_end + 1174+    has_distrib_col = disc_idx < len(block.cells) and not _looks_numeric(block.cells[disc_idx].text)175+    data_start = disc_idx + (1 if has_distrib_col else 0)176     numerics = [c.text.strip() for c in block.cells[data_start:] if _looks_numeric(c.text)]177     if len(numerics) < 3:178         return None179