From f4e215590d2d96c1a42c6197c0eae605e2ef1ba9 Mon Sep 17 00:00:00 2001 From: cagan Date: Sun, 12 Jul 2026 03:17:19 +0300 Subject: [PATCH 01/29] scrapers: a DKB filing that parses to zero transactions is a failure, not an empty filing Backfilling to 2015 stored 50 disclosures and produced ZERO transactions. The run reported SUCCESS. KAP changed the insider-filing format mid-2016. Measured disclosureType by month: 2016-03 DUY=57 2016-04 DUY=53 2016-05 DUY=138 2016-06 ODA=111 2016-07 ODA=50 2016-08 ODA=108 DUY-era filings are not KAP's structured form. The disclosure body just says the explanation "ekte yer almaktadir" and points at a file the insider mailed in (e.g. "nthol2.pdf") - free-form, often a scan, with no table to read. parse_dkb_transactions expects the ODA form and extracts nothing from them. Nothing said so. The ingest stored the disclosure row, wrote zero transactions, logged disclosure_processed, and finished the run as SUCCESS. This is the same failure mode as the list-endpoint cap: data lost quietly, with a green light on top. A backfill left to run would have spent hours downloading unreadable PDFs and reported success the whole way. Two changes: 1. A DKB disclosure that yields no transactions now logs dkb_yielded_no_transactions with its disclosureType, and a run where most filings come back empty logs dkb_parse_yield_low. The whole point of a Pay Alim Satim Bildirimi is that it reports at least one trade; zero is a parse failure and must read as one. 2. backfill_kap_insider.py defaults to 2016-06-01 - the first month of the parseable format, measured, not preferred. Earlier dates cost hours and yield nothing. Reading DUY-era filings would need OCR plus a free-form extractor. That is a separate project, not a parameter, and it is documented as out of scope rather than left as an implicit gap. At ~1,100 filings/year, 2016-06 to today is ~11,000 disclosures - comfortably past the ~784 events the power gate in signals/base_rate.py requires. The reachable sample was never the constraint; the silent failures were. --- scripts/backfill_kap_insider.py | 20 ++++++++++++-- src/trailing_edge/scrapers/kap/insider.py | 33 +++++++++++++++++++++++ 2 files changed, 51 insertions(+), 2 deletions(-) diff --git a/scripts/backfill_kap_insider.py b/scripts/backfill_kap_insider.py index 571b98a..84ec0bc 100644 --- a/scripts/backfill_kap_insider.py +++ b/scripts/backfill_kap_insider.py @@ -35,8 +35,24 @@ _log = get_logger(__name__) -# ~1,100 DKB disclosures/year measured; ~11 years is what the ~784-event power gate costs. -DEFAULT_START = date(2015, 1, 1) +# 2016-06 is the earliest date this pipeline can actually parse, not a preference. +# +# KAP's insider filings changed format mid-2016. Measured disclosureType by month: +# 2016-03 DUY=57 2016-04 DUY=53 2016-05 DUY=138 +# 2016-06 ODA=111 2016-07 ODA=50 2016-08 ODA=108 +# +# DUY-era filings are a free-form PDF the insider mailed in ("...açıklama ekte yer +# almaktadır", attachment "nthol2.pdf") - often a scan, with no structured table. +# parse_dkb_transactions expects KAP's ODA form and extracts nothing from them: a probe +# run over 2015 stored 50 disclosures and produced ZERO transactions, silently. +# +# Starting earlier than this costs hours of PDF downloads and yields no usable rows. +# Reading DUY-era filings at all would need OCR plus a free-form extractor - a separate +# project, not a parameter. +# +# ~1,100 DKB disclosures/year measured, so 2016-06 → today is ~11,000 filings: well past +# the ~784 events the power gate in signals/base_rate.py requires. +DEFAULT_START = date(2016, 6, 1) # Progressive WAF cooldown: (pre_sleep_seconds, label) # If the warmup GET is disconnected, the WAF has IP-throttled us. diff --git a/src/trailing_edge/scrapers/kap/insider.py b/src/trailing_edge/scrapers/kap/insider.py index a755072..52690d7 100644 --- a/src/trailing_edge/scrapers/kap/insider.py +++ b/src/trailing_edge/scrapers/kap/insider.py @@ -45,6 +45,7 @@ async def run(self, from_date: date, to_date: date) -> ScraperRunResult: run_id: int = run.id seen = inserted = updated = skipped = 0 + empty_dkb = 0 # DKB filings that parsed to zero transactions - see below error_msg: str | None = None try: @@ -92,6 +93,27 @@ async def run(self, from_date: date, to_date: date) -> ScraperRunResult: body = detail.get("disclosureBody", "") or "" txs = parse_oda_transactions(body, ticker=dto.ticker) + # A DKB disclosure that yields no transactions is a parse failure, + # not an empty filing - the whole point of a Pay Alim Satim + # Bildirimi is that it reports at least one trade. Storing the + # disclosure row while quietly dropping its transactions is how + # the pre-2016-06 format gap went unnoticed: every DUY-era filing + # (a free-form attachment, not KAP's structured table) parsed to + # zero, and the ingest reported success. Surface it. + if is_dkb and not txs: + empty_dkb += 1 + _log.warning( + "dkb_yielded_no_transactions", + kap_disclosure_id=kap_disclosure_id, + ticker=dto.ticker, + disclosure_type=disc.get("disclosureType"), + published_at=str(dto.published_at), + hint=( + "DUY-era (pre-2016-06) filings attach a free-form PDF " + "that parse_dkb_transactions cannot read" + ), + ) + async with get_session() as session: repo = KapRepository(session) @@ -151,12 +173,23 @@ async def run(self, from_date: date, to_date: date) -> ScraperRunResult: records_skipped=skipped, ) + # A run that stored disclosures but extracted no transactions from most of them + # is a failed run wearing a SUCCESS label. Surface the ratio, not just the counts. + if empty_dkb: + _log.warning( + "dkb_parse_yield_low" if empty_dkb * 2 >= seen else "dkb_parse_partial", + seen=seen, + empty_dkb=empty_dkb, + empty_pct=round(empty_dkb / seen * 100, 1) if seen else 0.0, + ) + _log.info( "scraper_done", seen=seen, inserted=inserted, updated=updated, skipped=skipped, + empty_dkb=empty_dkb, ) return ScraperRunResult( records_seen=seen, From ee25c9ef2f9c03499a8521c8d81d7631795ee245 Mon Sep 17 00:00:00 2001 From: cagan Date: Sun, 12 Jul 2026 03:56:17 +0300 Subject: [PATCH 02/29] scrapers: pre-2021 filings parsed to zero because the date regex only accepted slashes Five years of history were being thrown away by one character. KAP writes the transaction date both ways, depending on the filing's vintage: 14.06.2019 (dots) - filings up to ~2020 07/06/2023 (slashes) - filings from ~2021 _DATE_RE accepted slashes only: _DATE_RE = re.compile(r"\b(\d{2}/\d{2}/\d{4})\b") So no row in a pre-2021 filing ever anchored, _extract_table_rows returned nothing, and every filing before 2021 parsed to ZERO transactions - while the ingest logged disclosure_processed and finished the run as SUCCESS. The table was always there. Dumped side by side, the 2019 and 2023 layouts carry the identical 9-column row; only the separator differs. parse_kap_date already accepted both formats. Only this regex was turning four years of history away. Measured against live KAP, sampling four filings per year: year before after 2016 0% 100% 2017 0% 100% 2018 0% 75% 2019 0% 100% 2020 0% 75% 2021+ already working, unchanged (fixture test still green) Second bug, found while fixing the first: pdfminer sometimes emits two adjacent table cells on one line ("174.004.552,79 174.269.552,79"). As a single token that fails _TR_NUM_RE, and the collector silently SKIPS non-numeric tokens - so both values vanish, every later column shifts left by two, and post_tx_share_count / post_tx_ownership_pct are read out of the wrong cells. The row still parses, which is what makes it dangerous: it yields plausible-looking wrong numbers. This is the source of the implausible_ownership_pct warnings (an ownership percentage of 24.628.606,69). Such a line is now split - but only when EVERY part is a Turkish number. Splitting unconditionally shreds prose too: "18,45 - 18,48 TL" from the price sentence would yield a numeric token "18,45" that the collector reads as the first table column, and share_count comes back as a unit price. The existing fixture test caught exactly that on the first attempt. Also: the backfill's WAF handling slept 60s before EVERY chunk attempt, including the first, before anything had gone wrong - ~2 hours of a 122-month run spent asleep on the happy path. That is not protection, it is a tax on success. The escalation ladder is kept (60s -> 10min -> 20min) but now fires only AFTER httpx.RemoteProtocolError, which is the only thing a backoff can usefully do. The WAF is real - rapid probing tripped it during this work - and the reactive ladder is what handles it. Tests: 160 passing. The new suite pins the dotted date, the merged cell, and the prose line that must NOT be split. --- scripts/backfill_kap_insider.py | 48 ++++++-- src/trailing_edge/scrapers/kap/parser.py | 35 +++++- .../scrapers/kap/test_table_extraction.py | 115 ++++++++++++++++++ 3 files changed, 186 insertions(+), 12 deletions(-) create mode 100644 tests/unit/scrapers/kap/test_table_extraction.py diff --git a/scripts/backfill_kap_insider.py b/scripts/backfill_kap_insider.py index 84ec0bc..f28f432 100644 --- a/scripts/backfill_kap_insider.py +++ b/scripts/backfill_kap_insider.py @@ -54,10 +54,22 @@ # the ~784 events the power gate in signals/base_rate.py requires. DEFAULT_START = date(2016, 6, 1) -# Progressive WAF cooldown: (pre_sleep_seconds, label) -# If the warmup GET is disconnected, the WAF has IP-throttled us. -# Retry with increasing waits: 1 min → 10 min → 20 min. -_WAF_ATTEMPTS = [(60, "attempt_1"), (600, "waf_retry_2"), (1200, "waf_retry_3")] +# WAF backoff, applied AFTER a block - never before one. +# +# This used to be a list of (pre_sleep, label) pairs that the retry loop slept through +# *before every attempt, including the first*. So every chunk paid a 60s toll even when +# nothing had gone wrong: over 122 months that is ~2 hours of the run spent asleep on the +# happy path. That is not WAF protection, it is a tax on success. +# +# The escalation ladder itself is kept intact - if KAP's WAF does disconnect the warmup +# GET (httpx.RemoteProtocolError = IP throttled), back off 1 min, then 10, then 20. The +# protection is now reactive, which is the only thing a backoff can usefully be. +_WAF_BACKOFF_S = [60, 600, 1200] + +# A courtesy gap between chunks so a long backfill does not arrive as one unbroken +# burst. Small enough to be irrelevant to the runtime (122 x 2s = ~4 min), unlike the +# 60s it replaces. +_CHUNK_GAP_S = 2 def generate_monthly_chunks(start: date, end: date) -> list[tuple[date, date]]: @@ -103,11 +115,24 @@ async def get_completed_chunks() -> set[tuple[date, date]]: async def _run_chunk(from_date: date, to_date: date) -> None: - """Run one chunk with progressive WAF-cooldown retries.""" + """Run one chunk, backing off only if the WAF actually blocks us.""" last_exc: Exception | None = None - for pre_sleep, label in _WAF_ATTEMPTS: - _log.info("chunk_start", from_date=from_date, to_date=to_date, stage=label) - await asyncio.sleep(pre_sleep) + + for attempt in range(len(_WAF_BACKOFF_S) + 1): + if attempt: + backoff = _WAF_BACKOFF_S[attempt - 1] + _log.warning( + "chunk_waf_backoff", + from_date=from_date, + to_date=to_date, + attempt=attempt + 1, + sleep_s=backoff, + ) + await asyncio.sleep(backoff) + + _log.info( + "chunk_start", from_date=from_date, to_date=to_date, attempt=attempt + 1 + ) try: scraper = KapInsiderScraper(backfill=True) result = await scraper.run(from_date, to_date) @@ -121,16 +146,18 @@ async def _run_chunk(from_date: date, to_date: date) -> None: ) return except httpx.RemoteProtocolError as exc: + # Warmup GET disconnected -> KAP's WAF has IP-throttled us. last_exc = exc _log.warning( "chunk_waf_blocked", from_date=from_date, to_date=to_date, - stage=label, + attempt=attempt + 1, error=str(exc), ) + raise RuntimeError( - f"Chunk {from_date}-{to_date} failed after {len(_WAF_ATTEMPTS)} WAF retries" + f"Chunk {from_date}-{to_date} failed after {len(_WAF_BACKOFF_S)} WAF backoffs" ) from last_exc @@ -169,6 +196,7 @@ async def main_async( continue await _run_chunk(from_date, to_date) + await asyncio.sleep(_CHUNK_GAP_S) _log.info("backfill_complete", dry_run=dry_run) diff --git a/src/trailing_edge/scrapers/kap/parser.py b/src/trailing_edge/scrapers/kap/parser.py index 84c8a27..90974fa 100644 --- a/src/trailing_edge/scrapers/kap/parser.py +++ b/src/trailing_edge/scrapers/kap/parser.py @@ -13,7 +13,21 @@ # Matches Turkish formatted numbers: 1.234.567,89 or 1.234 or 18,45 or -2.500.000 _TR_NUM_RE = re.compile(r"-?[\d]{1,3}(?:\.[\d]{3})*(?:,[\d]+)?") -_DATE_RE = re.compile(r"\b(\d{2}/\d{2}/\d{4})\b") + +# KAP writes the transaction date BOTH ways depending on the filing's vintage: +# 14.06.2019 (dots) - filings up to ~2020 +# 07/06/2023 (slashes) - filings from ~2021 +# This pattern used to accept slashes only, so no row in a pre-2021 filing ever anchored +# and _extract_table_rows returned nothing: every filing before 2021 parsed to ZERO +# transactions while the ingest reported success. The table was always there - measured +# side by side, the 2019 and 2023 layouts carry the identical 9-column row - and the only +# difference was this separator. parse_kap_date already accepted both formats; only this +# regex was turning four years of history away. +# +# A dotted date cannot be confused with a Turkish number: _TR_NUM_RE requires groups of +# exactly 3 digits after a dot, and "14.06.2019" has 2. +_DATE_RE = re.compile(r"\b(\d{2}[./]\d{2}[./]\d{4})\b") + _PRICE_RANGE_RE = re.compile(r"([\d]+[,.][\d]+)\s*-\s*([\d]+[,.][\d]+)\s*TL") # Known Turkish column header aliases per logical field (for header-driven column detection). @@ -58,9 +72,26 @@ def _extract_table_rows(text: str) -> list[list[str]]: """ # Flatten to a clean token list tokens: list[str] = [] + # One line is normally one cell. But pdfminer sometimes emits two adjacent table cells + # on a single line ("174.004.552,79 174.269.552,79"). As one token that fails + # _TR_NUM_RE, and the collector below silently *skips* non-numeric tokens - so two + # values vanish, every later column shifts left by two, and post_tx_share_count / + # post_tx_ownership_pct are read out of the wrong cells. That is worse than a crash: + # it yields plausible-looking wrong numbers (the source of the + # implausible_ownership_pct warnings, e.g. an ownership percentage of 24.628.606,69). + # + # Split such a line, but ONLY when every part is a Turkish number. Splitting + # unconditionally would shred narrative text too - "18,45 - 18,48 TL" from the price + # sentence would yield a numeric token "18,45" that the collector would happily read + # as a table column. Requiring the whole line to be numeric keeps prose out. for line in text.splitlines(): line = line.strip().replace("\xa0", "") - if line: + if not line: + continue + parts = line.split() + if len(parts) > 1 and all(_TR_NUM_RE.fullmatch(p.lstrip("-")) for p in parts): + tokens.extend(parts) + else: tokens.append(line) rows: list[list[str]] = [] diff --git a/tests/unit/scrapers/kap/test_table_extraction.py b/tests/unit/scrapers/kap/test_table_extraction.py new file mode 100644 index 0000000..3b09639 --- /dev/null +++ b/tests/unit/scrapers/kap/test_table_extraction.py @@ -0,0 +1,115 @@ +"""Row extraction from the DKB transaction table. + +Two bugs lived here, both silent, both costing years of history or corrupting the rows +that did survive. These tests pin the exact text shapes that broke them. +""" +from trailing_edge.scrapers.kap.parser import _extract_table_rows + +# A pre-2021 filing: the date uses DOTS. The table is otherwise identical to a modern one +# (9 numeric columns). Taken from a real 2019 TUKAS filing. +LEGACY_2019 = """ +Islem Tarihi +14.06.2019 +1.845.337 +0 +1.845.337 +107.266.000 +109.111.337 +39,34 +39,34 +40,02 +40,02 +( * ) Net satis olmasi durumunda net nominal tutar +""" + +# A modern filing: the date uses SLASHES, and pdfminer has merged two adjacent cells +# onto one line ("174.004.552,79 174.269.552,79"). +MODERN_2023_MERGED = """ +Islem Tarihi +07/06/2023 +265.000 +0 +265.000 +174.004.552,79 174.269.552,79 +49,72 +49,72 +49,79 +49,79 +""" + +# The narrative price sentence. Its numbers must NEVER be mistaken for table cells. +NARRATIVE_PRICE = """ +paylari 18,45 - 18,48 TL fiyat araligindan satilmistir +25/05/2026 +0 +2.500.000 +2.500.000 +50.000.000 +47.500.000 +20,10 +20,10 +19,10 +19,10 +""" + + +def test_dotted_dates_anchor_a_row(): + """The whole pre-2021 archive hung on this. + + _DATE_RE accepted slashes only, so no row in a 2016-2020 filing ever anchored, + _extract_table_rows returned nothing, and every one of those filings parsed to ZERO + transactions while the ingest reported success. Measured against live KAP, the fix + took 2016-2020 from 0% parse yield to 75-100%. + """ + rows = _extract_table_rows(LEGACY_2019) + + assert len(rows) == 1 + assert rows[0][0] == "14.06.2019" + assert rows[0][1] == "1.845.337" # buy nominal + assert len(rows[0]) == 10 # date + 9 columns + + +def test_slashed_dates_still_anchor_a_row(): + """The modern format must keep working - the fix widens, it does not replace.""" + rows = _extract_table_rows(MODERN_2023_MERGED) + + assert len(rows) == 1 + assert rows[0][0] == "07/06/2023" + + +def test_merged_cells_are_split_so_columns_do_not_shift(): + """pdfminer sometimes puts two cells on one line. + + As a single token that line fails the number pattern, and the collector *skips* + non-numeric tokens - so both values vanish and every later column shifts left by two. + The row still parses, which is what makes it dangerous: post_tx_share_count and + post_tx_ownership_pct come back as plausible-looking numbers read from the wrong + cells (the source of the implausible_ownership_pct warnings). + """ + rows = _extract_table_rows(MODERN_2023_MERGED) + row = rows[0] + + assert row[4] == "174.004.552,79" # start nominal - both halves survived + assert row[5] == "174.269.552,79" # end nominal, NOT shifted into by the merge + assert row[8] == "49,79" # end capital % lands in its own column + assert len(row) == 10 + + +def test_narrative_numbers_are_not_read_as_table_cells(): + """Splitting merged lines must not shred prose. + + An unconditional whitespace split turns "18,45 - 18,48 TL" into a numeric token + "18,45", which the collector reads as the first table column - so share_count comes + back as a unit price. Only fully-numeric lines may be split. + """ + rows = _extract_table_rows(NARRATIVE_PRICE) + + assert len(rows) == 1 + assert rows[0][0] == "25/05/2026" + assert rows[0][2] == "2.500.000" # the real sell nominal + assert "18,45" not in rows[0] + assert "18,48" not in rows[0] + + +def test_text_with_no_table_yields_no_rows(): + assert _extract_table_rows("aciklama ekte yer almaktadir\nEk dosyalar\n1- nthol2.pdf") == [] From 00305ac13a6daf7c41b413557c3f2cfdeb09f698 Mon Sep 17 00:00:00 2001 From: cagan Date: Sun, 12 Jul 2026 04:11:01 +0300 Subject: [PATCH 03/29] scrapers: correct the parse floor to 2015-01 - the DUY diagnosis was a confound The floor in the previous commit (2016-06) was justified by the DUY->ODA format split: pre-2016-06 filings looked like free-form attachments with no table. That reasoning was wrong, and the error is worth recording because it is a textbook confound. A 2015 backfill did produce zero transactions. But the date-separator bug in parser._DATE_RE would have zeroed those years regardless of their format. Two causes, one symptom - and I attributed it entirely to the one I found first. Re-probing with the fixed parser separates them: year docs with a table parsed type 2013 0/3 0% DUY <- genuinely no table in the document 2014 0/3 0% DUY <- genuinely no table 2015 3/3 100% DUY <- parses fine; DUY was never the problem 2016 1/3 33% ODA DUY-era 2015 filings carry the same structured table as a modern one. The format label was a red herring. What actually ends is the table: 2013/2014 filings are a free-form PDF the insider mailed in, often a scan, with nothing to extract. Reading those would need OCR plus a free-form extractor - a separate project, not a parameter. The floor moves to 2015-01-01, gaining ~1.5 years. If some early-2015 filings do turn out to be table-less, they now surface as dkb_yielded_no_transactions rather than vanishing silently: a measured claim with a safety net under it, not an assumption. 2015-01 -> today is ~12,500 filings, well past the ~784 events the power gate requires, and long enough to test regime-conditionally instead of pooling a decade into one number. --- scripts/backfill_kap_insider.py | 38 +++++++++++++++++++++------------ 1 file changed, 24 insertions(+), 14 deletions(-) diff --git a/scripts/backfill_kap_insider.py b/scripts/backfill_kap_insider.py index f28f432..2c729b3 100644 --- a/scripts/backfill_kap_insider.py +++ b/scripts/backfill_kap_insider.py @@ -35,24 +35,34 @@ _log = get_logger(__name__) -# 2016-06 is the earliest date this pipeline can actually parse, not a preference. +# 2015-01 is where the documents stop having a table to read, measured - not a preference. # -# KAP's insider filings changed format mid-2016. Measured disclosureType by month: -# 2016-03 DUY=57 2016-04 DUY=53 2016-05 DUY=138 -# 2016-06 ODA=111 2016-07 ODA=50 2016-08 ODA=108 +# An earlier version of this comment blamed the DUY/ODA disclosureType split and set the +# floor at 2016-06. That was wrong, and worth recording because the mistake was a +# confound: a 2015 backfill did produce zero transactions, but the date-separator bug in +# parser._DATE_RE would have zeroed those years regardless of their format. Two causes, +# one symptom. Re-probing with the fixed parser separates them: # -# DUY-era filings are a free-form PDF the insider mailed in ("...açıklama ekte yer -# almaktadır", attachment "nthol2.pdf") - often a scan, with no structured table. -# parse_dkb_transactions expects KAP's ODA form and extracts nothing from them: a probe -# run over 2015 stored 50 disclosures and produced ZERO transactions, silently. +# year docs with a table parsed type +# 2013 0/3 0% DUY <- genuinely no table in the document +# 2014 0/3 0% DUY <- genuinely no table +# 2015 3/3 100% DUY <- parses fine; DUY was never the problem +# 2016 1/3 33% ODA # -# Starting earlier than this costs hours of PDF downloads and yields no usable rows. -# Reading DUY-era filings at all would need OCR plus a free-form extractor - a separate -# project, not a parameter. +# So the format label was a red herring: DUY-era 2015 filings carry the same structured +# table as a modern one. What actually ends is the table itself - 2013/2014 filings are +# a free-form PDF the insider mailed in ("...açıklama ekte yer almaktadır", attachment +# "nthol2.pdf"), often a scan, with nothing to extract. Reading those would need OCR plus +# a free-form extractor: a separate project, not a parameter. # -# ~1,100 DKB disclosures/year measured, so 2016-06 → today is ~11,000 filings: well past -# the ~784 events the power gate in signals/base_rate.py requires. -DEFAULT_START = date(2016, 6, 1) +# If some early-2015 filings do turn out to be table-less, they now surface as +# dkb_yielded_no_transactions rather than vanishing silently. The floor is a measured +# claim with a safety net under it, not an assumption. +# +# ~1,100 DKB filings/year measured, so 2015-01 → today is ~12,500 filings: well past the +# ~784 events the power gate in signals/base_rate.py requires, and long enough to test +# regime-conditionally rather than pooling a decade into one number. +DEFAULT_START = date(2015, 1, 1) # WAF backoff, applied AFTER a block - never before one. # From ba363bdee37c664782111a697ed2c9052ec0339a Mon Sep 17 00:00:00 2001 From: cagan Date: Sun, 12 Jul 2026 04:43:25 +0300 Subject: [PATCH 04/29] scrapers: validate every row against the form's own arithmetic; parse the 2015-2020 blotter 21% of stored transactions were silently wrong. Fixed indices assumed the canonical 9-column layout on every row; a missing or merged cell shifted every later column and the row still "parsed" - share counts of 3.02 (a percentage), ownership percentages of 25,923,015 (a share count), and 4 stored transactions with share_count=0 (the totals line of the table). The old policy for an implausible percentage was to null THAT field and keep the rest of the row, which is precisely backwards: a row that has proven its columns shifted cannot be trusted in any field. Two changes. 1. Canonical rows must now prove their layout before anything is stored, using the arithmetic the form itself guarantees: start_nominal + (buy - sell) == end_nominal |buy - sell| == |net| every percentage in [0, 100] A shifted layout essentially cannot satisfy the first identity by accident. Rows that fail are logged as dkb_row_rejected with a reason and dropped - a rejected row is recoverable from the logs, a silently corrupt one poisons every statistic built on it. Zero-volume totals lines and flat round-trips (net zero) are rejected too. Mixed intraday rows now net to the dominant side; the old code called any row with sell > 0 a SELL even when buys dominated. The token collector also gets a skip budget: it used to wander arbitrarily far into the document stitching together narrative numbers, which is where the 2015-era garbage came from. (find_column_index / header aliases stay unused for this table: in the pdfminer stream the headers are far from the cells and multi-line, while position-proven- by-arithmetic is strictly stronger for a fixed-form table.) 2. The 2015-2020 filing is a different document, now parsed: a per-trade blotter (date | Alim/Satim | adet | fiyat | tutar | holdings), not the netted one-row-per-day form used from ~2021. pdfminer emits its cells in an unstable order - the 2015 fixture interleaves column-major and row-major in one table - so per-trade positional reconstruction is guesswork. Two anchors are order-independent and self-checking, and only those are used: TOPLAM ALIS / TOPLAM SATIS -> (buy qty, sell qty, buy amount, sell amount) narrative price range -> implied avg price amount/qty must fall inside it Both fixtures check out to the last kurus: TEKFEN 2015, six buys netting 120.000 shares for 526.950 TL (avg 4,39125, inside the traded 4,36-4,41); METRO 2018, 479.593 bought at avg 1,1458 (narrated range 1,11-1,16) and 69.368 sold at exactly 1,13. Per-filing net summaries are also the honest granularity - the modern form nets same-day trades into one row anyway. Post-transaction holdings are NOT order-independently recoverable and stay NULL rather than guessed. Routing is by the TOPLAM marker, which no modern fixture contains. Both era PDFs are committed as fixtures (public KAP filings), so this never has to be developed against the live WAF-guarded endpoint again. Tests: 175 passing. The validator suite pins every rejection reason; the blotter suite pins the real numbers read off the two era documents. --- src/trailing_edge/scrapers/kap/parser.py | 289 ++++++++++++++---- tests/fixtures/kap/dkb_era2015.pdf | Bin 0 -> 10743 bytes tests/fixtures/kap/dkb_era2018.pdf | Bin 0 -> 8234 bytes .../unit/scrapers/kap/test_legacy_blotter.py | 81 +++++ tests/unit/scrapers/kap/test_parser.py | 15 +- .../unit/scrapers/kap/test_row_validation.py | 108 +++++++ 6 files changed, 437 insertions(+), 56 deletions(-) create mode 100644 tests/fixtures/kap/dkb_era2015.pdf create mode 100644 tests/fixtures/kap/dkb_era2018.pdf create mode 100644 tests/unit/scrapers/kap/test_legacy_blotter.py create mode 100644 tests/unit/scrapers/kap/test_row_validation.py diff --git a/src/trailing_edge/scrapers/kap/parser.py b/src/trailing_edge/scrapers/kap/parser.py index 90974fa..5d0b0c5 100644 --- a/src/trailing_edge/scrapers/kap/parser.py +++ b/src/trailing_edge/scrapers/kap/parser.py @@ -100,15 +100,22 @@ def _extract_table_rows(text: str) -> list[list[str]]: m = _DATE_RE.fullmatch(tokens[i]) if m: date_tok = tokens[i] - # Collect up to 9 numeric tokens after the date + # Collect up to 9 numeric tokens after the date. In a real table row the + # cells are consecutive, so tolerate at most a few non-numeric interruptions + # (page-break artifacts) - an unbounded skip used to let the collector wander + # arbitrarily far into the document and stitch together numbers from + # unrelated narrative, which is where the 2015-era garbage rows came from. numerics: list[str] = [] + skips_left = 3 j = i + 1 while j < len(tokens) and len(numerics) < 9: tok = tokens[j] if _TR_NUM_RE.fullmatch(tok.lstrip("-")): numerics.append(tok) elif tok and not _DATE_RE.fullmatch(tok): - pass # skip non-numeric, non-date tokens + skips_left -= 1 + if skips_left < 0: + break else: break j += 1 @@ -171,12 +178,218 @@ def _relation_type_from_context(text: str) -> str: return RelationType.KENDISI +# The canonical DKB table row: date + exactly these 9 numeric cells, in this order. +# Verified identical on 2019 (dotted dates) and 2023+ (slashed dates) filings. +# [0] buy nominal [1] sell nominal [2] |net| +# [3] start nominal [4] end nominal +# [5] start capital % [6] start vote % [7] end capital % [8] end vote % +_ROW_NUMERIC_COUNT = 9 +# Display values are exact decimals; the tolerance only absorbs rounding of the +# printed cells, not real mismatches. +_NOMINAL_TOL = Decimal("1") + + +def _map_canonical_row(numerics: list[str]) -> tuple[dict | None, str]: + """ + Map the 9 numeric cells of a table row to transaction fields - or reject. + + Position is the schema here (the form is fixed), but position alone is exactly + what produced 21% silently-wrong rows: a missing or merged cell shifts every + later column and the row still "parses", yielding share counts like 3.02 and + ownership percentages in the millions. So the mapping must PROVE the positions + are right before anything is stored, using the arithmetic the form itself + guarantees: + + start_nominal + (buy - sell) == end_nominal + |buy - sell| == |net| + every percentage in [0, 100] + + A shifted layout essentially cannot satisfy the first identity by accident. + Rows that fail are rejected with a reason - a rejected row is recoverable from + the logs, a silently corrupt one poisons every statistic built on top of it. + """ + if len(numerics) != _ROW_NUMERIC_COUNT: + return None, f"expected {_ROW_NUMERIC_COUNT} cells, got {len(numerics)}" + + try: + buy, sell, net, start_nom, end_nom = (parse_turkish_number(v) for v in numerics[:5]) + pcts = [parse_turkish_number(v) for v in numerics[5:9]] + except InvalidOperation: + return None, "unparseable numeric cell" + + if buy < 0 or sell < 0: + return None, "negative buy/sell nominal" + if buy == 0 and sell == 0: + # A Pay Alim Satim Bildirimi row with no volume is a totals/summary line, + # not a transaction. These used to be stored as BUY with share_count=0. + return None, "zero volume (summary row)" + + signed_net = buy - sell + if abs(abs(signed_net) - abs(net)) > _NOMINAL_TOL: + return None, "net column disagrees with buy-sell" + if abs(start_nom + signed_net - end_nom) > _NOMINAL_TOL: + return None, "start+net != end (shifted columns?)" + if any(p < 0 or p > 100 for p in pcts): + return None, "percentage outside [0,100]" + if signed_net == 0: + # Equal intraday buy and sell: no position change, no directional signal. + return None, "flat round-trip (net zero)" + + return { + "transaction_type": "BUY" if signed_net > 0 else "SELL", + "share_count": abs(signed_net), + "post_tx_share_count": end_nom, + "post_tx_ownership_pct": pcts[2], # end capital % + }, "" + + +# The 2015-2020 filing is a different document: a per-trade blotter +# (date | Alım/Satım | adet | fiyat | tutar | pre/post holdings), not the netted +# one-row-per-day form used from ~2021. pdfminer emits its cells in an unstable +# order - the 2015 fixture interleaves column-major and row-major within one +# table - so reconstructing individual trades positionally is exactly the kind of +# guesswork that produced silent corruption before. Two anchors in the document +# are order-independent and self-checking, and the parser uses only those: +# +# 1. The TOPLAM ALIŞ / TOPLAM SATIŞ summary block: the next four numerics are +# (buy qty, sell qty, buy amount, sell amount), verified on both era +# fixtures against the per-trade rows they summarise. +# 2. The narrative price range ("1,11 - 1,16 TL fiyat aralığından"): the +# implied average price amount/qty must fall inside it. +# +# A per-filing net summary is also the honest granularity: the modern form nets +# same-day trades into one row anyway, and the signal pipeline keys on +# (insider, ticker, date, direction, size). Post-transaction holdings are NOT +# extractable order-independently, so they are left NULL rather than guessed. +_LEGACY_BUY_LABEL = "TOPLAM ALI" # prefix-match: trailing Ş varies with encoding +_LEGACY_SELL_LABEL = "TOPLAM SATI" +_LEGACY_PRICE_MIN = Decimal("0.01") +_LEGACY_PRICE_MAX = Decimal("100000") + + +def _parse_legacy_blotter( + text: str, + ticker: str, + insider_name: str, + relation_type: str, +) -> list[KapInsiderTxDTO]: + lines = [ln.strip().replace("\xa0", "") for ln in text.splitlines() if ln.strip()] + + sell_idx = next( + (i for i, ln in enumerate(lines) if ln.upper().startswith(_LEGACY_SELL_LABEL)), + None, + ) + if sell_idx is None: + _log.warning("dkb_row_rejected", reason="legacy: TOPLAM SATIŞ label not found") + return [] + + # First four numeric cells after the totals labels: buy qty, sell qty, + # buy amount, sell amount (order verified on 2015 and 2018 fixtures). + numerics: list[Decimal] = [] + for ln in lines[sell_idx + 1:]: + for tok in ln.split(): + if _TR_NUM_RE.fullmatch(tok.lstrip("-")): + try: + numerics.append(parse_turkish_number(tok)) + except InvalidOperation: + pass + if len(numerics) >= 4: + break + if len(numerics) < 4: + _log.warning("dkb_row_rejected", reason="legacy: totals block incomplete") + return [] + buy_qty, sell_qty, buy_amt, sell_amt = numerics[:4] + + # Latest trade date in the document. The blotter's own rows and the narrative + # carry the same dates, so a plain scan is order-independent. + dates = [] + for tok in _DATE_RE.findall(text): + try: + dates.append(parse_kap_date(tok)) + except ValueError: + pass + if not dates: + _log.warning("dkb_row_rejected", reason="legacy: no parseable trade date") + return [] + tx_date = max(dates) + + # Narrative price range, if present, bounds the implied average price. + range_lo = range_hi = None + pm = _PRICE_RANGE_RE.search(text) + if pm: + try: + range_lo = parse_turkish_number(pm.group(1)) + range_hi = parse_turkish_number(pm.group(2)) + except InvalidOperation: + pass + + txs: list[KapInsiderTxDTO] = [] + for tx_type, qty, amt in (("BUY", buy_qty, buy_amt), ("SELL", sell_qty, sell_amt)): + if qty == 0 and amt == 0: + continue + if qty <= 0 or amt <= 0: + _log.warning( + "dkb_row_rejected", + reason=f"legacy: {tx_type} qty/amount inconsistent", + qty=float(qty), + amount=float(amt), + ) + continue + avg_price = (amt / qty).quantize(Decimal("0.0001")) + if not (_LEGACY_PRICE_MIN <= avg_price <= _LEGACY_PRICE_MAX): + _log.warning( + "dkb_row_rejected", + reason="legacy: implied price implausible", + avg_price=float(avg_price), + ) + continue + if range_lo is not None and range_hi is not None and range_lo > 0: + if not (range_lo * Decimal("0.9") <= avg_price <= range_hi * Decimal("1.1")): + _log.warning( + "dkb_row_rejected", + reason="legacy: implied price outside narrative range", + avg_price=float(avg_price), + range_lo=float(range_lo), + range_hi=float(range_hi), + ) + continue + txs.append( + KapInsiderTxDTO( + insider_name=insider_name, + relation_type=relation_type, + ticker=ticker, + transaction_date=tx_date, + transaction_type=tx_type, + share_count=qty, + price_try=avg_price, + post_tx_share_count=None, + post_tx_ownership_pct=None, + ) + ) + _log.info( + "dkb_legacy_summary", + ticker=ticker, + tx_type=tx_type, + qty=float(qty), + avg_price=float(avg_price), + ) + return txs + + def parse_dkb_transactions( pdf_bytes: bytes, ticker: str = "", insider_name: str = "", ) -> list[KapInsiderTxDTO]: - """Extract transactions from a Java-unwrapped DKB PDF.""" + """Extract transactions from a Java-unwrapped DKB PDF. + + Two document generations, routed by an unambiguous marker: filings carrying a + TOPLAM ALIŞ/SATIŞ totals block (2015-2020) go through the legacy blotter + summary; everything else through the canonical one-row-per-day table. Every + accepted row is validated against arithmetic the form itself guarantees; + rows that fail are logged as dkb_row_rejected and dropped - never stored + with guessed fields. + """ from pdfminer.high_level import extract_text text = extract_text(io.BytesIO(pdf_bytes)) @@ -186,6 +399,11 @@ def parse_dkb_transactions( relation_type = _relation_type_from_context(text) + # Legacy (2015-2020) blotter form: route by its unambiguous totals marker. + # The modern canonical form never contains it (checked on both modern fixtures). + if _LEGACY_BUY_LABEL in text.upper(): + return _parse_legacy_blotter(text, ticker, insider_name, relation_type) + # Extract price range from narrative price_try: Decimal | None = None pm = _PRICE_RANGE_RE.search(text) @@ -201,56 +419,25 @@ def parse_dkb_transactions( for row in rows: try: tx_date = parse_kap_date(row[0]) - # columns: date, buy_nominal, sell_nominal, net, start_nom, end_nom, - # start_cap_pct, start_vote_pct, end_cap_pct, end_vote_pct - buy_nominal = parse_turkish_number(row[1]) if len(row) > 1 else Decimal(0) - sell_nominal = parse_turkish_number(row[2]) if len(row) > 2 else Decimal(0) - - if sell_nominal > 0 and buy_nominal == 0: - tx_type = "SELL" - share_count = sell_nominal - elif buy_nominal > 0 and sell_nominal == 0: - tx_type = "BUY" - share_count = buy_nominal - elif sell_nominal > 0: - tx_type = "SELL" - share_count = sell_nominal - else: - tx_type = "BUY" - share_count = buy_nominal - - end_nom: Decimal | None = None - if len(row) > 5: - try: - end_nom = parse_turkish_number(row[5]) - except InvalidOperation: - pass + except ValueError as exc: + _log.warning("dkb_row_rejected", reason=str(exc), row=row) + continue - end_cap_pct: Decimal | None = None - if len(row) > 8: - try: - end_cap_pct = parse_turkish_number(row[8]) - except InvalidOperation: - pass - if end_cap_pct is not None and not (0 <= end_cap_pct <= 100): - _log.warning("implausible_ownership_pct", raw=row[8], parsed=float(end_cap_pct)) - end_cap_pct = None - - txs.append( - KapInsiderTxDTO( - insider_name=insider_name, - relation_type=relation_type, - ticker=ticker, - transaction_date=tx_date, - transaction_type=tx_type, - share_count=share_count, - price_try=price_try, - post_tx_share_count=end_nom, - post_tx_ownership_pct=end_cap_pct, - ) + fields, reason = _map_canonical_row(row[1:]) + if fields is None: + _log.warning("dkb_row_rejected", reason=reason, row=row) + continue + + txs.append( + KapInsiderTxDTO( + insider_name=insider_name, + relation_type=relation_type, + ticker=ticker, + transaction_date=tx_date, + price_try=price_try, + **fields, ) - except Exception as exc: - _log.warning("dkb_row_parse_error", row=row, error=str(exc)) + ) return txs diff --git a/tests/fixtures/kap/dkb_era2015.pdf b/tests/fixtures/kap/dkb_era2015.pdf new file mode 100644 index 0000000000000000000000000000000000000000..cf99570829785cda2ddd4d1ecd02a6adfb6c9a86 GIT binary patch literal 10743 zcmeHN2{e>#-AIp;cM4CmtzF zvM0N;#Ul}t5Jlfjo9BIczW001dB1bM@0@22XXc*kzyAKe>z@0%{?~RhU4klH4JEMs zRewX103RF;fkEhGR{;$TsF4c|ATeEh0Y?ZNY6OA9)KN&NDFm(#)q@}rNI294f`D3r z-=h#Xs6NCFf0%jzp8w>QPL#KMU(O3`^ z)Qkx@x=`43CIpVq(t;9PsBC}A`hj2wg1=>>ge09vvNBZ5Ubyfkdg}^wbGDaBUb4hrz1DkT|V_Py?DH;B5yc z4o5CtIME?+6d0g6)CizCvyVVf*u`i(7xP5IpyqTs8v@5H30Q#GKoD4{r4Ivu+L!}Q z&;xdG2#5iQ0E^88NNxgrN4m*mM_~sCvcrXyBK%h&k*G@!87oD0mB3(dMD0og7>vAq z0ZE$^5O(0of%EhU@PW{MQ~m}bi%nv(y_tZM03S?%Z~JyVQ-T2B4`4w&)qmot2T{lV z5-kW0^*tdh25JF8ssABH2L}gk89N>V4^RDOetZ5kk5*qHksV0AM;xBH>R%!WbD8fZ zIq~oWeG%TYV=cFkL_tSR%7D}(>1&q;PCx`u7fA}jIY`*uU4ifED_2Qyy{h1Q{EJMM zaQntCV}MO^B(X`5MaG#!O-OD43xa?zU6_%a!3#KY>0$|89Z-@=cZOQK0A2tStcMf8 z1ZWgsku_#a7aH4wLZZ6RoT1tbhA!Yl@}RP{z}I&k!k0m$L#MF;8XL5aTe7diq%-!= zz3moRiN>KJSQO$QbT5y7u{6_9a1(p~fIz zNq;f}5*u{&Jri?)MfYHS4H~36ooT@!QGkE(vdDHYLhxGx$R)&5hy(-#@s*PVBm}WU zK2XezNvBuzmHkHv^E1AZgj)#DT!RhP;UGGH9@%i*Ggg z#WW+hFj;H}3ih2m>Np%Y?=fF3nUY;C*lx>-RC{H_60Nr3 ztg+N99sRCW+gP@Hr$B;VrO9)Qzr=atJ=U)%iNT}&{fVL>NX3}*(I0GHQl2ZED)e6~ z87__J3-%u!BOi{bhCQz8%kH@oL?Snn!Vh4rI}EMx#*9mMDr{a`9B-M0sQYSeQ5{eD zbhdQke88u&!b0rI%PGH+!$Y#7RKl(%)E|-KF1pe5R{ybk^mYrKPvJ6roN1p@`Ny;weB%0-S~MddA?9EpB1k^k@< zrg&aU(8DFAm}rhLq|NwIlg-z^Lb<3o_R^Ko>{C+no8x6S_7}v)NJ;92sH-NK-!Fdc zGuA({Uh|x4(P(~Fs-5om6Tdvsz-{Z^c<2teReFftx!w_doTFK1TeDXylzKCt^+I(` z=^!1Qe{Y7ocM9)WZ#Zr5nLmG|4b|kE+wE5C&zjzE$LSFNz_D{hb~{bDOc$od_1%$` z3jRzx%x(Em`dTd6tygqzl|pdyCKC;B{yQ_wH*WgxeLp@>QSv=aL-#1Wt$l4!CVMm~ z*u73p<8JouaAnI9KFb{Q_0|`FprAvF67y1lTYLng_y=P74+m$e`s0q{y!$c$@gzDV zY5*R#sr$16&1XEx>K*;acHg*7dA6T!xb-v-r#`HtvT}W@Zwi*ZO`Ga+k~kUY5O?0R z=kUI`(GzLF>!W$r*{Rg=g29ZYd@X@I``?6&GD{F6Q>%^*UOAYe=@ehH@pbvVUEKU! z4h3G%esxD&ya)aD?ECf;1V^A?KWCYDeBFV4^nQ%+!%qilfh;?Rn;R9wBc|nLFI=8m z-)zlyOHar?&cJ3LGX+b;p3H1Zm=>KaY>S_^T)lZNyghP2^~M%H9eF)LT;kw_`^UXE zB{3?38CBb)(Wf&fEY6f(^u6LAv}=C6Q?#^TqVGJ{<8HnFE}_10+iAV>08+@kxu`>! z^T4V8Xon6#X*21F(BL}NAY*|s|1xSBr6&E2>&WdjV-vJ@&xATT{ISG4S3OSqaS;{b z?&oYdKD&MBQ6KS3+PZ~Ah0(B{DPDz55o@EiXZp3 zKRZeK(H`(jD!m(j<+qXNm8}Dge=9d{xf=dp1M8FIS*zWr>c@aezcuO5(aE>>R03GL zG~Oh#`#$EfqNi+oXH1R}#RrKcn*%yZ0*apSCW^1zcXqzj_PonewQ>Qc%X(1XvC;gp z3jX^;hJ+c<7k#CY`@#X|BFT)6gpqnrq;Bf`I7FcJNech&(PO-QdCCi%vp=?Ss!g7NTLdBsv=0y(YMwpJ#*loOzQ%IE5!?25$Z;tR4tZkv)P zEXT~tx!2_IU#7FlH*TVY3|@M=p@4eJ{ARwM^s5yfUFN2$gT{wAw=2n!ccz~5oq0^& zm%T?0$l8-kvaT>3u~Ql!p4{&75n=DLHUKcQd_UhHQ@s9}{R5G(16q6pgyMC-YQwvS zHLFE!5%Wrwyw|mOucaqg2AsJ!mwhnd_N~q{2aR((q%Std#+t2ENFYt-$&hJBW%&C$ zuE~>N8qSS7=YJZwu8w!*T`1f3FxLq0_*Xu8F#?8TH}_xQ zc{w9^a-9fGc0%;EY141=$AbWo z-n1&mh@;A$E-)LqR}Q?~y6RxsV8fY&p5znP7};djA>V8534NAXv+L$h@2cKl{8l0} z`GVE9`VUpVr(Zam{{C>{yzf+YNY%((J5TUiq-uTKF{2lLLjgi3hUO}HPz{3f7=Du% z!aSolukoHy;pdV3`0Q=3!}teT>@)sB;?57Rc{?Bn`ZW`&g#N;htXrd39P#J%&4dWQSv(SBv+ChBKudj$Den{ACGNcABp~d zKD9v!D9F-N=YQ<+Z4r%s`}={w!m!JKKl0)mF3RO?tQ?;|@KMAiJn}{gHL9jkn_Dz{ zZNM9;0T_AHk?0s>`J4;WTk8Ze>?eh&qAwaUm`;V=lQ~6sB6fM~w);J=XDcgdWgiLRLx7iTAS*9C?fU(8BvULp$B%ftTZ#y+6CF7b=9@ykvb`!*;^Eeg4ueKj|W3(ebWOI3Rji$elD6 zb?i&(;V;L$3dBFfRrZ_iO?(-=)imko1axNn;mfJMiw|eXGow8($L|+C_9UEZV3ahq zo_NYQaleSry>e@3$#nN&shxAqz2m2(DZfYb-&bC=^4 zS_GcBClpmV#9tIz>ppFwS>PqeSvzju^De5tsc-CKVoK7GkK2XOn%HcC;ObLJy9K*M zS`iu9Y>#RG4%*tczs|pY;9o&gA%s$%V1OiLzip;)|*G>~r0Pf!!~I=8F_& z2oqhY^Ks)^DK2gCb=OG|zm{=REF`UV^*zlA)8}XpSVC|C%#-a30R@06t#?o znoex`gncl?x$VIWDXx8V^k88HwG5IYz+sCZ4W7L?`W7*?;S1wtxssXANOq8 zQeLw4;jZ%|afx9kPimPScOzNto44zkZ|N#J7k3W8iMn4sHnHkvQ5AgyOCeG7dhzeJ z>8ervmhH@20*+$28cvO>UQqj-&&K;|`pW{W6ztS~f96s_d*Dyd8jrhTq0mxQ4^{m# z`tf5;pMm|>E#vg=tqVTKC-{MRrDG{GB@2GX3p777KlE1$o!ZqL%7q*DSamncjFg#i zi3sB+i)nIOs%5!SH^h9*OgFekxW8Id(0}B<&_(q-3dQ%*-JUDA9Hu@`{VaF=tywEG zO~BEbigVnU`%;h?S>)o1)2QWqG0C?-cE*MN;G_|$hV;B-sNgDRi>c>v;+FCl`C{hY z^HVA=d|tk3%8u3=rm>}*LvhsL#w5m_hWq2!b7%_%8j@43F+p@?IiIvMFJijXnuQ7% zu=P9T&k<$Y`U}Dlm*ah{8ngLf!CU)_10iJ= z_kM3?BseBaWJV07`9p#5kG;*gY3nY8(yv5p@~tkuK^wB#(sC(wNTpXi=ZIh5nuzSr z0#QA0%LBM}bUC3Kx~OlDTIG?NIVUqahteLwfpBT<0l;a>D6D$)PDH!M>ox*L+A{ha zZ<-ulxpyB?PU|Xt+;w0aVrZ;a_AH;j%usf(>&+a0O6?6CPo5ERA+Tz{X<0=eiBnJB z-bn2AOJbl`*_k|aRI=7FbJjd)!yy_<1;*R9NQxE6c(nS~HQvH@qAL&I!%2J8w{z^k z;YA>BPQ~h&=McT94M}1xl2Y!JRQzO!XAuOUKcWI6;F^qR7|0GD4tdNPrrI{g+{v*q zU`MtR#e#%tnM83>sl%4eA>CAbOH#cNN6+;}PHuhn5x>VfEFw$#-h*l@sGgB{P6Ze^ zIG`VPEmCRHNWSt`{m^czrE|hke}a!1%Pi;ptuBg-c%xjVONh7tQ zyY&(pcj{Te_(`^ya!@_rNAu;mUP*X6i1Bb5Rx#heS^b$TRtJ>;s^6>I`|anz{p;$@ zL44fU*1ge7xq|id$`uT6E=`!#W|O3!{N;tEI4Y5u?QZU}zfR^1cY&yG#LB$*Wwo{YR$J>wQQ7*( z9U^NAueE=?fv0?MYD)Q5TK~hfH)dDgT%=wNGzH0%*+Cp;i=~Fv^GREXRjqGkR%=|Y z0%OMuD|2p}IA?vICzcP3Q^QNn$ay3MUdNxqBR#~DL+iQ2;K73u%?MytY7wI-i9ip0 z@Ks&sQ=ei=TZE~&*@$Etfirm1)uV|OAFPeSneEH*m5<|iMB@72HIJm^Wz;)mU3jW; z4V}w6X}YP4kRQz=EP_PHxr!n z!Y)}of)@bQ*(>8HY-d-A*1pvD#56gCa&@Ajy?}PqWmwFG7U7pZW6+U$jXMH5v_oAT zwH0I3x3T>h1~2TF`0NhyIi@S-^`H1W77~i>7^*kA-KBP(pRd#VF_MyRpR>Aw*OqBucTS z2ZodPkq;!beDOL2+B0B_wJ+xUIO;MSEH`8Jf!y``*3pNKYsAkz2FUN(QCrn!_(v8} z@AcTv6YYy<`{@FU^Y7;TOV(uz&GS?+4-)Cw}U;bizGOi+wD;RR&cTy zZDsiV^87r!1sI$bd1-n8-I z&6!~zW;a!;^$d!OI8KH*C(5BOhgeP&z2%rdhYy$n@MJZc)08@*JjjcO+Z8olcF)tgq; z?rRfjsAbhE*9S~?A%ZXLi${tCiy9ey(6+TV5563XSZDY?STw}wRMP4|?A z{&)F-{hV*xOgIF%wTt|#{17NJm>CRZ2}ACOq0BR3`@v*q_SG7zhPA%qyCB41dxd&^ zU`@Xhx}bhuy?K=q%4|*P75MX2t~}M{ogI2ov1e@N)`#!k-C5eK`xt`h6xLY-dBios z+$`{mz?Zk7zP4eIa10h|;X!6EHCv30zJ~d|i|})Gqp`~y+*Lmcupwl?*@dPlGhS38 z195TGlsSMlh8Z(-fFmyYzD&Tv*TjF9(v*Cl*&2VQB)9AUE=cqr7%%S+7*sYYiyL*Y1Z zyB&srA`q&egeuF2#wK~I(pa)zjeNJG3$Q3m7Y5sfPJ=AkC6VbKY)u)NpIstQ@>^bR zj316@EVZQ^)hKi~s5glLg{#4!zfxi>i|!9f-~gzE{H2m3<%c7N2a~##k0S*NPyx5a z_6w*7|D_&S2V>(Os(*0zJw`V-=nnuEF#uKYI$*yIYz~6|S|IpnRTxGUj<$qj@o*R( zg;8EC0Zaq>FOok2`e}pCw4~FiT81PR17Mm13_26+742XyeSj-2Gx1IEUu+oD9bKG! zeia82kA^LeW0~Z?+W6KR`yW|ZF1ySQSQSSc}pc(4Rxg7b)_fK{WXkY^rZ1sJ&ivZAY zbu^l!io>a67wxL6l1MPLDu8r^}o(4C&;o1^K~{ZeG5)U@J;~z zaVuCF0{Qj7`}&#i|DvCNoct|t|AFftxc(Lbe@plucKrj_-$LMT3ID^c{}^0+KNl0= zSu9N%FL0^wi^a;~ft-K0%0ho%WMMGiBI_>~kqDIa*HvUav`%?H?D-6T(3(crb7V8J zt$C*2NHGk9;#FE>4qFwTHMi~R>g2fc~8mm;*Hn0p-3|bgj2m}m;h9ObtTo?)lZ^FQ=)nR6B0e7pR&)*rUsOhWn z8LB>oDk)l@Z`{G1!%sn>r%p^-!0KQaw13=svnz?_vGKOI!X+bY_up{1<$TvU!Y zhZycY{X($ZmrKTzdu?~#?$e|#=Z{I<9Sc-GU$aI=LV~oaX;-GiZvPcr!o9pZeu2=k Nvx9Ir=Ev#5{{q5vbbbH; literal 0 HcmV?d00001 diff --git a/tests/fixtures/kap/dkb_era2018.pdf b/tests/fixtures/kap/dkb_era2018.pdf new file mode 100644 index 0000000000000000000000000000000000000000..d06b383c2b665dd822598412e1a21820c4d15a18 GIT binary patch literal 8234 zcmd5i2|SeB+qZ>^E2)bXfZl%@9(?6@An_SxA&aqJm)#*Jlp#`=ShufZmEOSB}l0~ zZ@YO{3JYN%E;xS@OoA^@bRd@jQg5qiR46R;DqP6%5C zgDIw?FSF5?l{X*}#N|8kXbc2Q`fgQAKpsFXu@FH--x3EAM;myEgf@)854J^^Y?>)| zohJr>he#VOX#xbL39mwe!Vn>wK^r}b2<8|{7ozMyd4h#4 zI1Db6&GCa>*&Gv&fGzHi(?xeOMt3z$AT;x*@gW?M0RORw37<_1(lO-*F>RgxY#n|m zBI3W8v{)YEK5Wiru@F=oh`k&Q|=5i2N7|w--{(J<1SwLtoiyel*0`@vsfP{fL z2uca&umK_iNQeWGh_DH43Y)>^umx-huY^~@RuGv8+rX<~Ti6b^hp7++2RopO1Uthn z5JeAmhdm%YGE9T?^k4?eguzk}7DIsjVK#V^;XpVD4u&}}7v@1&ECJ@j0$2!#!eI~= zhlL~H^$-@!G-Ch(2L;g}96+&WvH?p3YypJBVPI!*54K`*g)|1r7TEC*d^{Egps~q- zTYhvtI2?o{kYFHOF&!cf4i)!cKqd$g2-vV-Y$DJk*pkg36EF~iE1N0w7kJ{xWS{_} ze-w(IsD;PlA@o6^=s4O&-{?amlK$2so-d|N#^IoE9}4o6#~+?xnPIWhkCV-KYT=6S=Y3o=Ip3_hDD z*8CRCR#fB@eo@&!toROgHIs|3%)=>LWB=n2GN9wIRZix#K=e> zTbpyiYDc#FU}s;jp0?UW_3S$;3E1iFQ(rH(c!2k}T+VIn?tS%am&fvaE&bHunU_w!m|A);@R!ZM z*w483Np8e;pDCuudn2^sWVq*skDK}`QY$i+RQV?OT-B&NCTB6JYPZQ68A{$K&EqYV zBrB(-uO#QDEp>nRkg$n#K64AHHB_-lAoW z7docCx-j*uiDcvp*+!eelys}Kp8_wJ9XYD{#sWV7C2aSuPN_|qD(i<{Ln*Z7NXvs@ z$#jVvTfZ~&AfFL~RV^~^E4}UWIlk`;^3S>p&wJnAQAFO_D{R=tn>0!JBxSWPHo`ag zINf}f=3bc`kT&gO%C5T4Atx&7@0JxQJYIDuyD)me9wmCJgOyQr zU&oV>Pt@fXsAfJpR?DuqNIS1;y!O<(>a2#gxLrxJil{H-q+b7Abv??-s&vZHB;##Z zuWkKJJmI3ab2D@*pUW8;vi*i4;*;8>)}Pm|=)SkGPa!epLVlT&Zq6JV2cNPfZ+rRc zZe9h~2PU_d86Duy$zsg;IdXY}5Q#X@*G+jeY=FHIuQaw#+Fp_n9MGucSZ2oNNGwiNvuaepK-Za_d%p|A# z4ohz4`>P!B*gay)O^fMIt*@GQ@aHPIClp1_09`+JSJ{UTXZR-uWhYfdX$(tlc$a@5 z_Dws!LneBz(EWhz_FQE?P4i|p=jD@?LBs|ZD+Qk*gEdmHY6}ajqI3Fv)xL7x2HVXM8 zGBf#RSAPk>2YV8oV{I~H^~3uV5z&aPOZIB@r#A@9ggpr?&XU5CNJ*{q~993Q*E=gFNV~r z6m>bw`edIc`^i!D=oN)ZaD2FlvSxf6&qXi5Da%DqaBmsS?#MdjCZZ!#e~D{uv`4P+ zoTbD2T8tDoZ_YVrqi;^Zx%_RD)#c}9axNV(Ju&la%o*v`zrGCUznrnFn`LDiVJZ#MZ)_9I802yW|n!3U^Y^J~yO@bXDG+cN1 zCPzg3V?_S-)L0Y&I>7($xiM@>07(^k(7%%9&tKLB!8F9{=EqE`<(fk{0XF{ z6Y?iCei)BM#PP*nynpz1Vnz26<6*Lh)#5{pem!w`khuQGuV-m`Jx8fVKIY@QUrJ{{ zliI?Xd|aE7CS_mTsbuh?r)Qbu^Hpi7hn3PyvlQzaTGyv5Qg?sODA;}XV`HJZ$%-EJ zyrCP*^ST;cbsCJxk&lDhue=DMR~)N97?yq^cp!nmp+A8gPdRr{Z45nW?jQC8i>z$c?=ELilxvbak zKR`*_eUlYK=)HB`BhQ;g)+%$FIg#NTBj4AmYzTMpJ{4L1WYhV1Rl6!j-o1#uQ1Qlf z0cToH)rPVM{@txiN@c6p+kLaXnv zFAM3KYo^Iwb$RwE#96c7<;JyBhnm?>PcPm(WvDLp(XO6~FL~WxT3)W4;ck%dcB84X zSJqJA>7jLRbeYDL&F8=NBlcbbV(cf?M{hj_`FJ_I-M-O$Kd1WMVEh#g zd|&xu!a&z2cI3z)&FyhQ^gX}3Z~He!goKv8sf)AE-l}SOUtxCp!UB1LwZ}TU-k4BN z=VN?pGUM|`JNoKmX&?7SyP4S9+I5AVDX9paTFBgZF95T)+uEi(w7Mo8d&+m1AlX(v zm*PKs|JnX8Lj(0u-n~zhBR5m*HwGU%)-har|H<0Ob1hFYA1=1>?!CP*VUhZU)iayDR8z z-nB@B)-S>`zm(XD^e=f;aLvk()W+}!f(Ij?hx!M)`tIJ|rEEO)>n-x+!?%#Bnz!RH z)TQ|@N|UUll_wQ!b5m=KT7CAxox`q|btWZFdHS~Mb<5Y`!yiiv)8AVs-Xy(pKA%k8 zf5u|w)n5a0ujROyweNTJ=w&*$_p>-7;Uxa+8LUn1UrC93YOF>yL$s&qG^lS5knM?A zKhO7!D^4LKYA@d_Rj_X$`AA4^(-9AP3ES&ccSpKxW4&^o#++7tT>tNuYFzD<=xA;1 z+ww>QyGQD=?kaU*kFULWUAD%eW?_!OjpUuiy85=Q<%Ll>w;vh|-m2KqeUtoU2zh7} zRzBZJqinyB+7z^J=fdknA#BO2XNxs*D=w6N@pOH@lV7gtr0JFtLw)mnr;W=axL|{> znx4v>u(}W17pjG%Oxw3LP-3W9A;9M`IYU2s$s-M~^oyS}4t#l<{HVO{O2fhh)VW3S zwyQnL;%OR5rNc9Bo9pfD{JF8#_!Uu`YPg?k zf}d=8bY{b^)|X4SY;u<_wZGq8G5fumx$XRgzn=O{vM5jfseMdH)>1ER{nQfPv?U3yVlkyp){-Jt6l-R18+&XB7uMhlffirU<7-|s)`7FG^9t?IU( z9;$1aTQEJj%X^YC!> zpbY$0;JI~9m;UTS?rDD8ShH_z!{$j7Y+8P<=P%Xoom+F`c-yWWGbPfr-(=2Dv%*w( z$rQ}-veC~EJZ)XGli4(zV5Y(;jy`bK(o?ZcS8=P9L|gH!F0ZZD-g4^qE2h>z6}Ipz zCDYiVlaFda`*lYjDsu=&FdG<@NzXV_5cnx{%QC8Bk^FL?1F=z>r}i5!UY~ks-dmnT z@%yM{j(l}JU;U^X*M#b)#fmo6NS`JC5v-+RXA$=Q$J?yRR9C0s5)UPEkNZSJ0> z>!rOO$X<3-_U|zaJWkTN$@)%gG1a!rrQPUmvxJJ4p9}82ZtBV3Zr)6&S3MxZ;4E#v zk;e*paip`1p+q=^nniH4($~nGfnGV1PwyMo5PZB}N}WmQmSC%CrS&fT&3l48pR@e0 z6~182Zy4u6d+pIJs|&*8(=QF4?cmFh@46{0ozWyUfau(-jN6-6opo&h;cV-{_lBQp zZ3vJd&v94Cmu}j=NL}Y~)ld7}>kd+vmq>l!$(O#rz03)4#}HURtlYA^?YQ_wEdDmb zo+JB>(qaOD_D8x)1My3r1KO`@%-J@)VM|TD`F@Fl%ca@TRzUc^W5Oq^sPJ0aSo7;p z;mZxLW+*1)A8@};2)lIGCpwn5@N;61^yc0+dQkB|`>}PM-Mx?Ato7MgrMour!y~C| z^~IhWt2DARkF(Xft~Y2bj7&&&YW#g9@9s8o|IP5gkBbtmk)Etz#b-k7BfeECwM2Ku zvsR0BxR}HFy}1{RC0g#*mTMd-?D{j4l9kt&+T+jMT*lDW3Ey(_IQTql zcVV~=NEE)Ry<#`xJi|-7EV+~#PFKl1R2Nc4KU6c%f>o>Sa zQ8P@;c+o$YSv^~?*+u?Abv)Viov!quR*U8|?YB!3k2m8DDja;2ZIhgDa#4a_ZOpq zHtgo^0pdFf5fov(#g~~_aYe>c6o&sP4jL7X1#)II-YQWw0$eJBP!8Oanxe%X9TJWV z>EOxW-UG<{4X{q3XcDKclS;C4CS}LQ`6ePhibCY!pFqUY?Zx(55@@;m08y7!uV4 zze(b--`tV|p@XU`{-)aW9r2|*`n$|P{xKVdNcw9c%nwKVD z)WUjcX`!`JW+(&UYqHToDI`i^A$kOab;7$jX~DK!ra8Fx(=^w|VXzbo2}{6$8(AD# z8-v%tU^KK~Gjs_ymopIykJTek@HjEnI2+c1qe%gi(USs}Gxh`5_e4AqVnKhvQ0AdO zh$DvKDX2Im!f*r%ki$e62AI_kFgy|5gigf65-305ktl>8)*xdrz*9_|i>LfJmw?9u z0Z+su0A_s;BV*CZ*hD-MnGAB}?_dS`qL!O2tA`6ctvG6QB8KLV@Sab}HL1!`$426!ca4e>gA&W)F0xSf> zAY++yJdwmClE_#Fi$ bytes: + return (FIXTURES / "dkb_era2015.pdf").read_bytes() + + +@pytest.fixture(scope="module") +def era2018_pdf() -> bytes: + return (FIXTURES / "dkb_era2018.pdf").read_bytes() + + +def test_2015_tekfen_filing_nets_to_one_buy(era2015_pdf): + """Six intraday buys (10+20+10+20+25+35 = 120 thousand) net to one BUY summary.""" + txs = parse_dkb_transactions(era2015_pdf, ticker="TKFEN") + + assert len(txs) == 1 + tx = txs[0] + assert tx.transaction_type == "BUY" + assert tx.share_count == Decimal("120000") + assert tx.transaction_date == date(2015, 6, 15) + # 526.950 / 120.000 = 4,39125 TL - inside the traded range 4,36..4,41 + assert Decimal("4.36") <= tx.price_try <= Decimal("4.41") + + +def test_2015_holdings_are_null_not_guessed(era2015_pdf): + """pdfminer emits this table in an unstable interleaved order, so post-transaction + holdings cannot be recovered order-independently. NULL is the honest value - + the previous parser stored numbers from whatever cells it happened to land on.""" + tx = parse_dkb_transactions(era2015_pdf, ticker="TKFEN")[0] + assert tx.post_tx_share_count is None + assert tx.post_tx_ownership_pct is None + + +def test_2018_metro_filing_yields_buy_and_sell(era2018_pdf): + """A filing with both directions produces two summary transactions.""" + txs = parse_dkb_transactions(era2018_pdf, ticker="METRO") + + assert len(txs) == 2 + by_type = {t.transaction_type: t for t in txs} + + buy = by_type["BUY"] + assert buy.share_count == Decimal("479593") + # 549.526,49 / 479.593 = 1,1458 - inside the narrated 1,11-1,16 range + assert Decimal("1.11") <= buy.price_try <= Decimal("1.16") + + sell = by_type["SELL"] + assert sell.share_count == Decimal("69368") + assert sell.price_try == Decimal("1.13") # 78.385,84 / 69.368 exactly + + assert all(t.transaction_date == date(2018, 6, 13) for t in txs) + + +def test_2015_insider_name_is_extracted(era2015_pdf): + txs = parse_dkb_transactions(era2015_pdf, ticker="TKFEN") + assert "YATIRIM HOLD" in txs[0].insider_name.upper() + + +def test_modern_fixture_does_not_take_the_legacy_path(dkb_pdf_bytes): + """The 2026 EGEPO fixture has no TOPLAM block; it must still parse canonically + with full holdings data - the legacy route must not swallow modern filings.""" + txs = parse_dkb_transactions(dkb_pdf_bytes, ticker="EGEPO") + assert txs + assert txs[0].post_tx_ownership_pct is not None diff --git a/tests/unit/scrapers/kap/test_parser.py b/tests/unit/scrapers/kap/test_parser.py index 7ad8b5f..b51204e 100644 --- a/tests/unit/scrapers/kap/test_parser.py +++ b/tests/unit/scrapers/kap/test_parser.py @@ -69,8 +69,15 @@ def test_oda_parser_does_not_crash(oda_html): assert isinstance(result, list) -def test_post_tx_ownership_pct_implausible_value_becomes_none(monkeypatch, dkb_pdf_bytes): - """Validation guard clamps an astronomic ownership % (e.g. 5,260,000) to None.""" +def test_row_with_implausible_ownership_pct_is_rejected_whole(monkeypatch, dkb_pdf_bytes): + """An out-of-range percentage means the columns are misaligned - the WHOLE row goes. + + The previous policy nulled the percentage and kept the rest of the row. That is + how 21% of stored transactions ended up with silently wrong share counts: a row + that has already proven its columns shifted cannot be trusted in any field. + Rejecting loses one recoverable row; keeping it poisons every statistic built on + top. This test pins the stricter contract. + """ orig = parser_mod._extract_table_rows def _inject_implausible(text: str): @@ -82,9 +89,7 @@ def _inject_implausible(text: str): monkeypatch.setattr(parser_mod, "_extract_table_rows", _inject_implausible) txs = parse_dkb_transactions(dkb_pdf_bytes, ticker="TEST") - assert txs, "Expected at least one transaction row from fixture" - for tx in txs: - assert tx.post_tx_ownership_pct is None or tx.post_tx_ownership_pct <= 100 + assert txs == [], "a row with an impossible percentage must be dropped, not patched" def test_column_mapping_missing_header_returns_none(): diff --git a/tests/unit/scrapers/kap/test_row_validation.py b/tests/unit/scrapers/kap/test_row_validation.py new file mode 100644 index 0000000..73da779 --- /dev/null +++ b/tests/unit/scrapers/kap/test_row_validation.py @@ -0,0 +1,108 @@ +"""_map_canonical_row: the arithmetic proof that the columns are what we think. + +Position is the schema in a DKB table, and position alone produced 21% silently-wrong +rows in production (share_count=0 summary lines, percentages stored as share counts, +ownership percentages in the millions). The validator uses the identities the form +itself guarantees - start + (buy - sell) == end, |buy-sell| == |net|, percentages in +[0,100] - so a shifted layout cannot survive by accident. +""" +from decimal import Decimal + +from trailing_edge.scrapers.kap.parser import _map_canonical_row + +# The NASMED fixture row, verbatim: sell of 2.5M, holdings 98M -> 95.5M. +GOOD_SELL = ["0", "2.500.000", "-2.500.000", "98.000.000", "95.500.000", + "19,6", "22,25", "19,1", "21,99"] +# A real 2019-style buy: 107.266.000 + 1.845.337 = 109.111.337. +GOOD_BUY = ["1.845.337", "0", "1.845.337", "107.266.000", "109.111.337", + "39,34", "39,34", "40,02", "40,02"] + + +def test_valid_sell_row_maps(): + fields, reason = _map_canonical_row(GOOD_SELL) + assert reason == "" + assert fields["transaction_type"] == "SELL" + assert fields["share_count"] == Decimal("2500000") + assert fields["post_tx_share_count"] == Decimal("95500000") + assert fields["post_tx_ownership_pct"] == Decimal("19.1") + + +def test_valid_buy_row_maps(): + fields, _ = _map_canonical_row(GOOD_BUY) + assert fields["transaction_type"] == "BUY" + assert fields["share_count"] == Decimal("1845337") + + +def test_short_row_is_rejected(): + """A partial row is precisely the case fixed indices used to mis-map.""" + fields, reason = _map_canonical_row(GOOD_BUY[:5]) + assert fields is None + assert "expected 9" in reason + + +def test_zero_volume_summary_row_is_rejected(): + """Production had 4 stored transactions with share_count=0 - totals lines.""" + fields, reason = _map_canonical_row( + ["0", "0", "0", "1.000.000", "1.000.000", "5", "5", "5", "5"] + ) + assert fields is None + assert "zero volume" in reason + + +def test_shifted_columns_fail_the_arithmetic_and_are_rejected(): + """Drop one cell and shift the rest left - the exact production corruption. + + The row still has plausible-looking numbers everywhere, which is why it survived + before. start + net == end is what unmasks it. + """ + shifted = GOOD_BUY[1:] + ["40,02"] # sell slid into buy's slot, etc. + fields, reason = _map_canonical_row(shifted) + assert fields is None + + +def test_net_column_disagreement_is_rejected(): + row = list(GOOD_BUY) + row[2] = "999" # |net| != |buy - sell| + fields, reason = _map_canonical_row(row) + assert fields is None + assert "net column" in reason + + +def test_percentage_out_of_range_rejects_the_whole_row(): + """Production stored ownership percentages like 25.923.015,31 - a share count + that landed in a percentage column. The row must go entirely: if one column is + provably misaligned, share_count is not trustworthy either.""" + row = list(GOOD_BUY) + row[7] = "25.923.015,31" + fields, reason = _map_canonical_row(row) + assert fields is None + assert "percentage" in reason + + +def test_flat_roundtrip_is_rejected(): + """Equal buy and sell: no position change, no directional information.""" + fields, reason = _map_canonical_row( + ["500", "500", "0", "1.000.000", "1.000.000", "5", "5", "5", "5"] + ) + assert fields is None + assert "net zero" in reason + + +def test_mixed_day_nets_to_the_dominant_side(): + """Intraday buy AND sell on one row: the honest direction is the net one. + + (The old code called any row with sell>0 a SELL, even if buys dominated.) + """ + fields, _ = _map_canonical_row( + ["3.000", "1.000", "2.000", "10.000", "12.000", "1", "1", "1,2", "1,2"] + ) + assert fields["transaction_type"] == "BUY" + assert fields["share_count"] == Decimal("2000") + + +def test_negative_nominal_is_rejected(): + row = list(GOOD_BUY) + row[0] = "-1.845.337" + fields, reason = _map_canonical_row(row) + assert fields is None + assert "negative" in reason From 0b895de71fa13204e068b64eb3bd792424b4bfe7 Mon Sep 17 00:00:00 2001 From: cagan Date: Sun, 12 Jul 2026 04:53:23 +0300 Subject: [PATCH 05/29] scrapers: let the checks choose the totals binding; parse prices era-aware The first clean backfill rejected ~65% of 2015-era filings. The reject log tallied to three defects, all in the legacy blotter path added one commit earlier: 18x "implied price outside narrative range" 18x "qty/amount inconsistent" 10x "totals block incomplete" Root causes, in order of blame: 1. The price regex could not read two of the three period spellings. It captured "1.234" out of "1.234,56 TL" (parse_turkish_number then read it as one thousand two hundred) and read the pre-2016 dot-decimal "4.36" as 436. Either way the implied average landed "outside" a narrative range that had been misparsed, and a correct filing was rejected. _parse_price_token now treats a dot followed by 1-2 digits as a decimal point; only exactly-3-digit dot groups are thousands. 2. One narrated range was applied to both directions. The filing template narrates the buy leg first, so the first range bounds the buy side and the last bounds the sell side; with a single range and both directions active, only the buy is range-checked. A sell narrated without its own range is no longer rejected against the buy's. 3. The TOPLAM cells' order is not stable: pdfminer emits (buy qty, sell qty, buy amt, sell amt) in some files and (buy qty, buy amt, sell qty, sell amt) in others. The parser assumed the first and rejected the second. Guessing the order is how columns got silently mis-read before, so it is not guessed: both bindings are validated against the same checks and the one that uniquely survives wins. Both-or-neither surviving with different substance rejects the filing loudly. This turns "position we hope is right" into "position the arithmetic proved", same principle as the canonical-row validator. avg_price now quantizes ROUND_HALF_UP like every other Decimal in the codebase. Tests: 186 passing, each defect pinned with a synthetic filing, including the case that must KEEP failing (a buy avg genuinely outside its own narrated range). --- src/trailing_edge/scrapers/kap/parser.py | 133 +++++++++++++----- .../unit/scrapers/kap/test_legacy_binding.py | 125 ++++++++++++++++ 2 files changed, 219 insertions(+), 39 deletions(-) create mode 100644 tests/unit/scrapers/kap/test_legacy_binding.py diff --git a/src/trailing_edge/scrapers/kap/parser.py b/src/trailing_edge/scrapers/kap/parser.py index 5d0b0c5..ed8e94d 100644 --- a/src/trailing_edge/scrapers/kap/parser.py +++ b/src/trailing_edge/scrapers/kap/parser.py @@ -3,7 +3,7 @@ import io import re -from decimal import Decimal, InvalidOperation +from decimal import ROUND_HALF_UP, Decimal, InvalidOperation from trailing_edge.core.logging import get_logger from trailing_edge.core.time import parse_kap_date @@ -28,7 +28,26 @@ # exactly 3 digits after a dot, and "14.06.2019" has 2. _DATE_RE = re.compile(r"\b(\d{2}[./]\d{2}[./]\d{4})\b") -_PRICE_RANGE_RE = re.compile(r"([\d]+[,.][\d]+)\s*-\s*([\d]+[,.][\d]+)\s*TL") +# Narrated prices come in three spellings across eras: "1,15" (comma decimal), +# "1.234,56" (thousands + comma decimal), and "4.36" (dot decimal, pre-2016 filings). +# The old pattern ([\d]+[,.][\d]+) matched only "1.234" out of "1.234,56" - which +# parse_turkish_number then read as one thousand two hundred - and read "4.36" as 436. +# Both produced false "price outside narrative range" rejections downstream. +_PRICE_NUM = r"(?:\d{1,3}(?:\.\d{3})+,\d+|\d+,\d+|\d+\.\d+)" +_PRICE_RANGE_RE = re.compile(rf"({_PRICE_NUM})\s*-\s*({_PRICE_NUM})\s*TL") + + +def _parse_price_token(s: str) -> Decimal: + """Price-aware number parse: a dot followed by 1-2 digits is a DECIMAL point. + + parse_turkish_number treats every dot as a thousands separator, which is right + for nominal amounts ("2.500.000") and wrong for old-era dot-decimal prices + ("4.36" must be 4.36 TL, not 436). Only exactly-3-digit dot groups are + thousands; anything else is a decimal written with a dot. + """ + if re.fullmatch(r"\d+\.\d{1,2}", s.strip()): + return Decimal(s.strip()) + return parse_turkish_number(s) # Known Turkish column header aliases per logical field (for header-driven column detection). _COLUMN_ALIASES: dict[str, list[str]] = { @@ -267,6 +286,26 @@ def _map_canonical_row(numerics: list[str]) -> tuple[dict | None, str]: _LEGACY_PRICE_MAX = Decimal("100000") +def _check_side( + qty: Decimal, + amt: Decimal, + price_range: tuple[Decimal, Decimal] | None, +) -> str | None: + """Validate one direction of a totals binding. Returns a reason, or None if OK.""" + if qty == 0 and amt == 0: + return None # legitimately inactive side + if qty <= 0 or amt <= 0: + return "qty/amount inconsistent" + avg = amt / qty + if not (_LEGACY_PRICE_MIN <= avg <= _LEGACY_PRICE_MAX): + return f"implied price implausible ({avg:.4f})" + if price_range is not None: + lo, hi = price_range + if lo > 0 and not (lo * Decimal("0.9") <= avg <= hi * Decimal("1.1")): + return f"implied price {avg:.4f} outside narrative range {lo}-{hi}" + return None + + def _parse_legacy_blotter( text: str, ticker: str, @@ -283,8 +322,15 @@ def _parse_legacy_blotter( _log.warning("dkb_row_rejected", reason="legacy: TOPLAM SATIŞ label not found") return [] - # First four numeric cells after the totals labels: buy qty, sell qty, - # buy amount, sell amount (order verified on 2015 and 2018 fixtures). + # The four numeric cells after the totals labels. Their ORDER is not stable: + # pdfminer emits this table column-major in some files and row-major in others, + # giving either (buy qty, sell qty, buy amt, sell amt) or + # (buy qty, buy amt, sell qty, sell amt). Guessing the order is how columns got + # silently mis-read before, so it is not guessed: both bindings are validated + # against the same checks (each active side must have qty>0 AND amount>0, an + # implied average price inside the plausible band, and inside the narrated + # range when one exists) and the binding that uniquely survives wins. If both + # or neither survive, the filing is rejected loudly rather than half-trusted. numerics: list[Decimal] = [] for ln in lines[sell_idx + 1:]: for tok in ln.split(): @@ -298,7 +344,7 @@ def _parse_legacy_blotter( if len(numerics) < 4: _log.warning("dkb_row_rejected", reason="legacy: totals block incomplete") return [] - buy_qty, sell_qty, buy_amt, sell_amt = numerics[:4] + n1, n2, n3, n4 = numerics[:4] # Latest trade date in the document. The blotter's own rows and the narrative # carry the same dates, so a plain scan is order-independent. @@ -313,46 +359,54 @@ def _parse_legacy_blotter( return [] tx_date = max(dates) - # Narrative price range, if present, bounds the implied average price. - range_lo = range_hi = None - pm = _PRICE_RANGE_RE.search(text) - if pm: + # Narrated price ranges. The filing template narrates the buy leg first + # ("... fiyat aralığından N adet alış işlemi ve/veya ... satış işlemi"), so the + # first range bounds the buy side and the last bounds the sell side. With a + # single range and both directions active, only the buy side is range-checked - + # a sell narrated without its own range must not be rejected against the buy's. + ranges: list[tuple[Decimal, Decimal]] = [] + for m in _PRICE_RANGE_RE.finditer(text): try: - range_lo = parse_turkish_number(pm.group(1)) - range_hi = parse_turkish_number(pm.group(2)) + ranges.append((_parse_price_token(m.group(1)), _parse_price_token(m.group(2)))) except InvalidOperation: pass - - txs: list[KapInsiderTxDTO] = [] - for tx_type, qty, amt in (("BUY", buy_qty, buy_amt), ("SELL", sell_qty, sell_amt)): - if qty == 0 and amt == 0: - continue - if qty <= 0 or amt <= 0: - _log.warning( - "dkb_row_rejected", - reason=f"legacy: {tx_type} qty/amount inconsistent", - qty=float(qty), - amount=float(amt), - ) - continue - avg_price = (amt / qty).quantize(Decimal("0.0001")) - if not (_LEGACY_PRICE_MIN <= avg_price <= _LEGACY_PRICE_MAX): + buy_range = ranges[0] if ranges else None + sell_range = ranges[-1] if len(ranges) > 1 else None + + bindings = { + "column-major": (n1, n2, n3, n4), # buy qty, sell qty, buy amt, sell amt + "row-major": (n1, n3, n2, n4), # buy qty, buy amt, sell qty, sell amt + } + survivors = {} + for name, (bq, sq, ba, sa) in bindings.items(): + if _check_side(bq, ba, buy_range) is None and _check_side(sq, sa, sell_range) is None: + survivors[name] = (bq, sq, ba, sa) + + if not survivors: + _log.warning( + "dkb_row_rejected", + reason="legacy: no totals binding satisfies the checks", + cells=[float(v) for v in (n1, n2, n3, n4)], + ) + return [] + if len(survivors) > 1: + # Both orders pass only when the bindings agree in substance (e.g. one side + # all zeros makes them identical); otherwise the filing is truly ambiguous. + vals = list(survivors.values()) + if vals[0] != vals[1]: _log.warning( "dkb_row_rejected", - reason="legacy: implied price implausible", - avg_price=float(avg_price), + reason="legacy: totals binding ambiguous", + cells=[float(v) for v in (n1, n2, n3, n4)], ) + return [] + binding_name, (buy_qty, sell_qty, buy_amt, sell_amt) = next(iter(survivors.items())) + + txs: list[KapInsiderTxDTO] = [] + for tx_type, qty, amt in (("BUY", buy_qty, buy_amt), ("SELL", sell_qty, sell_amt)): + if qty == 0: continue - if range_lo is not None and range_hi is not None and range_lo > 0: - if not (range_lo * Decimal("0.9") <= avg_price <= range_hi * Decimal("1.1")): - _log.warning( - "dkb_row_rejected", - reason="legacy: implied price outside narrative range", - avg_price=float(avg_price), - range_lo=float(range_lo), - range_hi=float(range_hi), - ) - continue + avg_price = (amt / qty).quantize(Decimal("0.0001"), rounding=ROUND_HALF_UP) txs.append( KapInsiderTxDTO( insider_name=insider_name, @@ -372,6 +426,7 @@ def _parse_legacy_blotter( tx_type=tx_type, qty=float(qty), avg_price=float(avg_price), + binding=binding_name, ) return txs @@ -409,7 +464,7 @@ def parse_dkb_transactions( pm = _PRICE_RANGE_RE.search(text) if pm: try: - price_try = parse_turkish_number(pm.group(1)) + price_try = _parse_price_token(pm.group(1)) except InvalidOperation: pass diff --git a/tests/unit/scrapers/kap/test_legacy_binding.py b/tests/unit/scrapers/kap/test_legacy_binding.py new file mode 100644 index 0000000..82e6d22 --- /dev/null +++ b/tests/unit/scrapers/kap/test_legacy_binding.py @@ -0,0 +1,125 @@ +"""Totals-binding disambiguation and price parsing in the legacy blotter. + +The first clean-backfill attempt rejected ~65% of 2015-era filings. The reasons +tallied to three concrete defects, each pinned here with a synthetic document: + + 18x "implied price outside narrative range" - the price regex read "1.234,56 TL" + as 1234 (grabbing "1.234") and dot-decimal "4.36" as 436; and a single + narrated BUY range was applied to the SELL side too. + 18x "qty/amount inconsistent" - pdfminer emits the TOPLAM cells column-major in + some files and row-major in others; assuming one order broke the other. + 10x "totals block incomplete" - still rejected loudly (needs the cells to exist). +""" +from decimal import Decimal + +from trailing_edge.scrapers.kap.parser import ( + _PRICE_RANGE_RE, + _parse_legacy_blotter, + _parse_price_token, +) + + +def _doc(totals_lines: str, narrative: str = "") -> str: + return f"""SÜREKLİ BİLGİLERE İLİŞKİN ÖZEL DURUM AÇIKLAMASI +{narrative} +İşlem Tarihi +15.06.2015 +Alım +TOPLAM ALIŞ +TOPLAM SATIŞ +{totals_lines} +""" + + +def parse(doc: str): + return _parse_legacy_blotter(doc, ticker="X", insider_name="A B", relation_type="KENDISI") + + +# --- price token parsing ----------------------------------------------------- + +def test_price_token_thousands_and_comma(): + assert _parse_price_token("1.234,56") == Decimal("1234.56") + + +def test_price_token_dot_decimal(): + """Old filings write '4.36' meaning 4 lira 36 kurus - not 436.""" + assert _parse_price_token("4.36") == Decimal("4.36") + + +def test_price_token_plain_comma(): + assert _parse_price_token("18,45") == Decimal("18.45") + + +def test_range_regex_captures_full_thousands_price(): + m = _PRICE_RANGE_RE.search("1.234,56 - 1.240,00 TL fiyat aralığından") + assert m.group(1) == "1.234,56" + assert _parse_price_token(m.group(1)) == Decimal("1234.56") + + +# --- totals binding ---------------------------------------------------------- + +def test_column_major_binding(): + """(buy qty, sell qty, buy amt, sell amt) - the 2015/2018 fixture order.""" + txs = parse(_doc("120.000\n0\n526.950\n0")) + assert len(txs) == 1 + assert txs[0].transaction_type == "BUY" + assert txs[0].share_count == Decimal("120000") + + +def test_row_major_binding_is_recognised(): + """(buy qty, buy amt, sell qty, sell amt) - the emission that used to be + rejected as 'qty/amount inconsistent'. 120.000 @ 526.950 implies 4.39 TL; + read column-major it implies buy amt 0 for qty 120.000, which fails - so + exactly one binding survives and it is the right one.""" + txs = parse(_doc("120.000\n526.950\n0\n0")) + assert len(txs) == 1 + assert txs[0].transaction_type == "BUY" + assert txs[0].share_count == Decimal("120000") + assert txs[0].price_try == Decimal("4.3913") + + +def test_truly_ambiguous_binding_is_rejected(): + """Four cells that pass BOTH bindings with different substance must not be + half-trusted. (100 sh @ 200 TL vs 100 sh @ 300 TL depending on order.)""" + txs = parse(_doc("100\n200\n300\n400")) + assert txs == [] + + +def test_single_buy_range_does_not_reject_the_sell_side(): + """One narrated range (the buy leg's). Sell at 9,50 is outside 1,11-1,16 - + that must NOT reject the sell: only the buy is checked against the buy range.""" + doc = _doc( + "479.593\n69.368\n549.526,49\n658.996", # sell avg 9.5 TL + narrative="1,11 - 1,16 TL fiyat aralığından 479.593 adet alış işlemi", + ) + txs = parse(doc) + assert {t.transaction_type for t in txs} == {"BUY", "SELL"} + + +def test_two_ranges_bind_buy_first_sell_last(): + doc = _doc( + "1.000\n2.000\n4.500\n19.000", # buy avg 4,50 ; sell avg 9,50 + narrative=( + "4,40 - 4,60 TL fiyat aralığından 1.000 adet alış işlemi ve " + "9,40 - 9,60 TL fiyat aralığından 2.000 adet satış işlemi" + ), + ) + txs = parse(doc) + assert len(txs) == 2 + by = {t.transaction_type: t for t in txs} + assert by["BUY"].price_try == Decimal("4.5000") + assert by["SELL"].price_try == Decimal("9.5000") + + +def test_buy_price_outside_its_own_range_still_rejects(): + """The range check must keep its teeth: a buy avg of 45 TL against a narrated + 4,40-4,60 range is a mis-binding and the filing must be rejected.""" + doc = _doc( + "1.000\n0\n45.000\n0", + narrative="4,40 - 4,60 TL fiyat aralığından 1.000 adet alış işlemi", + ) + assert parse(doc) == [] + + +def test_incomplete_totals_still_rejected(): + assert parse(_doc("120.000\n526.950")) == [] From fcf063071ff52d1a083b834c0e64df860fac4e44 Mon Sep 17 00:00:00 2001 From: cagan Date: Sun, 12 Jul 2026 05:08:16 +0300 Subject: [PATCH 06/29] scrapers: quarantine unparsed DKB PDFs instead of discarding the evidence A DKB filing that parses to zero transactions now keeps its PDF under reports/parse_failures/{disclosure_id}.pdf (gitignored). The bytes were already paid for during ingest; discarding them meant every parse failure had to be re-fought through KAP's WAF one probe at a time just to be looked at. With the evidence retained, remaining failure modes can be diagnosed offline in bulk and repaired with a single re-parse pass over the quarantine directory - no new requests at all. --- src/trailing_edge/scrapers/kap/insider.py | 27 +++++++++++++++-------- 1 file changed, 18 insertions(+), 9 deletions(-) diff --git a/src/trailing_edge/scrapers/kap/insider.py b/src/trailing_edge/scrapers/kap/insider.py index 52690d7..f97c3d7 100644 --- a/src/trailing_edge/scrapers/kap/insider.py +++ b/src/trailing_edge/scrapers/kap/insider.py @@ -1,6 +1,7 @@ """KAP insider scraper orchestrator.""" from dataclasses import dataclass from datetime import date +from pathlib import Path from trailing_edge.core.db import get_session from trailing_edge.core.http import RateLimitedClient @@ -19,6 +20,11 @@ SCRAPER_NAME = "kap_insider" +# DKB PDFs that parsed to zero transactions land here (gitignored under /reports), +# named by disclosure id, so parse failures keep their evidence for offline +# diagnosis and a later bulk repair pass. +_QUARANTINE_DIR = Path("reports/parse_failures") + @dataclass class ScraperRunResult: @@ -76,6 +82,7 @@ async def run(self, from_date: date, to_date: date) -> ScraperRunResult: dto = parse_disclosure_metadata(detail, list_item=disc) txs = [] + pdf_bytes: bytes | None = None # Route by the list API's disclosureClass (reliable DKB indicator) is_dkb = disc.get("disclosureClass") == "DKB" if is_dkb: @@ -95,23 +102,25 @@ async def run(self, from_date: date, to_date: date) -> ScraperRunResult: # A DKB disclosure that yields no transactions is a parse failure, # not an empty filing - the whole point of a Pay Alim Satim - # Bildirimi is that it reports at least one trade. Storing the - # disclosure row while quietly dropping its transactions is how - # the pre-2016-06 format gap went unnoticed: every DUY-era filing - # (a free-form attachment, not KAP's structured table) parsed to - # zero, and the ingest reported success. Surface it. + # Bildirimi is that it reports at least one trade. Surface it, + # and QUARANTINE the PDF: the bytes were already paid for, and + # a failure that keeps its evidence can be diagnosed offline and + # repaired in bulk later, instead of re-fought through the WAF + # one probe at a time. if is_dkb and not txs: empty_dkb += 1 + quarantined = None + if pdf_bytes: + _QUARANTINE_DIR.mkdir(parents=True, exist_ok=True) + quarantined = _QUARANTINE_DIR / f"{kap_disclosure_id}.pdf" + quarantined.write_bytes(pdf_bytes) _log.warning( "dkb_yielded_no_transactions", kap_disclosure_id=kap_disclosure_id, ticker=dto.ticker, disclosure_type=disc.get("disclosureType"), published_at=str(dto.published_at), - hint=( - "DUY-era (pre-2016-06) filings attach a free-form PDF " - "that parse_dkb_transactions cannot read" - ), + quarantined=str(quarantined) if quarantined else None, ) From 705e53a2f7f42b6185030071415de5f4894b1a63 Mon Sep 17 00:00:00 2001 From: cagan Date: Sun, 12 Jul 2026 05:33:57 +0300 Subject: [PATCH 07/29] scrapers: two more evidence-backed blotter emissions + offline quarantine repair The quarantine directory did its job on the first day. 165 unparseable PDFs accumulated during the running backfill; diagnosing them offline (zero extra requests against the WAF) surfaced two more real emissions of the totals block: 1. Separated labels - each label owns the pair that follows it: TOPLAM ALIS / qty / amt / TOPLAM SATIS / qty / amt The old code looked only after the SELL label, found the sell side's lonely zeros, and rejected the filing as "totals block incomplete" (31 times in month one alone). 2. Trade quantities first, totals pair after (quarantined filings 405071 and 404639): [186, 133, 5, 155, 479, 0, prices...] where 186+133+5+155 == 479 The totals pair is findable without knowing the layout: it is the adjacent pair that EXACTLY equals the sum of every numeric before it. Exact Decimal equality makes an accidental match effectively impossible - a position is again accepted only because arithmetic proved it. Trade amounts are interleaved beyond recovery in this emission, so price_try stays NULL rather than guessed; share_count is the field the signal pipeline actually needs. scripts/reparse_quarantine.py replays quarantined PDFs against the current parser and upserts what now succeeds - no network involved. Dry-run against the real quarantine: 81 of 165 recovered; of the rest, 36 are scanned images (OCR territory, out of scope) and 48 are a long tail of rarer emissions that stay quarantined with their evidence rather than being guessed at. Tests: 189 passing, including the separated-labels case verbatim from quarantined filing 405865. --- scripts/reparse_quarantine.py | 104 ++++++++++++++ src/trailing_edge/scrapers/kap/parser.py | 136 ++++++++++++------ .../unit/scrapers/kap/test_legacy_binding.py | 56 ++++++++ 3 files changed, 255 insertions(+), 41 deletions(-) create mode 100644 scripts/reparse_quarantine.py diff --git a/scripts/reparse_quarantine.py b/scripts/reparse_quarantine.py new file mode 100644 index 0000000..9108bef --- /dev/null +++ b/scripts/reparse_quarantine.py @@ -0,0 +1,104 @@ +"""Re-parse quarantined DKB PDFs with the current parser and store what now succeeds. + +The ingest quarantines every DKB PDF that parses to zero transactions +(reports/parse_failures/{disclosure_id}.pdf). After a parser improvement, this script +replays those files offline - no network, no WAF - and upserts the recovered +transactions against their already-stored disclosures. Files that still fail stay in +quarantine for the next diagnosis round; files that succeed are removed. + +Usage: + uv run python scripts/reparse_quarantine.py # repair + uv run python scripts/reparse_quarantine.py --dry-run # report only +""" +from __future__ import annotations + +import asyncio +import sys +from pathlib import Path + +import click + +sys.path.insert(0, "src") + +from sqlalchemy import select # noqa: E402 + +from trailing_edge.core.db import get_session, init_db # noqa: E402 +from trailing_edge.core.logging import configure_logging, get_logger # noqa: E402 +from trailing_edge.models.kap import KapDisclosure # noqa: E402 +from trailing_edge.scrapers.kap.parser import parse_dkb_transactions # noqa: E402 +from trailing_edge.storage.repository import KapRepository # noqa: E402 + +_log = get_logger(__name__) + +QUARANTINE_DIR = Path("reports/parse_failures") + + +async def main_async(dry_run: bool) -> None: + await init_db() + + pdfs = sorted(QUARANTINE_DIR.glob("*.pdf")) + if not pdfs: + click.echo("Quarantine is empty - nothing to repair.") + return + + recovered = still_failing = orphaned = tx_total = 0 + + for pdf_path in pdfs: + disclosure_id = pdf_path.stem + + async with get_session() as session: + row = ( + await session.execute( + select(KapDisclosure).where( + KapDisclosure.kap_disclosure_id == disclosure_id + ) + ) + ).scalar_one_or_none() + + if row is None: + # PDF outlived its disclosure row (e.g. the DB was reset since + # quarantine). Leave the file; a future backfill will re-store the + # disclosure and the next repair pass will pick it up. + orphaned += 1 + continue + + txs = parse_dkb_transactions(pdf_path.read_bytes(), ticker=row.ticker) + if not txs: + still_failing += 1 + continue + + inserted = 0 + if not dry_run: + repo = KapRepository(session) + result = await repo.upsert_transactions(row.id, txs) + inserted = result.inserted + + recovered += 1 + tx_total += len(txs) + _log.info( + "quarantine_repaired", + kap_disclosure_id=disclosure_id, + ticker=row.ticker, + transactions=len(txs), + inserted=inserted, + dry_run=dry_run, + ) + if not dry_run: + pdf_path.unlink() + + click.echo( + f"Quarantine: {len(pdfs)} files | repaired {recovered} " + f"({tx_total} transactions) | still failing {still_failing} | orphaned {orphaned}" + + (" [DRY RUN - nothing written or deleted]" if dry_run else "") + ) + + +@click.command() +@click.option("--dry-run", is_flag=True, help="Parse and report without writing.") +def main(dry_run: bool) -> None: + configure_logging() + asyncio.run(main_async(dry_run)) + + +if __name__ == "__main__": + main() diff --git a/src/trailing_edge/scrapers/kap/parser.py b/src/trailing_edge/scrapers/kap/parser.py index ed8e94d..3f5072c 100644 --- a/src/trailing_edge/scrapers/kap/parser.py +++ b/src/trailing_edge/scrapers/kap/parser.py @@ -314,37 +314,17 @@ def _parse_legacy_blotter( ) -> list[KapInsiderTxDTO]: lines = [ln.strip().replace("\xa0", "") for ln in text.splitlines() if ln.strip()] + buy_idx = next( + (i for i, ln in enumerate(lines) if ln.upper().startswith(_LEGACY_BUY_LABEL)), + None, + ) sell_idx = next( (i for i, ln in enumerate(lines) if ln.upper().startswith(_LEGACY_SELL_LABEL)), None, ) - if sell_idx is None: - _log.warning("dkb_row_rejected", reason="legacy: TOPLAM SATIŞ label not found") - return [] - - # The four numeric cells after the totals labels. Their ORDER is not stable: - # pdfminer emits this table column-major in some files and row-major in others, - # giving either (buy qty, sell qty, buy amt, sell amt) or - # (buy qty, buy amt, sell qty, sell amt). Guessing the order is how columns got - # silently mis-read before, so it is not guessed: both bindings are validated - # against the same checks (each active side must have qty>0 AND amount>0, an - # implied average price inside the plausible band, and inside the narrated - # range when one exists) and the binding that uniquely survives wins. If both - # or neither survive, the filing is rejected loudly rather than half-trusted. - numerics: list[Decimal] = [] - for ln in lines[sell_idx + 1:]: - for tok in ln.split(): - if _TR_NUM_RE.fullmatch(tok.lstrip("-")): - try: - numerics.append(parse_turkish_number(tok)) - except InvalidOperation: - pass - if len(numerics) >= 4: - break - if len(numerics) < 4: - _log.warning("dkb_row_rejected", reason="legacy: totals block incomplete") + if buy_idx is None or sell_idx is None: + _log.warning("dkb_row_rejected", reason="legacy: TOPLAM labels not found") return [] - n1, n2, n3, n4 = numerics[:4] # Latest trade date in the document. The blotter's own rows and the narrative # carry the same dates, so a plain scan is order-independent. @@ -373,22 +353,58 @@ def _parse_legacy_blotter( buy_range = ranges[0] if ranges else None sell_range = ranges[-1] if len(ranges) > 1 else None - bindings = { - "column-major": (n1, n2, n3, n4), # buy qty, sell qty, buy amt, sell amt - "row-major": (n1, n3, n2, n4), # buy qty, buy amt, sell qty, sell amt - } + def _numerics_in(segment: list[str], limit: int) -> list[Decimal]: + out: list[Decimal] = [] + for ln in segment: + for tok in ln.split(): + if _TR_NUM_RE.fullmatch(tok.lstrip("-")): + try: + out.append(parse_turkish_number(tok)) + except InvalidOperation: + pass + if len(out) >= limit: + break + return out[:limit] + + # Two emissions of the totals block exist, distinguished by whether the labels + # are adjacent. When they are SEPARATED, each label owns the pair that follows + # it - "TOPLAM ALIŞ / qty / amt / TOPLAM SATIŞ / qty / amt" - in the table's + # own header order (nominal before tutar). The old code assumed adjacency, + # looked only after the SELL label, found the sell side's lonely zeros, and + # rejected the filing as "totals block incomplete" (31 times in month one). + if sell_idx - buy_idx > 1 and buy_idx < sell_idx: + buy_cells = _numerics_in(lines[buy_idx + 1 : sell_idx], 2) + sell_cells = _numerics_in(lines[sell_idx + 1 :], 2) + if len(buy_cells) < 2 or len(sell_cells) < 2: + _log.warning("dkb_row_rejected", reason="legacy: totals block incomplete") + return [] + bindings = { + "label-adjacent": (buy_cells[0], sell_cells[0], buy_cells[1], sell_cells[1]), + } + else: + # Adjacent labels: four cells follow, but their ORDER is not stable - + # pdfminer emits (buy qty, sell qty, buy amt, sell amt) in some files and + # (buy qty, buy amt, sell qty, sell amt) in others. Guessing the order is + # how columns got silently mis-read before, so it is not guessed: both + # bindings run the same checks and the one that uniquely survives wins. + cells = _numerics_in(lines[sell_idx + 1 :], 4) + if len(cells) < 4: + # Third observed emission: the PER-TRADE quantities come first and the + # totals pair only after them. Handled below on the longer window. + cells = [] + if len(cells) == 4: + n1, n2, n3, n4 = cells + bindings = { + "column-major": (n1, n2, n3, n4), # buy qty, sell qty, buy amt, sell amt + "row-major": (n1, n3, n2, n4), # buy qty, buy amt, sell qty, sell amt + } + else: + bindings = {} survivors = {} for name, (bq, sq, ba, sa) in bindings.items(): if _check_side(bq, ba, buy_range) is None and _check_side(sq, sa, sell_range) is None: survivors[name] = (bq, sq, ba, sa) - if not survivors: - _log.warning( - "dkb_row_rejected", - reason="legacy: no totals binding satisfies the checks", - cells=[float(v) for v in (n1, n2, n3, n4)], - ) - return [] if len(survivors) > 1: # Both orders pass only when the bindings agree in substance (e.g. one side # all zeros makes them identical); otherwise the filing is truly ambiguous. @@ -397,16 +413,54 @@ def _parse_legacy_blotter( _log.warning( "dkb_row_rejected", reason="legacy: totals binding ambiguous", - cells=[float(v) for v in (n1, n2, n3, n4)], + cells=[float(v) for v in next(iter(bindings.values()))], + ) + return [] + + if survivors: + binding_name, (buy_qty, sell_qty, buy_amt, sell_amt) = next(iter(survivors.items())) + else: + # Fourth observed emission (quarantined filings 405071, 404639): the + # PER-TRADE quantities come first and only then the totals pair - + # [186, 133, 5, 155, 479, 0, prices...] where 186+133+5+155 == 479. The + # totals pair is findable without knowing the layout: it is the adjacent + # pair that EXACTLY equals the sum of every numeric before it. Exact + # Decimal equality makes an accidental match effectively impossible; a + # position is again accepted only because arithmetic proved it. The trade + # AMOUNTS are interleaved beyond recovery in this emission, so the implied + # price cannot be computed - price_try stays NULL (the honest value) and + # only quantities are stored. Mixed buy+sell filings are skipped here: + # without amounts the two directions cannot be range-checked apart. + seq = _numerics_in(lines[sell_idx + 1 :], 16) + matches = [] + for k in range(2, len(seq) - 1): + a, b = seq[k], seq[k + 1] + if a < 0 or b < 0 or a + b <= 0: + continue + if (a == 0) != (b == 0): # exactly one active side + if sum(seq[:k]) == a + b: + matches.append((a, b)) + unique = {m for m in matches} + if len(unique) != 1: + _log.warning( + "dkb_row_rejected", + reason="legacy: no totals binding satisfies the checks", + cells=[float(v) for v in seq[:6]], ) return [] - binding_name, (buy_qty, sell_qty, buy_amt, sell_amt) = next(iter(survivors.items())) + buy_qty, sell_qty = unique.pop() + buy_amt = sell_amt = None + binding_name = "qty-sum" txs: list[KapInsiderTxDTO] = [] for tx_type, qty, amt in (("BUY", buy_qty, buy_amt), ("SELL", sell_qty, sell_amt)): if qty == 0: continue - avg_price = (amt / qty).quantize(Decimal("0.0001"), rounding=ROUND_HALF_UP) + avg_price = ( + (amt / qty).quantize(Decimal("0.0001"), rounding=ROUND_HALF_UP) + if amt is not None + else None + ) txs.append( KapInsiderTxDTO( insider_name=insider_name, @@ -425,7 +479,7 @@ def _parse_legacy_blotter( ticker=ticker, tx_type=tx_type, qty=float(qty), - avg_price=float(avg_price), + avg_price=float(avg_price) if avg_price is not None else None, binding=binding_name, ) return txs diff --git a/tests/unit/scrapers/kap/test_legacy_binding.py b/tests/unit/scrapers/kap/test_legacy_binding.py index 82e6d22..6e09811 100644 --- a/tests/unit/scrapers/kap/test_legacy_binding.py +++ b/tests/unit/scrapers/kap/test_legacy_binding.py @@ -123,3 +123,59 @@ def test_buy_price_outside_its_own_range_still_rejects(): def test_incomplete_totals_still_rejected(): assert parse(_doc("120.000\n526.950")) == [] + + +# --- separated labels (each label owns its pair) ------------------------------ + +def test_separated_labels_bind_per_label(): + """The emission behind 31 'totals block incomplete' rejections in month one: + TOPLAM ALIŞ / qty / amt / TOPLAM SATIŞ / qty / amt. Looking only after the + SELL label finds the sell side's lonely zeros and nothing else. + + Cells verbatim from quarantined filing 405865: 395.321 shares bought for + 256.442,84 TL (avg 0,6487), nothing sold.""" + doc = """SÜREKLİ BİLGİLERE İLİŞKİN ÖZEL DURUM AÇIKLAMASI +İşlem Tarihi +05.01.2015 +Alım +TOPLAM ALIŞ +395.321 +256.442,84 +TOPLAM SATIŞ +0 +0 +""" + txs = parse(doc) + assert len(txs) == 1 + assert txs[0].transaction_type == "BUY" + assert txs[0].share_count == Decimal("395321") + assert txs[0].price_try == Decimal("0.6487") + + +def test_separated_labels_with_both_sides_active(): + doc = """SÜREKLİ BİLGİLERE İLİŞKİN ÖZEL DURUM AÇIKLAMASI +İşlem Tarihi +05.01.2015 +TOPLAM ALIŞ +1.000 +4.500 +TOPLAM SATIŞ +2.000 +19.000 +""" + txs = parse(doc) + by = {t.transaction_type: t for t in txs} + assert by["BUY"].share_count == Decimal("1000") + assert by["SELL"].share_count == Decimal("2000") + assert by["SELL"].price_try == Decimal("9.5000") + + +def test_separated_labels_missing_sell_pair_rejected(): + doc = """başlık +05.01.2015 +TOPLAM ALIŞ +1.000 +4.500 +TOPLAM SATIŞ +""" + assert parse(doc) == [] From 1cc941b3cf7606a6c13ad73f41755abd5dff105c Mon Sep 17 00:00:00 2001 From: cagan Date: Sun, 12 Jul 2026 09:23:51 +0300 Subject: [PATCH 08/29] scrapers: accept a legacy filing only when two independent sources agree to the kurus The dominant remaining quarantine bucket (89% of a fresh hour's failures) was a fifth emission: the totals labels sit BEFORE the table cells entirely, so every positional path - label-adjacent pairs, dual binding, qty-sum - had nothing to bind. The cells carry textbook trade triples (250 x 1,35 = 337,50; 9.850 x 3,3 = 32.505), but their positions are unusable. What was being overlooked: the filing narrates its own volume in words. "479.593 adet alis islemi" is data, present on every legacy filing, and it is INDEPENDENT of the table's emission order. The fallback extracts quantities per direction from the narrative and accepts them only when a second, disjoint source agrees exactly: - the table's exact qty*price==amount triples summing to the same quantity (price then comes out as amount_sum/qty_sum and must still clear the narrated range), or - the narrated totals appearing verbatim as table cells. Prose lines are excluded from the table-numeric pool, so the corroboration is genuinely cross-source - the narrative cannot confirm itself. One contract sharpened by a test that had to change: when the quantity is doubly corroborated but the amount contradicts the narrated price range, the old policy rejected the whole filing. Now the quantity is kept and the price is refused (NULL) - keep what two sources prove, never store what one source contradicts. And the fallback is not "accept anything with a TOPLAM label": an uncorroborated filing still dies, pinned by test. Dry-run over the real quarantine: 903 of 1,227 files recovered (74%, from 49%), 1,044 transactions. The remainder is scanned images (OCR, out of scope) and a long tail that stays quarantined with its evidence. Tests: 190 passing. --- src/trailing_edge/scrapers/kap/parser.py | 201 +++++++++++++----- .../unit/scrapers/kap/test_legacy_binding.py | 28 ++- 2 files changed, 170 insertions(+), 59 deletions(-) diff --git a/src/trailing_edge/scrapers/kap/parser.py b/src/trailing_edge/scrapers/kap/parser.py index 3f5072c..a39c282 100644 --- a/src/trailing_edge/scrapers/kap/parser.py +++ b/src/trailing_edge/scrapers/kap/parser.py @@ -306,6 +306,73 @@ def _check_side( return None +# "...fiyat aralığından 479.593 adet alış işlemi ve/veya 0 adet satış işlemi..." +# The narrative states the traded quantity in words on every legacy filing. Per-trade +# narrations list several "N adet" before one direction keyword; the next direction +# keyword after each match attributes it. +_NARRATIVE_QTY_RE = re.compile(r"(\d[\d.,]*)\s*adet", re.IGNORECASE) +_NARRATIVE_DIR_RE = re.compile(r"al[ıi][şsm]|sat[ıi][şsm]", re.IGNORECASE) + + +def _narrative_quantities(text: str) -> tuple[Decimal, Decimal] | None: + """(buy_qty, sell_qty) as narrated in the filing's prose, or None.""" + buy = sell = Decimal(0) + found = False + for m in _NARRATIVE_QTY_RE.finditer(text): + try: + qty = parse_turkish_number(m.group(1).rstrip(".,")) + except InvalidOperation: + continue + d = _NARRATIVE_DIR_RE.search(text, m.end()) + if d is None: + continue + found = True + if d.group(0).lower().startswith("al"): + buy += qty + else: + sell += qty + return (buy, sell) if found else None + + +def _exact_triples(numerics: list[Decimal]) -> tuple[Decimal, Decimal]: + """Greedy disjoint cover by exact qty*price==amount triples. + + Returns (sum of quantities, sum of amounts) over the cover. Exact Decimal + equality is the filter; the qty/price roles are assigned large/small, which + holds for every real trade in this corpus (thousands of shares at tens of + lira). Used only as corroboration - never as the sole source. + """ + from collections import Counter + + pool = Counter(numerics) + sum_q = sum_a = Decimal(0) + # Largest amounts first, so real trades claim their cells before small + # accidental products (2*3==6) can steal them. + candidates = sorted( + ( + (q * p, q, p) + for q in pool + for p in pool + if q > p > 0 + and _LEGACY_PRICE_MIN <= p <= _LEGACY_PRICE_MAX + and q >= 1 + ), + key=lambda t: -t[0], + ) + for a, q, p in candidates: + if pool[a] <= 0 or pool[q] <= 0 or pool[p] <= 0: + continue + if q == a or p == a: # degenerate (price 1.0 etc.) - not evidence + continue + need = Counter((a, q, p)) + if any(pool[v] < c for v, c in need.items()): + continue + pool.subtract(need) + sum_q += q + sum_a += a + return sum_q, sum_a + + def _parse_legacy_blotter( text: str, ticker: str, @@ -366,21 +433,24 @@ def _numerics_in(segment: list[str], limit: int) -> list[Decimal]: break return out[:limit] + resolved: tuple | None = None # (buy_qty, sell_qty, buy_amt|None, sell_amt|None, how) + reject_reason = "legacy: no totals binding satisfies the checks" + # Two emissions of the totals block exist, distinguished by whether the labels # are adjacent. When they are SEPARATED, each label owns the pair that follows # it - "TOPLAM ALIŞ / qty / amt / TOPLAM SATIŞ / qty / amt" - in the table's - # own header order (nominal before tutar). The old code assumed adjacency, - # looked only after the SELL label, found the sell side's lonely zeros, and - # rejected the filing as "totals block incomplete" (31 times in month one). + # own header order (nominal before tutar). Looking only after the SELL label + # finds the sell side's lonely zeros ("totals block incomplete", 31x in month 1). + bindings: dict[str, tuple] = {} if sell_idx - buy_idx > 1 and buy_idx < sell_idx: buy_cells = _numerics_in(lines[buy_idx + 1 : sell_idx], 2) sell_cells = _numerics_in(lines[sell_idx + 1 :], 2) - if len(buy_cells) < 2 or len(sell_cells) < 2: - _log.warning("dkb_row_rejected", reason="legacy: totals block incomplete") - return [] - bindings = { - "label-adjacent": (buy_cells[0], sell_cells[0], buy_cells[1], sell_cells[1]), - } + if len(buy_cells) == 2 and len(sell_cells) == 2: + bindings = { + "label-adjacent": (buy_cells[0], sell_cells[0], buy_cells[1], sell_cells[1]), + } + else: + reject_reason = "legacy: totals block incomplete" else: # Adjacent labels: four cells follow, but their ORDER is not stable - # pdfminer emits (buy qty, sell qty, buy amt, sell amt) in some files and @@ -388,10 +458,6 @@ def _numerics_in(segment: list[str], limit: int) -> list[Decimal]: # how columns got silently mis-read before, so it is not guessed: both # bindings run the same checks and the one that uniquely survives wins. cells = _numerics_in(lines[sell_idx + 1 :], 4) - if len(cells) < 4: - # Third observed emission: the PER-TRADE quantities come first and the - # totals pair only after them. Handled below on the longer window. - cells = [] if len(cells) == 4: n1, n2, n3, n4 = cells bindings = { @@ -399,58 +465,81 @@ def _numerics_in(segment: list[str], limit: int) -> list[Decimal]: "row-major": (n1, n3, n2, n4), # buy qty, buy amt, sell qty, sell amt } else: - bindings = {} + reject_reason = "legacy: totals block incomplete" + survivors = {} for name, (bq, sq, ba, sa) in bindings.items(): if _check_side(bq, ba, buy_range) is None and _check_side(sq, sa, sell_range) is None: survivors[name] = (bq, sq, ba, sa) - - if len(survivors) > 1: - # Both orders pass only when the bindings agree in substance (e.g. one side - # all zeros makes them identical); otherwise the filing is truly ambiguous. - vals = list(survivors.values()) - if vals[0] != vals[1]: - _log.warning( - "dkb_row_rejected", - reason="legacy: totals binding ambiguous", - cells=[float(v) for v in next(iter(bindings.values()))], - ) - return [] - + if len(survivors) > 1 and len(set(survivors.values())) > 1: + # Bindings that both pass but disagree in substance: truly ambiguous. + survivors = {} + reject_reason = "legacy: totals binding ambiguous" if survivors: - binding_name, (buy_qty, sell_qty, buy_amt, sell_amt) = next(iter(survivors.items())) - else: - # Fourth observed emission (quarantined filings 405071, 404639): the - # PER-TRADE quantities come first and only then the totals pair - - # [186, 133, 5, 155, 479, 0, prices...] where 186+133+5+155 == 479. The - # totals pair is findable without knowing the layout: it is the adjacent - # pair that EXACTLY equals the sum of every numeric before it. Exact - # Decimal equality makes an accidental match effectively impossible; a - # position is again accepted only because arithmetic proved it. The trade - # AMOUNTS are interleaved beyond recovery in this emission, so the implied - # price cannot be computed - price_try stays NULL (the honest value) and - # only quantities are stored. Mixed buy+sell filings are skipped here: - # without amounts the two directions cannot be range-checked apart. + name, (bq, sq, ba, sa) = next(iter(survivors.items())) + resolved = (bq, sq, ba, sa, name) + + if resolved is None: + # Emission #4 (quarantined 405071, 404639): per-trade quantities first, + # totals pair after - [186, 133, 5, 155, 479, 0, ...] where the pair is + # findable as the adjacent pair EXACTLY equal to the sum of everything + # before it. Trade amounts are interleaved beyond recovery here, so + # price_try stays NULL; share_count is what the pipeline needs. seq = _numerics_in(lines[sell_idx + 1 :], 16) - matches = [] + matches = set() for k in range(2, len(seq) - 1): a, b = seq[k], seq[k + 1] - if a < 0 or b < 0 or a + b <= 0: - continue - if (a == 0) != (b == 0): # exactly one active side + if a >= 0 and b >= 0 and a + b > 0 and (a == 0) != (b == 0): if sum(seq[:k]) == a + b: - matches.append((a, b)) - unique = {m for m in matches} - if len(unique) != 1: - _log.warning( - "dkb_row_rejected", - reason="legacy: no totals binding satisfies the checks", - cells=[float(v) for v in seq[:6]], - ) - return [] - buy_qty, sell_qty = unique.pop() - buy_amt = sell_amt = None - binding_name = "qty-sum" + matches.add((a, b)) + if len(matches) == 1: + bq, sq = matches.pop() + resolved = (bq, sq, None, None, "qty-sum") + + if resolved is None: + # Emission #5 (dominant in the 2015-Q2+ quarantine): the totals labels sit + # BEFORE the table cells entirely, so none of the positional paths apply. + # Two order-independent sources remain, and a filing is accepted only when + # they agree TO THE KURUŞ: + # - the narrative states the quantities in words ("479.593 adet alış"); + # - the table numerics contain either exact qty*price==amount trade + # triples summing to the same quantity, or the narrated totals as cells. + # Prose lines are excluded from the table numerics, so the corroboration + # is genuinely cross-source, not the narrative confirming itself. + narr = _narrative_quantities(text) + if narr is not None and narr[0] + narr[1] > 0: + nb, ns = narr + table_nums: list[Decimal] = [] + for ln in lines: + parts = ln.split() + if parts and all(_TR_NUM_RE.fullmatch(p.lstrip("-")) for p in parts): + for p_ in parts: + try: + table_nums.append(parse_turkish_number(p_)) + except InvalidOperation: + pass + + tq, ta = _exact_triples(table_nums) + single_dir = (nb == 0) != (ns == 0) + if tq > 0 and tq == nb + ns and single_dir: + # Triples reconstruct the exact narrated volume; their amount sum + # gives the average price, which must still clear the range check. + rng = buy_range if nb > 0 else (sell_range or buy_range) + if _check_side(nb + ns, ta, rng) is None: + resolved = (nb, ns, ta if nb > 0 else None, ta if ns > 0 else None, + "narrative+triples") + if resolved is None: + cells_present = (nb == 0 or nb in table_nums) and ( + ns == 0 or ns in table_nums + ) + if cells_present: + resolved = (nb, ns, None, None, "narrative+totals-cells") + + if resolved is None: + _log.warning("dkb_row_rejected", reason=reject_reason) + return [] + + buy_qty, sell_qty, buy_amt, sell_amt, binding_name = resolved txs: list[KapInsiderTxDTO] = [] for tx_type, qty, amt in (("BUY", buy_qty, buy_amt), ("SELL", sell_qty, sell_amt)): diff --git a/tests/unit/scrapers/kap/test_legacy_binding.py b/tests/unit/scrapers/kap/test_legacy_binding.py index 6e09811..a63f256 100644 --- a/tests/unit/scrapers/kap/test_legacy_binding.py +++ b/tests/unit/scrapers/kap/test_legacy_binding.py @@ -111,13 +111,35 @@ def test_two_ranges_bind_buy_first_sell_last(): assert by["SELL"].price_try == Decimal("9.5000") -def test_buy_price_outside_its_own_range_still_rejects(): - """The range check must keep its teeth: a buy avg of 45 TL against a narrated - 4,40-4,60 range is a mis-binding and the filing must be rejected.""" +def test_contradicted_price_is_refused_but_corroborated_qty_survives(): + """A buy amount implying 45 TL against a narrated 4,40-4,60 range: the AMOUNT + binding is provably wrong, so no price may be stored. But the QUANTITY is + corroborated by two independent sources (the narrative's '1.000 adet' and the + 1.000 totals cell), so refusing the whole filing would discard knowledge we + actually have. Contract: keep what two sources prove, NULL what is contradicted. + """ doc = _doc( "1.000\n0\n45.000\n0", narrative="4,40 - 4,60 TL fiyat aralığından 1.000 adet alış işlemi", ) + txs = parse(doc) + assert len(txs) == 1 + assert txs[0].transaction_type == "BUY" + assert txs[0].share_count == Decimal("1000") + assert txs[0].price_try is None # the contradicted amount must NOT become a price + + +def test_uncorroborated_filing_is_still_rejected(): + """No narrative quantities, no valid binding, no triples: nothing survives. + The fallback must not turn into 'accept anything with a TOPLAM label'.""" + doc = """SÜREKLİ BİLGİLERE İLİŞKİN ÖZEL DURUM AÇIKLAMASI +İşlem Tarihi +15.06.2015 +TOPLAM ALIŞ +TOPLAM SATIŞ +7 +11 +""" assert parse(doc) == [] From 625bea0c1d52bedcd0335c4375bba2705193a5c5 Mon Sep 17 00:00:00 2001 From: cagan Date: Sun, 12 Jul 2026 20:48:16 +0300 Subject: [PATCH 09/29] scrapers: stop retrying the WAF inline; recover the disclosures it drops A backfill was spending 83% of its wall clock waiting, and losing 12% of the data, for the same reason. Measured over one run: 4,200 disclosures processed, 575 lost to httpx.RemoteProtocolError ("Server disconnected without sending a response"). Each one cost ~33s and 555 of them were followed directly by a stall - 304 minutes, 85% of all stall time. The run reported SUCCESS anyway. Two defects, one cause: the code treated a WAF block as a network blip. 1. core/http._is_retryable listed RemoteProtocolError as retryable, so tenacity re-attempted it 5 times with exponential backoff (4+8+16+32s). But KAP's WAF does not lift in four seconds - retrying inline just re-pokes the block, burns up to a minute, and still fails. RemoteProtocolError is now NOT retried inline: it fails fast so the caller can handle it properly. ConnectError/ReadTimeout/429/503 - the failures that ARE transient - still retry as before. 2. KapInsiderScraper caught the exception, logged disclosure_error, and `continue`d - the disclosure was simply gone. This is the same silent-loss pattern as the list cap, the date regex and the column shift: data disappears, the run stays green. Dropped disclosures are now DEFERRED, retried once after a 90s cooldown (long enough for a volume-triggered throttle to lapse), and anything still missing downgrades the run to PARTIAL. PARTIAL matters: the resume ledger only counts SUCCESS as done, so an incomplete month is automatically re-run by the next pass, and because ingest is idempotent (disclosure_exists) that re-run costs only the disclosures actually missed. A month can no longer be silently half-ingested. The per-disclosure body is extracted to _ingest_one so the main pass and the retry pass share one implementation rather than duplicating it. Expected effect: the ~300 minutes of inline WAF backoff collapse to one cooldown per affected chunk, and the 12% loss becomes zero (or a PARTIAL flag that fixes itself). Tests: 199 passing, including the retry-policy contract - a WAF disconnect must not be retried inline, a genuine blip must be. --- src/trailing_edge/core/http.py | 16 +- src/trailing_edge/scrapers/kap/insider.py | 259 +++++++++++++--------- tests/unit/core/test_http_retry.py | 43 ++++ 3 files changed, 212 insertions(+), 106 deletions(-) create mode 100644 tests/unit/core/test_http_retry.py diff --git a/src/trailing_edge/core/http.py b/src/trailing_edge/core/http.py index fa480b0..be6fc5d 100644 --- a/src/trailing_edge/core/http.py +++ b/src/trailing_edge/core/http.py @@ -19,9 +19,23 @@ def _is_retryable(exc: BaseException) -> bool: + """Which failures are worth retrying *inline*, right where they happened. + + httpx.RemoteProtocolError ("Server disconnected without sending a response") is + deliberately NOT in this set. On KAP that error is not a network blip - it is the + WAF cutting the connection, and it does not lift in seconds. Retrying it inline + with exponential backoff (the old policy: 5 attempts, 4+8+16+32s) spent up to a + minute per occurrence re-poking the block, still failed ~12% of the time, and the + caller then dropped the disclosure. Measured over one backfill: 575 such failures, + 304 minutes burned - 83% of the wall clock - and 12% of the data silently lost. + + A WAF disconnect is instead surfaced immediately so the scraper can defer that + disclosure, wait out the throttle once per chunk, and retry it then (see + KapInsiderScraper: _WAF_COOLDOWN_S). Fail fast here, recover properly there. + """ if isinstance(exc, httpx.HTTPStatusError): return exc.response.status_code in (429, 503) - return isinstance(exc, (httpx.ConnectError, httpx.ReadTimeout, httpx.RemoteProtocolError)) + return isinstance(exc, (httpx.ConnectError, httpx.ReadTimeout)) class RateLimitedClient: diff --git a/src/trailing_edge/scrapers/kap/insider.py b/src/trailing_edge/scrapers/kap/insider.py index f97c3d7..f15baa5 100644 --- a/src/trailing_edge/scrapers/kap/insider.py +++ b/src/trailing_edge/scrapers/kap/insider.py @@ -1,4 +1,5 @@ """KAP insider scraper orchestrator.""" +import asyncio from dataclasses import dataclass from datetime import date from pathlib import Path @@ -25,6 +26,10 @@ # diagnosis and a later bulk repair pass. _QUARANTINE_DIR = Path("reports/parse_failures") +# Pause before re-attempting the disclosures KAP's WAF disconnected on. Long enough +# for a volume-triggered throttle to lapse, short enough to stay inside the chunk. +_WAF_COOLDOWN_S = 90 + @dataclass class ScraperRunResult: @@ -35,10 +40,98 @@ class ScraperRunResult: status: str +@dataclass +class _Counts: + inserted: int = 0 + updated: int = 0 + skipped: int = 0 + empty_dkb: int = 0 + + class KapInsiderScraper(AbstractScraper[ScraperRunResult]): def __init__(self, *, backfill: bool = False) -> None: self._backfill = backfill + async def _ingest_one(self, kap: KapClient, disc: dict, counts: _Counts) -> None: + """Fetch, parse and store one disclosure. + + Transport failures propagate: the caller decides whether to retry or defer. + Swallowing them here is what silently lost 12% of a backfill. + """ + kap_disclosure_id = str(disc.get("disclosureIndex", "")) + is_correction = bool(disc.get("isChanged") or disc.get("isCorrection") or False) + + async with get_session() as session: + repo = KapRepository(session) + already_exists = await repo.disclosure_exists(kap_disclosure_id) + + if already_exists and not is_correction: + counts.skipped += 1 + _log.debug("disclosure_skipped", kap_disclosure_id=kap_disclosure_id) + return + + detail = await kap.fetch_disclosure_detail(kap_disclosure_id) + # Pass list_item so metadata uses DKB class (not detail's DUY) + dto = parse_disclosure_metadata(detail, list_item=disc) + + txs = [] + pdf_bytes: bytes | None = None + # Route by the list API's disclosureClass (reliable DKB indicator) + is_dkb = disc.get("disclosureClass") == "DKB" + if is_dkb: + attachments = detail.get("attachments", []) + if attachments: + obj_id = attachments[0].get("objId", "") + if obj_id: + pdf_bytes = await kap.fetch_pdf(obj_id) + txs = parse_dkb_transactions( + pdf_bytes, ticker=dto.ticker, insider_name="" + ) + else: + body = detail.get("disclosureBody", "") or "" + txs = parse_oda_transactions(body, ticker=dto.ticker) + + # A DKB disclosure that yields no transactions is a parse failure, not an + # empty filing - the whole point of a Pay Alim Satim Bildirimi is that it + # reports at least one trade. Surface it, and QUARANTINE the PDF: the bytes + # were already paid for, and a failure that keeps its evidence can be + # diagnosed offline and repaired in bulk, instead of re-fought through the + # WAF one probe at a time. + if is_dkb and not txs: + counts.empty_dkb += 1 + quarantined = None + if pdf_bytes: + _QUARANTINE_DIR.mkdir(parents=True, exist_ok=True) + quarantined = _QUARANTINE_DIR / f"{kap_disclosure_id}.pdf" + quarantined.write_bytes(pdf_bytes) + _log.warning( + "dkb_yielded_no_transactions", + kap_disclosure_id=kap_disclosure_id, + ticker=dto.ticker, + disclosure_type=disc.get("disclosureType"), + published_at=str(dto.published_at), + quarantined=str(quarantined) if quarantined else None, + ) + + async with get_session() as session: + repo = KapRepository(session) + model, created = await repo.upsert_disclosure(dto) + result = await repo.upsert_transactions(model.id, txs) + + if created: + counts.inserted += 1 + else: + counts.updated += 1 + counts.inserted += result.inserted + + _log.info( + "disclosure_processed", + kap_disclosure_id=kap_disclosure_id, + ticker=dto.ticker, + created=created, + tx_inserted=result.inserted, + ) + async def run(self, from_date: date, to_date: date) -> ScraperRunResult: # Create the audit run record run_meta = ( @@ -50,8 +143,9 @@ async def run(self, from_date: date, to_date: date) -> ScraperRunResult: run = await repo.create_scraper_run(SCRAPER_NAME, metadata=run_meta) run_id: int = run.id - seen = inserted = updated = skipped = 0 - empty_dkb = 0 # DKB filings that parsed to zero transactions - see below + counts = _Counts() + seen = 0 + lost = 0 error_msg: str | None = None try: @@ -61,96 +155,41 @@ async def run(self, from_date: date, to_date: date) -> ScraperRunResult: disclosures = await kap.fetch_disclosure_list(from_date, to_date) seen = len(disclosures) + # KAP's WAF intermittently disconnects mid-request. Those failures used + # to be logged and skipped, dropping the disclosure outright - 12% of + # them on a real backfill, silently, while the run still said SUCCESS. + # A disconnect is transient, so a dropped disclosure is retried once + # after a cooldown; anything still missing downgrades the run to PARTIAL. + deferred: list[dict] = [] for disc in disclosures: - disclosure_index = str(disc.get("disclosureIndex", "")) - # Use disclosureIndex as the stable ID (list API has no disclosureId) - kap_disclosure_id = disclosure_index - is_correction = bool(disc.get("isChanged") or disc.get("isCorrection") or False) - - async with get_session() as session: - repo = KapRepository(session) - already_exists = await repo.disclosure_exists(kap_disclosure_id) - - if already_exists and not is_correction: - skipped += 1 - _log.debug("disclosure_skipped", kap_disclosure_id=kap_disclosure_id) - continue - try: - detail = await kap.fetch_disclosure_detail(disclosure_index) - # Pass list_item so metadata uses DKB class (not detail's DUY) - dto = parse_disclosure_metadata(detail, list_item=disc) - - txs = [] - pdf_bytes: bytes | None = None - # Route by the list API's disclosureClass (reliable DKB indicator) - is_dkb = disc.get("disclosureClass") == "DKB" - if is_dkb: - attachments = detail.get("attachments", []) - if attachments: - obj_id = attachments[0].get("objId", "") - if obj_id: - pdf_bytes = await kap.fetch_pdf(obj_id) - txs = parse_dkb_transactions( - pdf_bytes, - ticker=dto.ticker, - insider_name="", - ) - else: - body = detail.get("disclosureBody", "") or "" - txs = parse_oda_transactions(body, ticker=dto.ticker) - - # A DKB disclosure that yields no transactions is a parse failure, - # not an empty filing - the whole point of a Pay Alim Satim - # Bildirimi is that it reports at least one trade. Surface it, - # and QUARANTINE the PDF: the bytes were already paid for, and - # a failure that keeps its evidence can be diagnosed offline and - # repaired in bulk later, instead of re-fought through the WAF - # one probe at a time. - if is_dkb and not txs: - empty_dkb += 1 - quarantined = None - if pdf_bytes: - _QUARANTINE_DIR.mkdir(parents=True, exist_ok=True) - quarantined = _QUARANTINE_DIR / f"{kap_disclosure_id}.pdf" - quarantined.write_bytes(pdf_bytes) - _log.warning( - "dkb_yielded_no_transactions", - kap_disclosure_id=kap_disclosure_id, - ticker=dto.ticker, - disclosure_type=disc.get("disclosureType"), - published_at=str(dto.published_at), - quarantined=str(quarantined) if quarantined else None, - ) - - - async with get_session() as session: - repo = KapRepository(session) - model, created = await repo.upsert_disclosure(dto) - result = await repo.upsert_transactions(model.id, txs) - - if created: - inserted += 1 - else: - updated += 1 - inserted += result.inserted - - _log.info( - "disclosure_processed", - kap_disclosure_id=kap_disclosure_id, - ticker=dto.ticker, - created=created, - tx_inserted=result.inserted, - ) - + await self._ingest_one(kap, disc, counts) except Exception as exc: - _log.error( - "disclosure_error", - kap_disclosure_id=kap_disclosure_id, + deferred.append(disc) + _log.warning( + "disclosure_deferred", + kap_disclosure_id=str(disc.get("disclosureIndex", "")), error=str(exc), - exc_info=True, ) + if deferred: + _log.info( + "waf_cooldown", deferred=len(deferred), sleep_s=_WAF_COOLDOWN_S + ) + await asyncio.sleep(_WAF_COOLDOWN_S) + retrying, deferred = deferred, [] + for disc in retrying: + try: + await self._ingest_one(kap, disc, counts) + except Exception as exc: + deferred.append(disc) + _log.error( + "disclosure_error", + kap_disclosure_id=str(disc.get("disclosureIndex", "")), + error=str(exc), + ) + lost = len(deferred) + except Exception as exc: error_msg = str(exc) _log.error("scraper_failed", error=error_msg, exc_info=True) @@ -162,48 +201,58 @@ async def run(self, from_date: date, to_date: date) -> ScraperRunResult: run_obj, status="FAILED", records_seen=seen, - records_inserted=inserted, - records_updated=updated, - records_skipped=skipped, + records_inserted=counts.inserted, + records_updated=counts.updated, + records_skipped=counts.skipped, error_message=error_msg, ) raise + # A month that lost even one disclosure is not a SUCCESS. PARTIAL keeps it out + # of the resume ledger's completed set so a later pass comes back for it; the + # ingest is idempotent (disclosure_exists skips what is already stored), so the + # re-run costs only the disclosures that were actually missed. + status = "SUCCESS" if lost == 0 else "PARTIAL" async with get_session() as session: run_obj = await session.get(ScraperRun, run_id) if run_obj: repo = KapRepository(session) await repo.finish_scraper_run( run_obj, - status="SUCCESS", + status=status, records_seen=seen, - records_inserted=inserted, - records_updated=updated, - records_skipped=skipped, + records_inserted=counts.inserted, + records_updated=counts.updated, + records_skipped=counts.skipped, + error_message=f"{lost} disclosures unrecovered" if lost else None, ) + if lost: + _log.warning("chunk_incomplete", seen=seen, lost=lost, status=status) + # A run that stored disclosures but extracted no transactions from most of them # is a failed run wearing a SUCCESS label. Surface the ratio, not just the counts. - if empty_dkb: + if counts.empty_dkb: _log.warning( - "dkb_parse_yield_low" if empty_dkb * 2 >= seen else "dkb_parse_partial", + "dkb_parse_yield_low" if counts.empty_dkb * 2 >= seen else "dkb_parse_partial", seen=seen, - empty_dkb=empty_dkb, - empty_pct=round(empty_dkb / seen * 100, 1) if seen else 0.0, + empty_dkb=counts.empty_dkb, + empty_pct=round(counts.empty_dkb / seen * 100, 1) if seen else 0.0, ) _log.info( "scraper_done", seen=seen, - inserted=inserted, - updated=updated, - skipped=skipped, - empty_dkb=empty_dkb, + inserted=counts.inserted, + updated=counts.updated, + skipped=counts.skipped, + empty_dkb=counts.empty_dkb, + lost=lost, ) return ScraperRunResult( records_seen=seen, - records_inserted=inserted, - records_updated=updated, - records_skipped=skipped, - status="SUCCESS", + records_inserted=counts.inserted, + records_updated=counts.updated, + records_skipped=counts.skipped, + status=status, ) diff --git a/tests/unit/core/test_http_retry.py b/tests/unit/core/test_http_retry.py new file mode 100644 index 0000000..70d7716 --- /dev/null +++ b/tests/unit/core/test_http_retry.py @@ -0,0 +1,43 @@ +"""Which transport failures get retried inline, and which are deferred. + +The retry policy is not a detail: the old one (retry RemoteProtocolError, 5 attempts, +4+8+16+32s exponential) burned 83% of a backfill's wall clock re-poking KAP's WAF and +still lost 12% of the disclosures. A WAF disconnect must fail fast so the scraper can +defer it to a proper cooldown pass; a genuine network blip must still be retried. +""" +import httpx +import pytest + +from trailing_edge.core.http import _is_retryable + + +def test_waf_disconnect_is_not_retried_inline(): + """'Server disconnected without sending a response' is the WAF, not a blip. + + It does not lift in seconds, so an inline retry just re-triggers the block. It + is deferred to the scraper's per-chunk cooldown instead. + """ + assert _is_retryable(httpx.RemoteProtocolError("Server disconnected")) is False + + +def test_transient_network_failures_are_retried_inline(): + assert _is_retryable(httpx.ConnectError("connection refused")) is True + assert _is_retryable(httpx.ReadTimeout("timed out")) is True + + +@pytest.mark.parametrize("status", [429, 503]) +def test_backpressure_statuses_are_retried_inline(status): + resp = httpx.Response(status, request=httpx.Request("GET", "https://kap.org.tr")) + exc = httpx.HTTPStatusError("throttled", request=resp.request, response=resp) + assert _is_retryable(exc) is True + + +@pytest.mark.parametrize("status", [400, 403, 404, 500]) +def test_other_statuses_are_not_retried(status): + resp = httpx.Response(status, request=httpx.Request("GET", "https://kap.org.tr")) + exc = httpx.HTTPStatusError("nope", request=resp.request, response=resp) + assert _is_retryable(exc) is False + + +def test_unrelated_exceptions_are_not_retried(): + assert _is_retryable(ValueError("bad parse")) is False From 23686ebeb28a55822c6e8c2f863ba10a0c68a112 Mon Sep 17 00:00:00 2001 From: cagan Date: Sun, 12 Jul 2026 21:09:32 +0300 Subject: [PATCH 10/29] scrapers: escalate the WAF recovery, batch the existence check, fetch fresh months first Three fixes to a backfill that was sweeping the archive without advancing the frontier. 1. One 90s cooldown recovered only ~53% of the disclosures KAP's WAF dropped, so the month still finished PARTIAL. The resume ledger counts only SUCCESS as done, so a PARTIAL month gets re-run on the next pass - during which the WAF drops a fresh handful and the month goes PARTIAL again. Converges, but each round costs a full re-scan. The cooldown is now a ladder (90s, 4min, 10min): the throttle is volume-triggered, so waiting longer lifts it, only months that actually lost something pay, and a month that recovers everything reaches SUCCESS and is never revisited. 2. disclosure_exists ran per filing, opening a fresh DB session for each of ~200 a month. That is most of the cost of re-running an already-ingested month - and PARTIAL months get re-run by design. Now one query per chunk. 3. Chunk order: untouched months FIRST, PARTIAL stragglers LAST. A PARTIAL month is already ~95% ingested and only needs its WAF-dropped filings; an untouched month has nothing. Replaying PARTIALs in calendar order meant re-listing hundreds of stored filings before any new history landed - observed live: 11 chunks processed, ~180 of every ~200 filings skipped as already-present, and the frontier stuck at 2017-01. Fresh data first; stragglers are cheap to collect at the end. Tests: 199 passing. --- scripts/backfill_kap_insider.py | 45 +++++++++----- src/trailing_edge/scrapers/kap/insider.py | 76 ++++++++++++++++++----- 2 files changed, 88 insertions(+), 33 deletions(-) diff --git a/scripts/backfill_kap_insider.py b/scripts/backfill_kap_insider.py index 2c729b3..5bb9d70 100644 --- a/scripts/backfill_kap_insider.py +++ b/scripts/backfill_kap_insider.py @@ -99,7 +99,7 @@ def generate_monthly_chunks(start: date, end: date) -> list[tuple[date, date]]: return chunks -async def get_completed_chunks() -> set[tuple[date, date]]: +async def _chunks_with_status(status: str) -> set[tuple[date, date]]: from sqlalchemy import select from trailing_edge.core.db import get_session @@ -107,21 +107,25 @@ async def get_completed_chunks() -> set[tuple[date, date]]: async with get_session() as session: stmt = select(ScraperRun.metadata_).where( - ScraperRun.status == "SUCCESS", + ScraperRun.status == status, ScraperRun.metadata_["backfill"].astext == "true", ) rows = (await session.execute(stmt)).scalars().all() - completed: set[tuple[date, date]] = set() + out: set[tuple[date, date]] = set() for meta in rows: if meta and meta.get("from_date") and meta.get("to_date"): - completed.add( + out.add( ( date.fromisoformat(meta["from_date"]), date.fromisoformat(meta["to_date"]), ) ) - return completed + return out + + +async def get_completed_chunks() -> set[tuple[date, date]]: + return await _chunks_with_status("SUCCESS") async def _run_chunk(from_date: date, to_date: date) -> None: @@ -181,28 +185,35 @@ async def main_async( end = date.today() chunks = generate_monthly_chunks(start, end) completed = await get_completed_chunks() + partial = await _chunks_with_status("PARTIAL") + + todo = [c for c in chunks if c not in completed or (c[0].year, c[0].month) in forced] + + # Untouched months FIRST, months that only need their WAF-dropped stragglers SECOND. + # + # A PARTIAL month is already 95% ingested - it just lost a handful of filings to a + # WAF disconnect. An untouched month has nothing. Replaying the PARTIALs in calendar + # order means re-listing and re-checking hundreds of already-stored filings before + # any new history lands, and if the WAF drops a fresh straggler the month goes + # PARTIAL again - so the archive can be swept repeatedly while the frontier never + # moves. Fresh data first; the stragglers are cheap to collect at the end. + fresh = [c for c in todo if c not in partial] + stragglers = [c for c in todo if c in partial] + ordered = fresh + stragglers _log.info( "backfill_start", total_chunks=len(chunks), already_done=len(completed), + fresh=len(fresh), + partial_retry=len(stragglers), dry_run=dry_run, forced_months=len(forced), ) - for from_date, to_date in chunks: - is_forced = (from_date.year, from_date.month) in forced - if (from_date, to_date) in completed and not is_forced: - _log.info( - "chunk_skip", - from_date=from_date, - to_date=to_date, - reason="already_ingested", - ) - continue - + for from_date, to_date in ordered: if dry_run: - _log.info("chunk_dry_run", from_date=from_date, to_date=to_date, forced=is_forced) + _log.info("chunk_dry_run", from_date=from_date, to_date=to_date) continue await _run_chunk(from_date, to_date) diff --git a/src/trailing_edge/scrapers/kap/insider.py b/src/trailing_edge/scrapers/kap/insider.py index f15baa5..e56c6d1 100644 --- a/src/trailing_edge/scrapers/kap/insider.py +++ b/src/trailing_edge/scrapers/kap/insider.py @@ -4,10 +4,12 @@ from datetime import date from pathlib import Path +from sqlalchemy import select + from trailing_edge.core.db import get_session from trailing_edge.core.http import RateLimitedClient from trailing_edge.core.logging import get_logger -from trailing_edge.models.kap import ScraperRun +from trailing_edge.models.kap import KapDisclosure, ScraperRun from trailing_edge.scrapers.base import AbstractScraper from trailing_edge.scrapers.kap.client import KapClient from trailing_edge.scrapers.kap.parser import ( @@ -26,9 +28,20 @@ # diagnosis and a later bulk repair pass. _QUARANTINE_DIR = Path("reports/parse_failures") -# Pause before re-attempting the disclosures KAP's WAF disconnected on. Long enough -# for a volume-triggered throttle to lapse, short enough to stay inside the chunk. -_WAF_COOLDOWN_S = 90 +# Escalating pauses before re-attempting the disclosures KAP's WAF disconnected on. +# +# A single 90s cooldown recovered only ~53% of them, so the chunk still finished +# PARTIAL - and because the resume ledger only counts SUCCESS as done, a PARTIAL month +# forces a whole extra sweep of the archive on the next pass, during which the WAF drops +# a fresh handful and the month goes PARTIAL again. That converges, but each round costs +# a full re-scan (~2 min per already-ingested month, spent almost entirely on +# disclosure_exists checks). +# +# Recovering inside the chunk is strictly cheaper: the throttle is volume-triggered, so +# waiting longer lifts it, and only the months that actually lost something pay. The +# ladder ends when everything is recovered - a month reaches SUCCESS, and the ledger +# never has to come back for it. +_WAF_COOLDOWNS_S = (90, 240, 600) @dataclass @@ -52,7 +65,30 @@ class KapInsiderScraper(AbstractScraper[ScraperRunResult]): def __init__(self, *, backfill: bool = False) -> None: self._backfill = backfill - async def _ingest_one(self, kap: KapClient, disc: dict, counts: _Counts) -> None: + async def _already_stored(self, ids: list[str]) -> set[str]: + """Which of these disclosure ids are already in the database. + + One query for the whole month. Asking per-disclosure opened a fresh session + for each of ~200 filings, which is most of the cost of re-running an + already-ingested month - and PARTIAL months get re-run by design. + """ + if not ids: + return set() + async with get_session() as session: + rows = await session.execute( + select(KapDisclosure.kap_disclosure_id).where( + KapDisclosure.kap_disclosure_id.in_(ids) + ) + ) + return set(rows.scalars().all()) + + async def _ingest_one( + self, + kap: KapClient, + disc: dict, + counts: _Counts, + already_stored: set[str], + ) -> None: """Fetch, parse and store one disclosure. Transport failures propagate: the caller decides whether to retry or defer. @@ -60,10 +96,7 @@ async def _ingest_one(self, kap: KapClient, disc: dict, counts: _Counts) -> None """ kap_disclosure_id = str(disc.get("disclosureIndex", "")) is_correction = bool(disc.get("isChanged") or disc.get("isCorrection") or False) - - async with get_session() as session: - repo = KapRepository(session) - already_exists = await repo.disclosure_exists(kap_disclosure_id) + already_exists = kap_disclosure_id in already_stored if already_exists and not is_correction: counts.skipped += 1 @@ -154,6 +187,9 @@ async def run(self, from_date: date, to_date: date) -> ScraperRunResult: await kap.warmup() disclosures = await kap.fetch_disclosure_list(from_date, to_date) seen = len(disclosures) + already_stored = await self._already_stored( + [str(d.get("disclosureIndex", "")) for d in disclosures] + ) # KAP's WAF intermittently disconnects mid-request. Those failures used # to be logged and skipped, dropping the disclosure outright - 12% of @@ -163,7 +199,7 @@ async def run(self, from_date: date, to_date: date) -> ScraperRunResult: deferred: list[dict] = [] for disc in disclosures: try: - await self._ingest_one(kap, disc, counts) + await self._ingest_one(kap, disc, counts, already_stored) except Exception as exc: deferred.append(disc) _log.warning( @@ -172,20 +208,28 @@ async def run(self, from_date: date, to_date: date) -> ScraperRunResult: error=str(exc), ) - if deferred: + for attempt, cooldown in enumerate(_WAF_COOLDOWNS_S, start=1): + if not deferred: + break _log.info( - "waf_cooldown", deferred=len(deferred), sleep_s=_WAF_COOLDOWN_S + "waf_cooldown", + attempt=attempt, + deferred=len(deferred), + sleep_s=cooldown, ) - await asyncio.sleep(_WAF_COOLDOWN_S) + await asyncio.sleep(cooldown) retrying, deferred = deferred, [] for disc in retrying: try: - await self._ingest_one(kap, disc, counts) + await self._ingest_one(kap, disc, counts, already_stored) except Exception as exc: deferred.append(disc) - _log.error( - "disclosure_error", + last = attempt == len(_WAF_COOLDOWNS_S) + _log.log( + 40 if last else 30, # ERROR on the final round, else WARNING + "disclosure_error" if last else "disclosure_deferred", kap_disclosure_id=str(disc.get("disclosureIndex", "")), + attempt=attempt, error=str(exc), ) lost = len(deferred) From 1041b2d30c2021aa613695c74abdbf95497d7cbc Mon Sep 17 00:00:00 2001 From: cagan Date: Sun, 12 Jul 2026 21:39:11 +0300 Subject: [PATCH 11/29] signals: a trade cannot predate its own filing; a third of the clusters cannot vanish Ran the pipeline end to end for the first time on real data (4,500 filings, 2015-2017, 910 clusters). It produced a convincing result and two reasons not to believe it. 1. Look-ahead guard fired: 91 transactions were dated AFTER the disclosure reporting them - physically impossible. The legacy blotter took max() over EVERY date in the document, so a reference or validity date in the prose could outrank the real trade date. The filing's own publication date is a hard ceiling (a trade cannot be disclosed before it happens), so candidates above it are now discarded, and the canonical path rejects such rows outright. parse_dkb_transactions takes published_on; both the ingest and the quarantine repair pass it. signals/returns.py's AssertionError was right to fire. Guards that stop the pipeline are worth more than results that flow through it. 2. SURVIVORSHIP_BIASED, a new verdict, and the reason this commit exists. The first honest numbers looked like an edge: +2.45% abnormal return at 20 days, t=6.07, p<0.0001, 95% CI [54.4, 62.3] excluding the null. Then: of 910 clusters, only 584 had any price data. The other 326 (36%) were dropped - and their tickers (ACSEL, ANELT, ARBUL, BISAS, BMEKS, ...) are overwhelmingly DELISTED small caps. yfinance does not serve dead BIST names, so the deletion is precisely of the worst outcomes. What remains is a survivor sample, and its mean is not an estimate of anything. compute_base_rate now measures that attrition and, above 10%, returns SURVIVORSHIP_BIASED instead of a verdict - the point estimates still print, but nothing may read them as evidence. This gate exists because +2.45% at t=6.07 was the most convincing wrong answer this pipeline has ever produced, and it would have been believed. Fixing it needs a price source that covers delisted tickers. Until then the honest output is "unusable", and the code now says so instead of leaving it to be discovered. Tests: 199 passing. --- scripts/check_forward_returns.py | 13 +++++++ scripts/fix_company_names.py | 1 - scripts/radar_diff.py | 4 +-- scripts/reparse_quarantine.py | 6 +++- scripts/tsg_acquire_fixtures.py | 4 +-- src/trailing_edge/scrapers/kap/insider.py | 5 ++- src/trailing_edge/scrapers/kap/parser.py | 41 +++++++++++++++++++--- src/trailing_edge/signals/base_rate.py | 42 +++++++++++++++++++++-- 8 files changed, 103 insertions(+), 13 deletions(-) diff --git a/scripts/check_forward_returns.py b/scripts/check_forward_returns.py index 42cee5d..93d9fe3 100644 --- a/scripts/check_forward_returns.py +++ b/scripts/check_forward_returns.py @@ -43,10 +43,18 @@ EDGE_DETECTED, INSUFFICIENT_POWER, NEGATIVE_EDGE, + SURVIVORSHIP_BIASED, compute_base_rate, ) _VERDICT_NOTE = { + SURVIVORSHIP_BIASED: ( + "Too many clusters could not be priced at all, and the tickers behind them are " + "overwhelmingly DELISTED companies. Their absence deletes the worst outcomes and " + "inflates every figure above. This is not a small correction - it is the " + "difference between an edge and an artefact. Nothing here is usable until the " + "price history covers delisted names." + ), INSUFFICIENT_POWER: ( "Sample too small to separate any edge from chance. The estimates above are " "NOT evidence in either direction - do not read a hit rate off them." @@ -115,6 +123,11 @@ async def main() -> None: f" n={s.signals_with_outcome}; " f"~{s.required_n_for_power} needed to detect a 55% hit rate at 80% power." ) + if s.attrition_pct > 0: + print( + f" {s.attrition_pct:.1f}% of clusters had no price data and were dropped " + "(mostly delisted tickers - a NON-random deletion of bad outcomes)." + ) print( f" raw return would have claimed {s.mean_raw_return_pct:+.2f}%, " f"of which the market alone gave {s.mean_benchmark_return_pct:+.2f}%." diff --git a/scripts/fix_company_names.py b/scripts/fix_company_names.py index a331ac3..3d292ec 100644 --- a/scripts/fix_company_names.py +++ b/scripts/fix_company_names.py @@ -25,7 +25,6 @@ async def main() -> None: - from trailing_edge.core.db import get_session, init_db from trailing_edge.core.logging import configure_logging, get_logger configure_logging() diff --git a/scripts/radar_diff.py b/scripts/radar_diff.py index c5e3b06..ae33f97 100644 --- a/scripts/radar_diff.py +++ b/scripts/radar_diff.py @@ -16,7 +16,6 @@ import argparse import asyncio import json -import os import sys from pathlib import Path @@ -49,9 +48,10 @@ def _dedupe_clusters(clusters: list[dict]) -> dict[str, dict]: async def _get_current_clusters() -> list[dict]: """Query insider_clusters and return raw rows as a list of dicts.""" - from trailing_edge.core.db import get_session, init_db from sqlalchemy import text + from trailing_edge.core.db import get_session, init_db + await init_db() async with get_session() as session: rows = ( diff --git a/scripts/reparse_quarantine.py b/scripts/reparse_quarantine.py index 9108bef..b65dfc2 100644 --- a/scripts/reparse_quarantine.py +++ b/scripts/reparse_quarantine.py @@ -62,7 +62,11 @@ async def main_async(dry_run: bool) -> None: orphaned += 1 continue - txs = parse_dkb_transactions(pdf_path.read_bytes(), ticker=row.ticker) + txs = parse_dkb_transactions( + pdf_path.read_bytes(), + ticker=row.ticker, + published_on=row.published_at.date() if row.published_at else None, + ) if not txs: still_failing += 1 continue diff --git a/scripts/tsg_acquire_fixtures.py b/scripts/tsg_acquire_fixtures.py index af7dc56..3f01db4 100644 --- a/scripts/tsg_acquire_fixtures.py +++ b/scripts/tsg_acquire_fixtures.py @@ -23,9 +23,9 @@ async def main() -> None: - from playwright.async_api import async_playwright + from pdfminer.high_level import extract_text as pdf_extract - import re + from playwright.async_api import async_playwright FIXTURES_DIR.mkdir(parents=True, exist_ok=True) diff --git a/src/trailing_edge/scrapers/kap/insider.py b/src/trailing_edge/scrapers/kap/insider.py index e56c6d1..9e76316 100644 --- a/src/trailing_edge/scrapers/kap/insider.py +++ b/src/trailing_edge/scrapers/kap/insider.py @@ -118,7 +118,10 @@ async def _ingest_one( if obj_id: pdf_bytes = await kap.fetch_pdf(obj_id) txs = parse_dkb_transactions( - pdf_bytes, ticker=dto.ticker, insider_name="" + pdf_bytes, + ticker=dto.ticker, + insider_name="", + published_on=dto.published_at.date() if dto.published_at else None, ) else: body = detail.get("disclosureBody", "") or "" diff --git a/src/trailing_edge/scrapers/kap/parser.py b/src/trailing_edge/scrapers/kap/parser.py index a39c282..a7385eb 100644 --- a/src/trailing_edge/scrapers/kap/parser.py +++ b/src/trailing_edge/scrapers/kap/parser.py @@ -3,6 +3,7 @@ import io import re +from datetime import date from decimal import ROUND_HALF_UP, Decimal, InvalidOperation from trailing_edge.core.logging import get_logger @@ -378,6 +379,7 @@ def _parse_legacy_blotter( ticker: str, insider_name: str, relation_type: str, + published_on: date | None = None, ) -> list[KapInsiderTxDTO]: lines = [ln.strip().replace("\xa0", "") for ln in text.splitlines() if ln.strip()] @@ -395,14 +397,27 @@ def _parse_legacy_blotter( # Latest trade date in the document. The blotter's own rows and the narrative # carry the same dates, so a plain scan is order-independent. + # + # But a blind max() over every date in the document also picks up dates that are + # not trade dates at all, and it produced 91 transactions dated AFTER the filing + # that reports them - physically impossible, and it tripped the look-ahead guard + # in signals/returns.py (which was right to fire). A trade cannot be disclosed + # before it happens, so the filing's own publication date is a hard ceiling: any + # candidate above it is not a trade date and is discarded. dates = [] for tok in _DATE_RE.findall(text): try: - dates.append(parse_kap_date(tok)) + d = parse_kap_date(tok) except ValueError: - pass + continue + if published_on is not None and d > published_on: + continue + dates.append(d) if not dates: - _log.warning("dkb_row_rejected", reason="legacy: no parseable trade date") + _log.warning( + "dkb_row_rejected", + reason="legacy: no trade date at or before the publication date", + ) return [] tx_date = max(dates) @@ -578,6 +593,7 @@ def parse_dkb_transactions( pdf_bytes: bytes, ticker: str = "", insider_name: str = "", + published_on: date | None = None, ) -> list[KapInsiderTxDTO]: """Extract transactions from a Java-unwrapped DKB PDF. @@ -587,6 +603,12 @@ def parse_dkb_transactions( accepted row is validated against arithmetic the form itself guarantees; rows that fail are logged as dkb_row_rejected and dropped - never stored with guessed fields. + + ``published_on`` (the filing's own KAP publication date) is a hard ceiling on + every transaction date: a trade cannot be disclosed before it happens. Rows + dated after it are rejected rather than stored - they are, by construction, a + misread date. Omitting it keeps the old permissive behaviour, so callers that + genuinely have no publication date still work. """ from pdfminer.high_level import extract_text @@ -600,7 +622,9 @@ def parse_dkb_transactions( # Legacy (2015-2020) blotter form: route by its unambiguous totals marker. # The modern canonical form never contains it (checked on both modern fixtures). if _LEGACY_BUY_LABEL in text.upper(): - return _parse_legacy_blotter(text, ticker, insider_name, relation_type) + return _parse_legacy_blotter( + text, ticker, insider_name, relation_type, published_on + ) # Extract price range from narrative price_try: Decimal | None = None @@ -621,6 +645,15 @@ def parse_dkb_transactions( _log.warning("dkb_row_rejected", reason=str(exc), row=row) continue + if published_on is not None and tx_date > published_on: + _log.warning( + "dkb_row_rejected", + reason="transaction dated after its own disclosure", + tx_date=str(tx_date), + published_on=str(published_on), + ) + continue + fields, reason = _map_canonical_row(row[1:]) if fields is None: _log.warning("dkb_row_rejected", reason=reason, row=row) diff --git a/src/trailing_edge/signals/base_rate.py b/src/trailing_edge/signals/base_rate.py index cf307ee..35df05c 100644 --- a/src/trailing_edge/signals/base_rate.py +++ b/src/trailing_edge/signals/base_rate.py @@ -34,7 +34,7 @@ from dataclasses import dataclass from decimal import ROUND_HALF_UP, Decimal -from sqlalchemy import select +from sqlalchemy import func, select from trailing_edge.core.config import get_config from trailing_edge.core.db import get_session @@ -52,6 +52,19 @@ NO_EDGE_DETECTED = "NO_EDGE_DETECTED" EDGE_DETECTED = "EDGE_DETECTED" NEGATIVE_EDGE = "NEGATIVE_EDGE" +SURVIVORSHIP_BIASED = "SURVIVORSHIP_BIASED" + +# Above this share of clusters dropped for want of price data, no verdict is safe. +# +# A cluster with no price series contributes nothing to the mean - and the tickers +# that have no price series are overwhelmingly the ones that were DELISTED. Measured +# on a real run: 326 of 910 clusters (36%) silently vanished this way, and the names +# behind them (ACSEL, ANELT, ARBUL, BISAS, BMEKS...) are exactly the small caps that +# went to zero. Their absence is not random; it removes the worst outcomes and leaves +# the mean looking like an edge. This gate exists because the number it suppresses - +# +2.45% abnormal at 20 days, t=6.07 - was the most convincing wrong answer this +# pipeline has ever produced. +_MAX_ATTRITION = 0.10 @dataclass @@ -77,6 +90,10 @@ class BaseRateStats: mean_raw_return_pct: Decimal mean_benchmark_return_pct: Decimal + # Share of clusters that could not be priced at all. Their tickers are mostly + # DELISTED, so the attrition is not random - it deletes the worst outcomes. + attrition_pct: Decimal + required_n_for_power: int verdict: str @@ -142,6 +159,7 @@ def _empty(horizon_days: int, total: int, need: int) -> BaseRateStats: worst_abnormal_return_pct=_ZERO, mean_raw_return_pct=_ZERO, mean_benchmark_return_pct=_ZERO, + attrition_pct=_ZERO, required_n_for_power=need, verdict=INSUFFICIENT_POWER, ) @@ -179,6 +197,17 @@ async def compute_base_rate( ) rows = (await session.execute(stmt)).all() + # Clusters that produced NO priceable outcome at all. Counting only the rows that + # HAVE an outcome hides them - and they are not missing at random. + async with get_session() as session: + total_clusters = ( + await session.execute( + select(func.count()) + .select_from(InsiderCluster) + .where(InsiderCluster.cluster_score >= min_score) + ) + ).scalar_one() + total_signals = len(rows) scored = [r for r in rows if r.abnormal_return_pct is not None] @@ -206,7 +235,14 @@ async def compute_base_rate( t_stat, p_value = t_test_vs_zero(ar) mean_ar = statistics.fmean(ar) - if n < need: + attrition = (total_clusters - n) / total_clusters if total_clusters else 0.0 + + if attrition > _MAX_ATTRITION: + # Not a footnote. A third of the clusters vanishing because their companies + # were delisted removes the worst outcomes and manufactures an edge; no + # verdict computed on the survivors can be trusted. + verdict = SURVIVORSHIP_BIASED + elif n < need: verdict = INSUFFICIENT_POWER elif p_value >= 0.05: verdict = NO_EDGE_DETECTED @@ -222,6 +258,7 @@ async def compute_base_rate( required_n=need, mean_abnormal_return_pct=round(mean_ar, 4), p_value=round(p_value, 4), + attrition_pct=round(attrition * 100, 1), verdict=verdict, ) @@ -241,6 +278,7 @@ async def compute_base_rate( worst_abnormal_return_pct=_dec(min(ar)), mean_raw_return_pct=_dec(statistics.fmean(raw)) if raw else _ZERO, mean_benchmark_return_pct=_dec(statistics.fmean(bench)) if bench else _ZERO, + attrition_pct=_dec(attrition * 100), required_n_for_power=need, verdict=verdict, ) From 2fdfe845f13a6fb65f875befc6de7eb25e9ed2bc Mon Sep 17 00:00:00 2001 From: cagan Date: Sun, 12 Jul 2026 22:36:57 +0300 Subject: [PATCH 12/29] data: fetch prices per ticker, not per batch - a third of the "survivorship" was a bug The base rate reported SURVIVORSHIP_BIASED because 326 of 910 clusters (36%) had no price data, and their tickers looked exactly like delisted small caps. Most of that was true. A large part was not - it was fetch_and_store_prices. One request for all 245 tickers looked efficient. yfinance answers a batch containing an unknown symbol with an error naming SEVERAL ("Quote not found for symbol: GEREL, VRSGS.IS") and drops the innocent ones with it. ACSEL, ANELT and ATLAS each have 750+ days of history in yfinance and still ended up with zero price rows. A cluster with no prices is silently excluded from the base rate, and the tickers a batch is most likely to poison are the obscure ones - so the loss wore survivorship bias as a costume. Prices are now fetched in batches of 20, and every ticker a batch fails to deliver is retried alone. One bad symbol can no longer take its neighbours down. Second cause, also not survivorship: two tickers were RENAMED, not delisted. KAP files them under the old code and yfinance only serves the new one (GYHOL -> GLYHO, AKFEN -> AKFGY, each verified individually to return a full 2015-2017 series under the new symbol and nothing under the old). _TICKER_ALIASES maps them. Attrition: 35.8% -> 31.0%. The remaining 31% is the real thing, and it does not yield to engineering. MCTAS (118 clusters), MENBA (59), OZBAL (42), IZTAR (25) and others are genuinely dead, and yfinance serves nothing for a delisted BIST name - not even the years it was trading. For a 2015-2017 study that is a systematic hole exactly where the worst outcomes live, so SURVIVORSHIP_BIASED still fires and the numbers below it remain unusable: 20d N=634 hit 60.2% AR +2.85% t=7.10 p<0.0001 <- NOT evidence Closing that gap needs a price source that covers delisted tickers. Until then the honest output is "unusable", and the pipeline says so rather than leaving it to be discovered by whoever trusts the number. Tests: 199 passing. --- src/trailing_edge/data/prices.py | 80 ++++++++++++++++++++++++++++---- 1 file changed, 71 insertions(+), 9 deletions(-) diff --git a/src/trailing_edge/data/prices.py b/src/trailing_edge/data/prices.py index e3366fb..3bd6567 100644 --- a/src/trailing_edge/data/prices.py +++ b/src/trailing_edge/data/prices.py @@ -17,6 +17,21 @@ _log = get_logger(__name__) +# BIST tickers that were RENAMED, not delisted. KAP files them under the old code; +# yfinance only serves the new one, so without this map their price history looks +# missing - and a cluster with no prices is silently dropped from the base rate, +# which reads exactly like survivorship bias. Verified individually against yfinance +# (each new symbol returns a full 2015-2017 series; each old one returns nothing). +_TICKER_ALIASES: dict[str, str] = { + "GYHOL": "GLYHO", # Global Yatirim Holding + "AKFEN": "AKFGY", # Akfen -> Akfen Gayrimenkul +} + + +def _yf_symbol(ticker: str) -> str: + return f"{_TICKER_ALIASES.get(ticker, ticker)}.IS" + + def _sync_yf_download(yf_tickers: list[str], start: str, end: str): import yfinance as yf @@ -39,14 +54,62 @@ async def fetch_and_store_prices( Fetch OHLCV for each ticker from yfinance and upsert into price_history. tickers: BIST short codes (e.g. ["ASELS", "ISCTR"]) - ".IS" suffix added here. - Returns {ticker: rows_upserted}. Delisted / missing tickers are skipped with a warning. + Returns {ticker: rows_upserted}. Tickers yfinance genuinely has no data for are + reported as 0 rows. + + Downloaded in SMALL BATCHES, and every batch that comes back short is retried one + ticker at a time. + + One request for all 245 tickers looked efficient and quietly lost a third of them: + yfinance answers a batch containing an unknown symbol with an error naming several + ("Quote not found for symbol: GEREL, VRSGS.IS") and drops the innocent ones with it. + ACSEL, ANELT and ATLAS each have 750+ days of history and still ended up with no + price rows at all. Because a cluster with no prices is silently excluded from the + base rate, and the tickers most likely to be dropped are the obscure ones, the loss + read exactly like survivorship bias - it was a batching bug wearing its costume. """ - import pandas as pd - if not tickers: return {} - yf_tickers = [f"{t}.IS" for t in tickers] + results: dict[str, int] = {} + batch_size = 20 + + for i in range(0, len(tickers), batch_size): + batch = tickers[i : i + batch_size] + got = await _fetch_batch(batch, start_date, end_date) + + # Any ticker the batch did not deliver is retried alone, so one bad symbol + # cannot take its neighbours down with it. + missing = [t for t in batch if got.get(t, 0) == 0] + for ticker in missing: + solo = await _fetch_batch([ticker], start_date, end_date) + got[ticker] = solo.get(ticker, 0) + if got[ticker] == 0: + _log.warning("price_ticker_no_data", ticker=ticker) + + results.update(got) + + recovered = sum(1 for v in results.values() if v > 0) + _log.info( + "prices_fetch_done", + requested=len(tickers), + with_data=recovered, + without_data=len(tickers) - recovered, + ) + return results + + +async def _fetch_batch( + tickers: list[str], + start_date: date, + end_date: date, +) -> dict[str, int]: + """Download and store one batch. Returns {ticker: rows_stored}, 0 where absent.""" + import pandas as pd + + yf_tickers = [_yf_symbol(t) for t in tickers] + # Map the downloaded symbol back to the ticker KAP files under. + by_symbol = {_yf_symbol(t): t for t in tickers} loop = asyncio.get_event_loop() try: @@ -60,12 +123,11 @@ async def fetch_and_store_prices( ), ) except Exception as exc: - _log.error("yfinance_download_failed", error=str(exc)) - return {} + _log.warning("yfinance_batch_failed", tickers=len(tickers), error=str(exc)) + return dict.fromkeys(tickers, 0) if df is None or df.empty: - _log.warning("yfinance_empty_response") - return {} + return dict.fromkeys(tickers, 0) results: dict[str, int] = {} is_multi = isinstance(df.columns, pd.MultiIndex) @@ -81,7 +143,7 @@ async def fetch_and_store_prices( async with get_session() as session: for yf_ticker in yf_tickers: - ticker = yf_ticker.replace(".IS", "") + ticker = by_symbol[yf_ticker] try: if is_multi: level_vals = df.columns.get_level_values(ticker_level) From e9c1d7ff30d965619289bb7b4ad36072a48c79d0 Mon Sep 17 00:00:00 2001 From: cagan Date: Sun, 12 Jul 2026 23:08:48 +0300 Subject: [PATCH 13/29] data: load the exchange's own bulletin - survivorship bias 31% -> 1.5%, and the 60-day sign flips yfinance is survivorship-contaminated for BIST: it serves nothing for a delisted ticker, not even the years it was actively trading. 31% of insider clusters had no price data, and the names behind them (MCTAS, MENBA, OZBAL, IZTAR...) are exactly the small caps that went to zero. compute_base_rate correctly refused to return a verdict. Borsa Istanbul's official end-of-day bulletin (PP_GUNSONUFIYATHACIM) is survivorship-clean by construction - it records what actually traded each day, so a company that later delisted is still there for every day it lived. 2015-2026 carries 1,204,914 equity rows across 729 tickers; yfinance had 185. cluster attrition: 31.0% -> 1.5% MCTAS 743 days, OZBAL 1,525, IZTAR 1,597, SERVE 1,645 - all previously unpriceable Corporate actions. The bulletin quotes raw prints, so a bonus issue halves the close overnight and a naive window return reads -50%. The archive has no adjustment file, but the bulletin restates one field itself: ONCEKI KAPANIS FIYATI, the previous close as the exchange adjusts it. So close(t)/prev_recorded(t) is the true one-day total return, and chaining those gives a series whose ratios are correct across any corporate action. Verified: BOSSA 2016-06-08 reads +0.39% chained, where the raw print shows -9%. 30 of 24,967 day-pairs (0.12%) carry such an adjustment. What the clean data says - and it is not what the biased data said: horizon N hit% mean AR t verdict 5d 934 64.2% +1.46% 7.86 EDGE_DETECTED 20d 934 60.4% +2.04% 5.99 EDGE_DETECTED 60d 934 41.0% -2.80% -4.58 NEGATIVE_EDGE N clears the 784 power gate and attrition clears the 10% gate, so these verdicts are the first this pipeline has been entitled to return. The 60-day horizon FLIPPED SIGN: +4.18% under survivorship bias, -2.80% clean. The dead companies' collapses were being deleted from the sample, and they were most of what happened at 60 days. That single number is the clearest measure of how large the bias was - and of why the gate existed. Not a green light. Costs are still not deducted, and these are the illiquid names where spread and impact are worst; no VBTS tradability filter is applied; and the window is 2015-2017, a single regime. The honest reading is that a short-horizon effect is statistically present before costs, and that holding past ~20 days destroys it. The bulletin also carries BRUT TAKAS and GECICI DURDURMA - the VBTS flags the pipeline lacks. That is the obvious next use of this source. Tests: 199 passing. --- scripts/load_official_prices.py | 282 ++++++++++++++++++++++++++++++++ 1 file changed, 282 insertions(+) create mode 100644 scripts/load_official_prices.py diff --git a/scripts/load_official_prices.py b/scripts/load_official_prices.py new file mode 100644 index 0000000..d7caba0 --- /dev/null +++ b/scripts/load_official_prices.py @@ -0,0 +1,282 @@ +"""Load Borsa Istanbul's official end-of-day bulletin into price_history. + +Why this exists +--------------- +yfinance is survivorship-contaminated for BIST. It serves nothing for a delisted +ticker - not even the years it was actively trading - so a 2015-2017 study loses +every company that died afterwards. Measured on this dataset: 31% of insider +clusters had no price data, and the names behind them (MCTAS, MENBA, OZBAL, +IZTAR...) are exactly the small caps that went to zero. That deletion is not +random; it removes the worst outcomes and manufactures an edge. compute_base_rate +returns SURVIVORSHIP_BIASED for precisely this reason. + +The exchange's own bulletin (PP_GUNSONUFIYATHACIM, "Pay Piyasasi Gun Sonu Fiyat +Hacim") is survivorship-clean by construction: it records what actually traded each +day, so a company that later delisted is still in the file for every day it lived. +One 2016 month carries 420 distinct equities against yfinance's 185 for the whole +universe, and 8 of the 10 tickers yfinance could not price are present with a full +month of closes. + +Notes +----- +- ZIPs are read in place, never extracted: the archive is ~1.8 GB and the disk has + little headroom. +- The bulletin also carries BRUT TAKAS (gross settlement) and GECICI DURDURMA + (suspended) - the VBTS tradability flags the pipeline currently lacks. Not loaded + yet; recorded here as the obvious next use of this source. +Corporate actions +----------------- +The bulletin quotes RAW traded prices, so a bonus issue or dividend drop looks like a +loss: a 100% bedelsiz halves the printed close overnight, and a naive return over that +window reads -50%. There is no adjustment file in the archive (corporate_actions/ is +empty), but the bulletin adjusts one field itself - ONCEKI KAPANIS FIYATI, the previous +close as the exchange restates it. Where a corporate action occurred, that value is the +ADJUSTED prior close, so + + close(t) / previous_close_as_recorded(t) + +is the true one-day total return, dividends and bonus issues already netted out. +Measured over 2016-06..2016-08: 30 of 24,967 day-pairs disagree with the actual prior +close (0.12%), with factors like 0.953 (ANHYT) and 0.908 (BOSSA) - exactly the dividend +drops it should catch. + +This loader therefore stores a chained TOTAL-RETURN series, not the raw print: + + A(first) = close(first) + A(t) = A(t-1) * close(t) / previous_close_as_recorded(t) + +so (A(exit) / A(entry) - 1) is a correct cumulative return across any corporate action. +The stored value is an index level, not the traded price - which is what the return +math needs, and the only thing it is used for. + +Usage: + uv run python scripts/load_official_prices.py --from 2015-01 +""" +from __future__ import annotations + +import asyncio +import csv +import io +import sys +import zipfile +from datetime import datetime +from decimal import Decimal, InvalidOperation +from pathlib import Path + +import click + +sys.path.insert(0, "src") + +from sqlalchemy.dialects.postgresql import insert as pg_insert # noqa: E402 + +from trailing_edge.core.db import get_session, init_db # noqa: E402 +from trailing_edge.core.logging import configure_logging, get_logger # noqa: E402 +from trailing_edge.models.signal import PriceHistory # noqa: E402 + +_log = get_logger(__name__) + +ARCHIVE = Path( + r"C:\Users\cagan\bist-trading-system\data\bist_datastore_archive\prices_official" +) + +# Column positions in the bulletin (0-based), verified against the 2016-06 header. +C_DATE = 0 +C_CODE = 1 +C_TYPE = 7 +C_PREV_CLOSE = 16 # ONCEKI KAPANIS FIYATI - restated by the exchange across a CA +C_OPEN = 17 +C_LOW = 20 +C_HIGH = 21 +C_CLOSE = 22 +C_VOLUME = 29 + +# Ignore adjustment factors within this band: they are rounding in the printed +# previous close, not a corporate action. +_CA_EPS = Decimal("0.005") + +# Equities only. The bulletin also lists warrants, ETFs, rights and so on. +EQUITY_TYPE = "MSPOTEQT" + + +def _dec(raw: str) -> Decimal | None: + raw = raw.strip() + if not raw or raw in {"0", "0.0", "0.00"}: + return None + try: + return Decimal(raw) + except InvalidOperation: + return None + + +def _rows_from_csv(text: str) -> list[dict]: + reader = csv.reader(io.StringIO(text), delimiter=";") + out: list[dict] = [] + for parts in reader: + if len(parts) <= C_VOLUME: + continue # header lines (Turkish + English) and any short row + if parts[C_TYPE].strip() != EQUITY_TYPE: + continue + try: + price_date = datetime.strptime(parts[C_DATE].strip(), "%Y-%m-%d").date() + except ValueError: + continue + + close = _dec(parts[C_CLOSE]) + if close is None or close <= 0: + continue # a day with no trade has no close to measure against + + # "ACSEL.E" -> "ACSEL". The suffix marks the instrument series, not the company. + ticker = parts[C_CODE].strip().split(".")[0].upper() + if not ticker: + continue + + volume_raw = parts[C_VOLUME].strip() + try: + volume = int(float(volume_raw)) if volume_raw else None + except ValueError: + volume = None + + out.append( + { + "ticker": ticker, + "price_date": price_date, + "open_try": _dec(parts[C_OPEN]), + "high_try": _dec(parts[C_HIGH]), + "low_try": _dec(parts[C_LOW]), + "close_try": close, + "volume": volume, + "_prev_close": _dec(parts[C_PREV_CLOSE]), + } + ) + return out + + +def chain_total_return(rows: list[dict]) -> list[dict]: + """Replace raw closes with a chained total-return index, per ticker. + + A raw close series is wrong across a corporate action - a bonus issue halves the + print and a naive window return reads it as a 50% loss. The exchange restates + ONCEKI KAPANIS FIYATI across such an event, so close(t)/prev_recorded(t) is the + true one-day total return. Chaining those gives a series whose ratios are correct + returns over any interval, which is all the pipeline asks of it. + + OHLC fields are scaled by the same running factor so they stay on the series' + scale rather than silently mixing raw prints with adjusted closes. + """ + by_ticker: dict[str, list[dict]] = {} + for r in rows: + by_ticker.setdefault(r["ticker"], []).append(r) + + out: list[dict] = [] + for series in by_ticker.values(): + series.sort(key=lambda r: r["price_date"]) + level = series[0]["close_try"] + factor = Decimal(1) + prev_raw_close = series[0]["close_try"] + + for i, r in enumerate(series): + raw_close = r["close_try"] + if i > 0: + prev_recorded = r.pop("_prev_close", None) + base = ( + prev_recorded + if prev_recorded and prev_recorded > 0 + else prev_raw_close + ) + if base > 0: + level = level * raw_close / base + factor = level / raw_close if raw_close > 0 else factor + else: + r.pop("_prev_close", None) + + def _scale(v: Decimal | None) -> Decimal | None: + return (v * factor).quantize(Decimal("0.0001")) if v is not None else None + + prev_raw_close = raw_close + r["close_try"] = level.quantize(Decimal("0.0001")) + r["open_try"] = _scale(r["open_try"]) + r["high_try"] = _scale(r["high_try"]) + r["low_try"] = _scale(r["low_try"]) + out.append(r) + return out + + +def _read_month(path: Path) -> list[dict]: + if path.suffix.lower() == ".zip": + # Read in place - extracting 1.8 GB of archive is not affordable here. + with zipfile.ZipFile(path) as zf: + name = next((n for n in zf.namelist() if n.lower().endswith(".csv")), None) + if name is None: + return [] + raw = zf.read(name) + else: + raw = path.read_bytes() + return _rows_from_csv(raw.decode("utf-8", errors="replace")) + + +async def _store(rows: list[dict]) -> int: + if not rows: + return 0 + async with get_session() as session: + for i in range(0, len(rows), 4000): + chunk = rows[i : i + 4000] + stmt = pg_insert(PriceHistory.__table__).values(chunk) + await session.execute( + stmt.on_conflict_do_update( + constraint="uq_price_ticker_date", + set_={ + "close_try": stmt.excluded.close_try, + "open_try": stmt.excluded.open_try, + "high_try": stmt.excluded.high_try, + "low_try": stmt.excluded.low_try, + "volume": stmt.excluded.volume, + }, + ) + ) + return len(rows) + + +async def main_async(start: str) -> None: + await init_db() + + if not ARCHIVE.is_dir(): + raise SystemExit(f"Official price archive not found: {ARCHIVE}") + + files = sorted( + f + for f in ARCHIVE.iterdir() + if f.suffix.lower() in {".zip", ".csv"} + and (m := f.name.split(".M.")[-1][:6]) + and m.isdigit() + and m >= start.replace("-", "") + ) + if not files: + raise SystemExit(f"No bulletin files at or after {start}") + + click.echo(f"Reading {len(files)} monthly bulletins from {files[0].name} ...") + + # The whole span is read before anything is stored: the total-return chain needs a + # ticker's series unbroken, and a corporate action does not respect month boundaries. + raw: list[dict] = [] + for path in files: + raw.extend(_read_month(path)) + + tickers = {r["ticker"] for r in raw} + click.echo(f" {len(raw):,} equity rows, {len(tickers)} distinct tickers") + + rows = chain_total_return(raw) + click.echo(" chained to a corporate-action-adjusted total-return series") + + stored = await _store(rows) + click.echo(f"Done: {stored:,} price rows stored for {len(tickers)} tickers.") + + +@click.command() +@click.option("--from", "start", default="2015-01", help="First month, YYYY-MM.") +def main(start: str) -> None: + configure_logging() + asyncio.run(main_async(start)) + + +if __name__ == "__main__": + main() From 80bbe41488e9e37c0d2014ffae75b9d8741ef580 Mon Sep 17 00:00:00 2001 From: cagan Date: Mon, 13 Jul 2026 00:46:58 +0300 Subject: [PATCH 14/29] scrapers: stop sleeping inside every month; sweep fast, recover in passes Half the backfill was spent asleep. Measured over 8 chunks: 124 minutes elapsed, 64 of them (52%) inside the per-chunk WAF cooldown ladder - 7.1 minutes per month, some 14 hours across the archive - while the frontier stood still. The ladder (90s, 4min, 10min) existed to recover a month fully before moving on. It is unnecessary, because a blocked request is now nearly free: RemoteProtocolError fails fast (core/http._is_retryable), so a chunk takes whatever the WAF lets through, defers the rest, and finishes PARTIAL. The resume ledger counts only SUCCESS as done, and the ingest is idempotent - so a later pass re-fetches exactly what was lost and skips the hundreds already stored. Recovery belongs in a second sweep, not in a sleep inside every month. - cooldown ladder -> a single 90s round (it clears ~half the deferrals immediately, measured 76 -> 34, which keeps the PARTIAL set small) - the backfill now re-sweeps the PARTIAL months in up to 4 recovery passes, each opening with a 30-minute pause, stopping when nothing is left or a pass makes no progress The pause is the part that matters. Measured: 40 minutes of silence took the block rate from 61% to 0%, while cutting the request rate did nothing (1 RPS and 3 RPS both hit 100% blocked once the throttle was hot). KAP's WAF meters cumulative volume, not instantaneous rate - so the right shape is burst then rest, not a slow trickle. The rate limit is raised to 3 RPS accordingly: when requests are getting through, 40 of them take 29s at 3 RPS against 88s at 1 RPS, and when they are not, the rate is irrelevant. --- scripts/backfill_kap_insider.py | 58 ++++++++++++++++++++++- src/trailing_edge/scrapers/kap/insider.py | 27 ++++++----- 2 files changed, 72 insertions(+), 13 deletions(-) diff --git a/scripts/backfill_kap_insider.py b/scripts/backfill_kap_insider.py index 5bb9d70..de9b680 100644 --- a/scripts/backfill_kap_insider.py +++ b/scripts/backfill_kap_insider.py @@ -81,6 +81,13 @@ # 60s it replaces. _CHUNK_GAP_S = 2 +# After the archive has been swept once, re-sweep whatever the WAF cost us. Silence is +# the only thing that measurably clears KAP's throttle - 40 minutes of it took the block +# rate from 61% to 0%, while cutting the request rate did nothing - so each pass opens +# with a real pause rather than trickling straight back in. +_RECOVERY_PASSES = 4 +_RECOVERY_PAUSE_S = 1800 # 30 minutes + def generate_monthly_chunks(start: date, end: date) -> list[tuple[date, date]]: chunks = [] @@ -219,7 +226,56 @@ async def main_async( await _run_chunk(from_date, to_date) await asyncio.sleep(_CHUNK_GAP_S) - _log.info("backfill_complete", dry_run=dry_run) + if dry_run: + _log.info("backfill_complete", dry_run=True) + return + + # Recovery passes. + # + # A month that lost disclosures to the WAF is PARTIAL, and the per-chunk cooldown no + # longer waits for it - that wait cost half the runtime while the frontier stood + # still. Instead the archive is swept once at speed, and then re-swept for exactly + # what was missed. The ingest is idempotent, so a re-swept month re-fetches only its + # lost disclosures and skips the hundreds already stored. + # + # A pause between passes is the one thing that actually clears KAP's throttle + # (measured: 40 minutes of silence took the block rate from 61% to 0%, while slowing + # the request rate did nothing). Passes stop when nothing is left, when a pass makes + # no progress, or after the cap. + for attempt in range(1, _RECOVERY_PASSES + 1): + remaining = await _chunks_with_status("PARTIAL") + remaining = [c for c in remaining if c in set(ordered)] + if not remaining: + break + + _log.info( + "recovery_pass_start", + attempt=attempt, + months=len(remaining), + pause_s=_RECOVERY_PAUSE_S, + ) + await asyncio.sleep(_RECOVERY_PAUSE_S) + + for from_date, to_date in sorted(remaining): + await _run_chunk(from_date, to_date) + await asyncio.sleep(_CHUNK_GAP_S) + + still = [c for c in await _chunks_with_status("PARTIAL") if c in set(ordered)] + _log.info( + "recovery_pass_done", + attempt=attempt, + before=len(remaining), + after=len(still), + ) + if len(still) >= len(remaining): + _log.warning("recovery_stalled", months=len(still)) + break + + left = [c for c in await _chunks_with_status("PARTIAL") if c in set(ordered)] + if left: + _log.warning("backfill_incomplete", partial_months=len(left)) + + _log.info("backfill_complete", dry_run=False, partial_months=len(left)) @click.command() diff --git a/src/trailing_edge/scrapers/kap/insider.py b/src/trailing_edge/scrapers/kap/insider.py index 9e76316..cdfe04a 100644 --- a/src/trailing_edge/scrapers/kap/insider.py +++ b/src/trailing_edge/scrapers/kap/insider.py @@ -28,20 +28,23 @@ # diagnosis and a later bulk repair pass. _QUARANTINE_DIR = Path("reports/parse_failures") -# Escalating pauses before re-attempting the disclosures KAP's WAF disconnected on. +# One short pause before re-attempting the disclosures KAP's WAF disconnected on. # -# A single 90s cooldown recovered only ~53% of them, so the chunk still finished -# PARTIAL - and because the resume ledger only counts SUCCESS as done, a PARTIAL month -# forces a whole extra sweep of the archive on the next pass, during which the WAF drops -# a fresh handful and the month goes PARTIAL again. That converges, but each round costs -# a full re-scan (~2 min per already-ingested month, spent almost entirely on -# disclosure_exists checks). +# This was an escalating ladder (90s, 4min, 10min) that recovered a month fully before +# moving on. It worked, and it was the wrong place to spend the time: measured over 8 +# chunks, 52% of the entire run was spent asleep inside it - 7.1 minutes per month, some +# 14 hours across the archive - while the frontier stood still. # -# Recovering inside the chunk is strictly cheaper: the throttle is volume-triggered, so -# waiting longer lifts it, and only the months that actually lost something pay. The -# ladder ends when everything is recovered - a month reaches SUCCESS, and the ledger -# never has to come back for it. -_WAF_COOLDOWNS_S = (90, 240, 600) +# The ladder is unnecessary because a blocked request is now nearly free. +# RemoteProtocolError fails fast (core/http._is_retryable), so a chunk can take whatever +# the WAF lets through, defer the rest, and finish PARTIAL. The resume ledger counts only +# SUCCESS as done, so a later pass comes back for exactly the missed disclosures - and the +# ingest is idempotent, so that re-fetch costs only what was actually lost. Recovery +# belongs in a second pass over the archive, not in a sleep inside every month. +# +# One 90s round is kept: it is cheap and clears roughly half the deferrals immediately +# (measured 76 -> 34), which keeps the PARTIAL set small enough for the recovery pass. +_WAF_COOLDOWNS_S = (90,) @dataclass From 1799071b68042bc341b360129f8846e23beda0e9 Mon Sep 17 00:00:00 2001 From: cagan Date: Mon, 13 Jul 2026 01:12:40 +0300 Subject: [PATCH 15/29] signals: the edge is real and it is not tradeable - the spread eats it The answer, measured end to end on survivorship-clean data: horizon N=1032 gross AR cost net AR t verdict 5d +0.57% 3.34% -2.77% -13.11 loses money 20d +1.76% 3.34% -1.58% -4.04 loses money 60d +2.09% 3.34% -1.24% -1.98 loses money round-trip cost: median 1.94%, p75 4.24% The signal has genuine predictive content - gross abnormal return is significantly positive at every horizon (20d: +1.76%, t=5.12). It is also not tradeable, because insider clusters fire in illiquid BIST small caps and the bid-ask spread on those names is larger than the alpha. Nothing survives the round trip. The cost is not assumed, it is estimated per trade from that stock's own OHLC in the 30 sessions before entry (Abdi-Ranaldo 2017, RFS 30(12)) - the only estimator that works on the delisted names the exchange bulletin carries and no quote feed does. A flat fee would have flattered the answer; the measured median is 1.94% and the upper quartile 4.24%. Three bugs had to be fixed first, and each one had moved the number: 1. My own cost script dropped 96% of the sample. abdi_ranaldo_spread returned None when a quiet window drove gamma >= 0, and I treated that as "cannot estimate" and excluded the trade. The survivors' gross AR came out NEGATIVE where the full sample's was positive - the exclusion selects on exactly the price behaviour the signal is about. A quiet window is not a free trade: the estimate now widens the window and is floored at one tick, and nothing is dropped for it. 2. The exchange bulletin changed format at 2015-12, and the older layout is a different file: session-level rows, DD.MM.YYYY dates, and a decimal separator that switches from comma to dot mid-year without touching the header. Read with the modern column map, 2015-01..2015-11 loaded ZERO rows. Reading '54.50' as Turkish gave fifty-four thousand five hundred, a thousandfold error that overflowed NUMERIC(20,4) - which was lucky, because a subtler scale error would have gone straight into the returns. The chain now also rejects any one-day ratio outside [0.5, 2.0]: BIST's daily limit is +/-10%, so anything past that is a data fault, and chaining it compounds through the whole series. 3. get_price_and_date_after_days returned the first row after the signal date however far away it was. With 2015 prices missing, 1,278 outcomes were "entered" on 2015-12-01 - where the series happened to begin - for signals fired months earlier. Those are not late entries, they are fabricated ones: nobody could have bought at a price that had not printed yet. Entry now refuses a gap wider than 10 days. With those fixed the sample is 1,055 clusters, 6% unpriceable, and the verdict is stable. What this settles: the gross signal is not the question, and never was. The question is whether it clears the cost of the names it fires in, and it does not. --- scripts/load_official_prices.py | 160 ++++++++++++++++++++++- scripts/net_of_cost.py | 165 +++++++++++++++++++++++ src/trailing_edge/data/prices.py | 33 +++++ src/trailing_edge/signals/costs.py | 201 +++++++++++++++++++++++++++++ 4 files changed, 555 insertions(+), 4 deletions(-) create mode 100644 scripts/net_of_cost.py create mode 100644 src/trailing_edge/signals/costs.py diff --git a/scripts/load_official_prices.py b/scripts/load_official_prices.py index d7caba0..17d982e 100644 --- a/scripts/load_official_prices.py +++ b/scripts/load_official_prices.py @@ -59,7 +59,7 @@ import io import sys import zipfile -from datetime import datetime +from datetime import date, datetime from decimal import Decimal, InvalidOperation from pathlib import Path @@ -94,6 +94,12 @@ # previous close, not a corporate action. _CA_EPS = Decimal("0.005") +# BIST's daily price limit is +/-10% (20% for some segments), so a one-day total return +# beyond this band cannot be a market move. It is a data fault, and chaining it would +# compound the error through the rest of the series. +_MIN_DAY_RATIO = Decimal("0.5") +_MAX_DAY_RATIO = Decimal("2.0") + # Equities only. The bulletin also lists warrants, ETFs, rights and so on. EQUITY_TYPE = "MSPOTEQT" @@ -108,6 +114,131 @@ def _dec(raw: str) -> Decimal | None: return None +# The bulletin changed format at the end of 2015. Files up to 2015-11 use a narrower, +# session-level layout with Turkish decimals: +# +# TARIH;SEANS NO;PAZAR;PAY KODU;SERI KODU;MIN;MAX;ONCEKI KAPANIS;KAPANIS;AOF;HACIM;MIKTAR +# 02.01.2015;1;L;ACSEL;E;50,50;53;52,30;50,95;51,24;... +# +# Two rows per ticker per day (one per session), dates as DD.MM.YYYY, commas for decimal +# points, and no instrument-type column. Read with the modern column map these yield ZERO +# rows - which is exactly what happened: 2015-01..2015-11 never loaded, 551 of 934 +# clusters had no history to price against, and entries silently snapped to 2015-12-01 +# (the first date that DID load) for signals fired months earlier. +_LEGACY_DATE = 0 +_LEGACY_SESSION = 1 +_LEGACY_CODE = 3 +_LEGACY_SERIES = 4 +_LEGACY_LOW = 5 +_LEGACY_HIGH = 6 +_LEGACY_PREV_CLOSE = 7 +_LEGACY_CLOSE = 8 +_LEGACY_VOLUME = 11 + + +def _tr_dec(raw: str) -> Decimal | None: + """Parse a legacy bulletin price. The decimal separator is not stable. + + The pre-2015-12 files switch convention mid-year without changing the header: + + 2015-01 ACSEL min='50,50' close='50,95' <- comma is the decimal point + 2015-11 ACSEL min='54.50' close='54.50' <- dot is the decimal point + + Assuming Turkish convention throughout reads '54.50' as fifty-four thousand five + hundred - a thousandfold error that made the chained series overflow a + NUMERIC(20,4) column outright. It overflowed, which was lucky: a subtler scale + error would have passed straight through into the returns. + + So the separator is inferred, not assumed. A comma always marks the decimal; + failing that, a lone dot does. Only a dot with exactly three digits after it, and + no comma anywhere, is read as a thousands group. + """ + raw = raw.strip() + if not raw: + return None + + if "," in raw: + raw = raw.replace(".", "").replace(",", ".") + elif raw.count(".") == 1: + whole, frac = raw.split(".") + if len(frac) == 3 and len(whole) <= 3: + raw = whole + frac + else: + raw = raw.replace(".", "") + + try: + v = Decimal(raw) + except InvalidOperation: + return None + return v if v > 0 else None + + +def _rows_from_legacy_csv(text: str) -> list[dict]: + """Parse the pre-2015-12 session-level bulletin. + + Sessions collapse to one row per ticker-day: the LAST session's close is the day's + close (that is what the next day's ONCEKI KAPANIS refers to), high/low span both + sessions, and volume is their sum. + """ + reader = csv.reader(io.StringIO(text), delimiter=";") + days: dict[tuple[str, date], dict] = {} + + for parts in reader: + if len(parts) <= _LEGACY_VOLUME: + continue + if parts[_LEGACY_SERIES].strip().upper() != "E": + continue + try: + price_date = datetime.strptime(parts[_LEGACY_DATE].strip(), "%d.%m.%Y").date() + session_no = int(parts[_LEGACY_SESSION].strip() or 0) + except ValueError: + continue + + close = _tr_dec(parts[_LEGACY_CLOSE]) + if close is None: + continue + ticker = parts[_LEGACY_CODE].strip().upper() + if not ticker: + continue + + v = _tr_dec(parts[_LEGACY_VOLUME]) + volume = int(v) if v is not None else 0 + + key = (ticker, price_date) + low, high = _tr_dec(parts[_LEGACY_LOW]), _tr_dec(parts[_LEGACY_HIGH]) + prev = _tr_dec(parts[_LEGACY_PREV_CLOSE]) + + cur = days.get(key) + if cur is None: + days[key] = { + "ticker": ticker, + "price_date": price_date, + "open_try": None, + "high_try": high, + "low_try": low, + "close_try": close, + "volume": volume, + "_prev_close": prev, + "_session": session_no, + } + continue + + if session_no >= cur["_session"]: + cur["close_try"] = close + cur["_session"] = session_no + if high is not None: + cur["high_try"] = high if cur["high_try"] is None else max(cur["high_try"], high) + if low is not None: + cur["low_try"] = low if cur["low_try"] is None else min(cur["low_try"], low) + cur["volume"] = (cur["volume"] or 0) + volume + + out = [] + for r in days.values(): + r.pop("_session", None) + out.append(r) + return out + + def _rows_from_csv(text: str) -> list[dict]: reader = csv.reader(io.StringIO(text), delimiter=";") out: list[dict] = [] @@ -183,8 +314,23 @@ def chain_total_return(rows: list[dict]) -> list[dict]: if prev_recorded and prev_recorded > 0 else prev_raw_close ) - if base > 0: - level = level * raw_close / base + # A single day's total return outside this band is not a return, it is a + # data fault - a mis-parsed decimal separator, a unit change, a stale + # previous close. Chaining it compounds the error through every later + # day of the series. Reject the ratio and carry the level forward flat, + # loudly: the day is wrong, the rest of the history is not. + ratio = raw_close / base if base > 0 else Decimal(1) + if not (_MIN_DAY_RATIO <= ratio <= _MAX_DAY_RATIO): + _log.warning( + "implausible_daily_ratio", + ticker=r["ticker"], + date=str(r["price_date"]), + ratio=float(ratio), + close=float(raw_close), + prev=float(base), + ) + ratio = Decimal(1) + level = level * ratio factor = level / raw_close if raw_close > 0 else factor else: r.pop("_prev_close", None) @@ -211,7 +357,13 @@ def _read_month(path: Path) -> list[dict]: raw = zf.read(name) else: raw = path.read_bytes() - return _rows_from_csv(raw.decode("utf-8", errors="replace")) + text = raw.decode("utf-8", errors="replace") + # Route on the header, not the filename: the layout changed at 2015-12 but the + # naming did not. + head = text[:400].upper() + if "PAY KODU" in head and "SEANS NO" in head: + return _rows_from_legacy_csv(text) + return _rows_from_csv(text) async def _store(rows: list[dict]) -> int: diff --git a/scripts/net_of_cost.py b/scripts/net_of_cost.py new file mode 100644 index 0000000..fded6f4 --- /dev/null +++ b/scripts/net_of_cost.py @@ -0,0 +1,165 @@ +"""Does the abnormal return survive the cost of trading it? + +The base rate says +1.46% at 5 days and +2.04% at 20 days, before costs. This is the +test that matters, and it is a hard one here: insider clusters fire in illiquid BIST +small caps - the names with the widest spreads. A flat fee would flatter the answer, so +the spread is estimated per trade from that stock's own OHLC (Abdi-Ranaldo 2017), which +is also the only estimator that works on the delisted names the exchange bulletin gives +us and no quote feed does. + +Clusters whose spread cannot be estimated are DROPPED, not priced at zero: a trade whose +cost is unknown is not a trade with no cost. + +Usage: + uv run python scripts/net_of_cost.py + uv run python scripts/net_of_cost.py --order-try 50000 +""" +from __future__ import annotations + +import asyncio +import math +import statistics +import sys +from decimal import Decimal + +import click + +sys.path.insert(0, "src") + +from sqlalchemy import text # noqa: E402 + +from trailing_edge.core.db import get_session, init_db # noqa: E402 +from trailing_edge.core.logging import configure_logging # noqa: E402 +from trailing_edge.signals.base_rate import wilson_interval # noqa: E402 +from trailing_edge.signals.costs import round_trip_cost # noqa: E402 + +_LOOKBACK = 30 # trading days of OHLC before entry, for the spread estimator + + +async def main_async(order_try: Decimal) -> None: + await init_db() + + async with get_session() as session: + rows = ( + await session.execute( + text( + """ + SELECT o.horizon_days, o.abnormal_return_pct, o.entry_date, c.ticker + FROM signal_outcomes o + JOIN insider_clusters c ON c.id = o.cluster_id + WHERE o.abnormal_return_pct IS NOT NULL AND o.entry_date IS NOT NULL + """ + ) + ) + ).all() + + # One OHLC pull per (ticker, entry_date): the estimator needs the 30 sessions + # before entry, which is information available at entry - no look-ahead. + cache: dict[tuple[str, object], tuple | None] = {} + + async def window(ticker: str, entry) -> tuple | None: + key = (ticker, entry) + if key in cache: + return cache[key] + px = ( + await session.execute( + text( + """ + SELECT close_try, high_try, low_try, volume + FROM price_history + WHERE ticker = :t AND price_date < :d + ORDER BY price_date DESC LIMIT :n + """ + ), + {"t": ticker, "d": entry, "n": _LOOKBACK}, + ) + ).all() + if len(px) < 22: + cache[key] = None + return None + px = list(reversed(px)) + closes = [r[0] for r in px] + highs = [r[1] or r[0] for r in px] + lows = [r[2] or r[0] for r in px] + vols = [r[3] or 0 for r in px] + adv = Decimal( + str(statistics.fmean(float(c) * float(v) for c, v in zip(closes, vols, strict=True))) + ) + cache[key] = (closes, highs, lows, adv) + return cache[key] + + by_h: dict[int, list[tuple[float, float]]] = {} + dropped = 0 + costs: list[float] = [] + + for horizon, ar, entry, ticker in rows: + w = await window(ticker, entry) + if w is None: + dropped += 1 + continue + closes, highs, lows, adv = w + rt = round_trip_cost(closes, highs, lows, order_try, adv) + if rt is None: + dropped += 1 + continue + costs.append(float(rt.total_pct)) + by_h.setdefault(horizon, []).append((float(ar), float(rt.total_pct))) + + if not by_h: + click.echo("No priceable clusters.") + return + + click.echo("") + click.echo(f"=== Abnormal return, NET of round-trip cost (order {order_try:,.0f} TRY) ===") + click.echo(" spread: Abdi-Ranaldo (2017) from the stock's own OHLC, per trade") + click.echo(f" dropped (no cost estimate): {dropped}") + if costs: + cs = sorted(costs) + click.echo( + f" round-trip cost: median {cs[len(cs)//2]:.2f}% " + f"p25 {cs[len(cs)//4]:.2f}% p75 {cs[3*len(cs)//4]:.2f}%" + ) + click.echo("") + hdr = f"{'HORIZON':>7} {'N':>5} {'GROSS AR%':>10} {'COST%':>7} {'NET AR%':>8} {'HIT%':>6} {'95% CI':>15} {'t':>6} VERDICT" + click.echo(hdr) + click.echo("-" * len(hdr)) + + for horizon in sorted(by_h): + pairs = by_h[horizon] + gross = [a for a, _ in pairs] + net = [a - c for a, c in pairs] + n = len(net) + mean_net = statistics.fmean(net) + sd = statistics.stdev(net) if n > 1 else 0.0 + t = mean_net / (sd / math.sqrt(n)) if sd > 0 else 0.0 + hits = sum(1 for v in net if v > 0) + lo, hi = wilson_interval(hits, n) + + if abs(t) < 1.96: + verdict = "NO EDGE (net)" + elif mean_net > 0: + verdict = "EDGE SURVIVES COSTS" + else: + verdict = "LOSES MONEY (net)" + + click.echo( + f"{horizon:>6}d {n:>5} {statistics.fmean(gross):>10.2f} " + f"{statistics.fmean([c for _, c in pairs]):>7.2f} {mean_net:>8.2f} " + f"{hits/n*100:>6.1f} {f'[{lo*100:.1f}, {hi*100:.1f}]':>15} {t:>6.2f} {verdict}" + ) + click.echo("") + + +@click.command() +@click.option( + "--order-try", + default=25000.0, + help="Position size in TRY. Retail default; impact scales with sqrt(size/ADV).", +) +def main(order_try: float) -> None: + configure_logging() + asyncio.run(main_async(Decimal(str(order_try)))) + + +if __name__ == "__main__": + main() diff --git a/src/trailing_edge/data/prices.py b/src/trailing_edge/data/prices.py index 3bd6567..06fbce5 100644 --- a/src/trailing_edge/data/prices.py +++ b/src/trailing_edge/data/prices.py @@ -256,11 +256,18 @@ async def get_price_after_days( return price +# A trading day the pipeline treats as "the next session" cannot be an arbitrary +# distance away. Beyond this many calendar days from the reference date, the row is not +# the next session at all - it is the first session after a hole in the data. +_MAX_SESSION_GAP_DAYS = 10 + + async def get_price_and_date_after_days( ticker: str, from_date: date, horizon_days: int, session: AsyncSession | None = None, + max_gap_days: int = _MAX_SESSION_GAP_DAYS, ) -> tuple[Decimal | None, date | None]: """ Close price AND its trading date, horizon_days trading days after from_date. @@ -272,9 +279,35 @@ async def get_price_and_date_after_days( / VBTS measure), which is exactly the population insider clusters concentrate in - so taking offsets on each series independently would silently compare mismatched windows. + + Entry (horizon_days=1) additionally REFUSES a row that is not actually adjacent. + The query asks for the first row after from_date, and if the price history simply + has no rows near from_date it happily returns one from months later - which the + caller then books as a t+1 entry. Measured: 1,278 outcomes were entered on + 2015-12-01 because that is where the price series began, for signals fired months + earlier. Those are not late entries, they are fabricated ones: nobody could have + bought at a price that had not printed yet. A gap wider than max_gap_days yields + None, and the cluster is dropped as unpriceable - which is what it is. """ async def _query(s: AsyncSession) -> tuple[Decimal | None, date | None]: + # The FIRST session after from_date anchors the window. If the series has no + # row anywhere near from_date, that first row is not "tomorrow" - it is the + # far side of a hole, and nothing may be entered against it. + first = ( + await s.execute( + select(PriceHistory.price_date) + .where( + PriceHistory.ticker == ticker, + PriceHistory.price_date > from_date, + ) + .order_by(PriceHistory.price_date.asc()) + .limit(1) + ) + ).scalar_one_or_none() + if first is None or (first - from_date).days > max_gap_days: + return None, None + result = await s.execute( select(PriceHistory.close_try, PriceHistory.price_date) .where( diff --git a/src/trailing_edge/signals/costs.py b/src/trailing_edge/signals/costs.py new file mode 100644 index 0000000..9600280 --- /dev/null +++ b/src/trailing_edge/signals/costs.py @@ -0,0 +1,201 @@ +"""Round-trip transaction cost for an insider-cluster trade. + +Why this decides the question +----------------------------- +The base rate reports a +1.46% abnormal return at 5 days and +2.04% at 20 days, +before costs. Whether either survives is not a detail - it IS the question. And it +is a hard question here precisely because the signal fires where trading is most +expensive: insider clusters concentrate in illiquid BIST small caps, exactly the +names with the widest spreads. + +So the cost has to be estimated PER TRADE from that stock's own price action, not +assumed as a flat fee. A flat 0.2% would flatter the result; the spread on a thin +name is routinely several times that. + +The model +--------- + round_trip = 2 x (commission + BSMV) + spread + 2 x impact + +- Spread: Abdi & Ranaldo (2017), "A Simple Estimation of Bid-Ask Spreads from Daily + Close, High, and Low Prices", RFS 30(12). It recovers the effective spread from + OHLC alone, which is all the exchange bulletin gives us - and it gives it for + delisted names too, where no quote data survives at all. + + c_t = ln(close_t) + m_t = (ln(high_t) + ln(low_t)) / 2 + gamma = mean[(c_t - m_t)(c_{t-1} - m_t)] over the window + spread = 2 sqrt(max(-gamma, 0)) + + Crossing the spread once on entry and once on exit costs it in full, which is why + it enters the round trip at 1x, not 2x. + +- Impact: Kyle (1985) / Almgren et al. (2005) square-root law, + impact = lambda * sigma_daily * sqrt(order_value / ADV), paid on each side. + At retail size (order << ADV) this is small - and showing that it is small is the + point, because it is the one cost that a small account genuinely escapes. + +- Commission is charged per side; BSMV (5%) applies to the commission itself. + Gains tax is zero for a domestic individual under Gecici-67, so it is not modelled. + +Every number is an estimate and the direction of error matters: an underestimated +cost turns a null into an edge. Where the data is too thin to estimate a spread at +all, the caller gets None rather than a cheap default. +""" +from __future__ import annotations + +import math +from dataclasses import dataclass +from decimal import Decimal + +# BIST retail defaults. Deliberately mid-range rather than best-case: a cost model +# tuned to the cheapest broker on the friendliest name is a way of assuming the answer. +COMMISSION_PCT = Decimal("0.0015") # 15 bps per side +BSMV_RATE = Decimal("0.05") # levied on the commission +KYLE_LAMBDA = Decimal("1.0") # uncalibrated; see kyle_impact_pct + +_SPREAD_WINDOW = 21 + + +@dataclass(frozen=True) +class RoundTrip: + spread_pct: Decimal + impact_pct: Decimal + commission_pct: Decimal + total_pct: Decimal + + +def _gamma( + closes: list[Decimal], highs: list[Decimal], lows: list[Decimal], window: int +) -> float | None: + n = min(len(closes), len(highs), len(lows)) + if n < window + 1: + return None + c_s, h_s, l_s = closes[-(window + 1) :], highs[-(window + 1) :], lows[-(window + 1) :] + + diffs: list[float] = [] + for c, h, low in zip(c_s, h_s, l_s, strict=True): + if c <= 0 or h <= 0 or low <= 0: + return None + mid = (math.log(float(h)) + math.log(float(low))) / 2 + diffs.append(math.log(float(c)) - mid) + + products = [diffs[i] * diffs[i - 1] for i in range(1, len(diffs))] + if not products: + return None + return sum(products) / len(products) + + +def tick_floor_pct(price: Decimal) -> Decimal: + """Smallest spread the exchange's price grid permits, as a fraction. + + A stock cannot be quoted tighter than one tick, so this is the hard floor under any + spread estimate. BIST's equity tick is 0.01 TRY below 20 TRY and 0.02 TRY above - + the coarse grid is why a 3 TRY small cap cannot have a spread under ~33bp however + quiet its prints look. + """ + if price <= 0: + return Decimal(0) + tick = Decimal("0.01") if price < 20 else Decimal("0.02") + return tick / price + + +def abdi_ranaldo_spread_pct( + closes: list[Decimal], + highs: list[Decimal], + lows: list[Decimal], + window: int = _SPREAD_WINDOW, +) -> Decimal | None: + """Effective bid-ask spread as a fraction, from OHLC alone (Abdi-Ranaldo 2017). + + The estimator produces gamma >= 0 whenever a window happens to be quiet, and the + formula then reports a spread of exactly zero. Treating that as "cannot estimate" + and dropping the trade is a trap I walked into: it discarded 96% of the sample, and + the survivors' gross abnormal return came out NEGATIVE where the full sample's was + positive. The dropout is not random - it selects on precisely the price behaviour + the signal is about, so the exclusion silently inverts the answer. + + A quiet window is not evidence of a free trade. So a degenerate gamma widens the + window rather than voiding the estimate, and whatever survives is floored at one + tick - the tightest quote the exchange's own price grid allows. None is returned + only when there is genuinely not enough price history to look at. + """ + for w in (window, 60, 120): + g = _gamma(closes, highs, lows, w) + if g is None: + continue + if g < 0: + spread = 2 * math.sqrt(-g) + est = Decimal(str(spread)) + break + else: + est = Decimal(0) + + if not closes: + return None + if _gamma(closes, highs, lows, min(window, max(len(closes) - 1, 2))) is None: + return None + + floor = tick_floor_pct(closes[-1]) + return max(est, floor).quantize(Decimal("0.000001")) + + +def kyle_impact_pct( + closes: list[Decimal], + order_value_try: Decimal, + adv_try: Decimal, + window: int = _SPREAD_WINDOW, + lambda_kyle: Decimal = KYLE_LAMBDA, +) -> Decimal: + """One-sided square-root market impact. + + impact = lambda * sigma_daily * sqrt(order_value / ADV) + + lambda is an uncalibrated 1.0 - calibrating it needs execution data this project + does not have. At retail order sizes the term is small enough that the error does + not change any conclusion; at institutional size it would, and the number should + not be trusted there. + """ + if adv_try <= 0 or order_value_try <= 0 or len(closes) < 3: + return Decimal(0) + + tail = closes[-(window + 1) :] + rets = [ + math.log(float(tail[i]) / float(tail[i - 1])) + for i in range(1, len(tail)) + if tail[i] > 0 and tail[i - 1] > 0 + ] + if len(rets) < 2: + return Decimal(0) + + mean = sum(rets) / len(rets) + sigma = math.sqrt(sum((r - mean) ** 2 for r in rets) / (len(rets) - 1)) + impact = float(lambda_kyle) * sigma * math.sqrt(float(order_value_try / adv_try)) + return Decimal(str(impact)).quantize(Decimal("0.000001")) + + +def round_trip_cost( + closes: list[Decimal], + highs: list[Decimal], + lows: list[Decimal], + order_value_try: Decimal, + adv_try: Decimal, +) -> RoundTrip | None: + """Total cost of entering and exiting one position, as a percent of notional. + + None when the spread cannot be estimated: a trade whose cost is unknown must be + excluded from the sample, not priced at zero. + """ + spread = abdi_ranaldo_spread_pct(closes, highs, lows) + if spread is None: + return None + + impact = kyle_impact_pct(closes, order_value_try, adv_try) + commission = 2 * COMMISSION_PCT * (1 + BSMV_RATE) + + total = spread + 2 * impact + commission + return RoundTrip( + spread_pct=spread * 100, + impact_pct=impact * 100, + commission_pct=commission * 100, + total_pct=(total * 100).quantize(Decimal("0.0001")), + ) From ffe1f07c2ea6c309bb2a06e6f84c133f637ef201 Mon Sep 17 00:00:00 2001 From: cagan Date: Mon, 13 Jul 2026 02:11:46 +0300 Subject: [PATCH 16/29] data: load the VBTS tradability flags - and they turn out not to be the constraint The exchange bulletin carries BRUT TAKAS (gross settlement) and GECICI DURDURMA (suspended). A name under gross settlement cannot be round-tripped the way a backtest assumes; a suspended one cannot be traded at all. The pipeline had been booking entries in both, and insider clusters fire precisely in the volatile small caps VBTS targets - so this looked like it might matter a great deal. Measured: 143,012 gross-settlement ticker-days across 451 names (11% of all price rows). But of 1,055 cluster entries, only 11 (1.0%) land on a restricted day, and none on a suspended one. So the flags are now loaded and the gap is closed, and the honest result is that it was never the binding constraint. The cost of crossing the spread is - a median 1.94% round trip against a gross abnormal return of 1.76% at 20 days. Naming a limitation and then measuring it away is worth as much as naming it and finding it fatal; what is not worth anything is leaving it unmeasured and implied. Tests: 199 passing. --- migrations/versions/0007_tradability_flags.py | 38 +++++++++++++++++++ scripts/load_official_prices.py | 12 +++++- src/trailing_edge/models/signal.py | 12 ++++++ 3 files changed, 60 insertions(+), 2 deletions(-) create mode 100644 migrations/versions/0007_tradability_flags.py diff --git a/migrations/versions/0007_tradability_flags.py b/migrations/versions/0007_tradability_flags.py new file mode 100644 index 0000000..5c8bc3f --- /dev/null +++ b/migrations/versions/0007_tradability_flags.py @@ -0,0 +1,38 @@ +"""Add VBTS tradability flags to price_history. + +Borsa Istanbul's Volatilite Bazli Tedbir Sistemi escalates measures on a volatile +stock: short-selling ban, then GROSS SETTLEMENT (brut takas), then a single-price +auction. A name under gross settlement cannot be entered and exited the way a +backtest assumes, and a suspended name cannot be traded at all - yet the pipeline +has been booking entries in both. + +The exchange's own bulletin carries these as columns (BRUT TAKAS, GECICI DURDURMA), +so the filter costs nothing but reading them. + +Revision ID: 0007 +Revises: 0006 +Create Date: 2026-07-13 +""" +import sqlalchemy as sa +from alembic import op + +revision = "0007" +down_revision = "0006" +branch_labels = None +depends_on = None + + +def upgrade() -> None: + op.add_column( + "price_history", + sa.Column("gross_settlement", sa.Boolean(), nullable=False, server_default="false"), + ) + op.add_column( + "price_history", + sa.Column("suspended", sa.Boolean(), nullable=False, server_default="false"), + ) + + +def downgrade() -> None: + op.drop_column("price_history", "suspended") + op.drop_column("price_history", "gross_settlement") diff --git a/scripts/load_official_prices.py b/scripts/load_official_prices.py index 17d982e..3158cca 100644 --- a/scripts/load_official_prices.py +++ b/scripts/load_official_prices.py @@ -83,6 +83,8 @@ C_DATE = 0 C_CODE = 1 C_TYPE = 7 +C_GROSS_SETTLE = 13 # BRUT TAKAS - VBTS measure: cash-and-carry only +C_SUSPENDED = 15 # GECICI DURDURMA - trading halted C_PREV_CLOSE = 16 # ONCEKI KAPANIS FIYATI - restated by the exchange across a CA C_OPEN = 17 C_LOW = 20 @@ -218,6 +220,8 @@ def _rows_from_legacy_csv(text: str) -> list[dict]: "low_try": low, "close_try": close, "volume": volume, + "gross_settlement": False, # not published in the pre-2015-12 bulletin + "suspended": False, "_prev_close": prev, "_session": session_no, } @@ -276,6 +280,8 @@ def _rows_from_csv(text: str) -> list[dict]: "low_try": _dec(parts[C_LOW]), "close_try": close, "volume": volume, + "gross_settlement": parts[C_GROSS_SETTLE].strip() == "1", + "suspended": parts[C_SUSPENDED].strip() == "1", "_prev_close": _dec(parts[C_PREV_CLOSE]), } ) @@ -370,8 +376,8 @@ async def _store(rows: list[dict]) -> int: if not rows: return 0 async with get_session() as session: - for i in range(0, len(rows), 4000): - chunk = rows[i : i + 4000] + for i in range(0, len(rows), 3000): + chunk = rows[i : i + 3000] stmt = pg_insert(PriceHistory.__table__).values(chunk) await session.execute( stmt.on_conflict_do_update( @@ -382,6 +388,8 @@ async def _store(rows: list[dict]) -> int: "high_try": stmt.excluded.high_try, "low_try": stmt.excluded.low_try, "volume": stmt.excluded.volume, + "gross_settlement": stmt.excluded.gross_settlement, + "suspended": stmt.excluded.suspended, }, ) ) diff --git a/src/trailing_edge/models/signal.py b/src/trailing_edge/models/signal.py index 81f133e..7b3a709 100644 --- a/src/trailing_edge/models/signal.py +++ b/src/trailing_edge/models/signal.py @@ -6,6 +6,7 @@ from sqlalchemy import ( BigInteger, + Boolean, Date, ForeignKey, Index, @@ -32,6 +33,17 @@ class PriceHistory(Base): low_try: Mapped[Decimal | None] = mapped_column(Numeric(20, 4)) close_try: Mapped[Decimal] = mapped_column(Numeric(20, 4), nullable=False) volume: Mapped[int | None] = mapped_column(BigInteger) + + # VBTS tradability, straight from the exchange bulletin. A name under gross + # settlement cannot be round-tripped the way a backtest assumes, and a suspended + # one cannot be traded at all - the signal fires precisely in the names this + # happens to. + gross_settlement: Mapped[bool] = mapped_column( + Boolean, nullable=False, server_default="false" + ) + suspended: Mapped[bool] = mapped_column( + Boolean, nullable=False, server_default="false" + ) fetched_at: Mapped[datetime] = mapped_column( TIMESTAMP(timezone=True), server_default=func.now(), nullable=False ) From 4c554674526d725f34f92b1082920317f4267b33 Mon Sep 17 00:00:00 2001 From: cagan Date: Mon, 13 Jul 2026 02:13:16 +0300 Subject: [PATCH 17/29] docs: the README states the finding, not the aspiration The repository is the argument, and it was making the wrong one - opening on what the pipeline scrapes rather than what it found. A reader had to reach docs/METHODOLOGY.md to learn that the question has an answer. The answer leads now: the insider-cluster signal has genuine predictive content (20d gross abnormal return +1.76%, t=5.12, N=1,032, survivorship-clean) and is not tradeable, because the names it fires in cost a median 1.94% to round-trip and an upper- quartile 4.24%. A gross number is not an edge; an edge is what is left after the market takes its cut. The six silent data faults are listed as a table rather than buried in commit history - each one had moved the number, and a reader deciding whether to trust the result is entitled to see what had to be true for it to be trustworthy. The sample-output block is replaced by the commands that reproduce the result, including net_of_cost.py, with its real output. Claims a reader cannot re-run are decoration. --- README.md | 130 ++++++++++++++++++++++++++++++++++-------------------- 1 file changed, 81 insertions(+), 49 deletions(-) diff --git a/README.md b/README.md index aec831c..e867b40 100644 --- a/README.md +++ b/README.md @@ -1,9 +1,9 @@ # TrailingEdge -> **Asynchronous Python data engine that ingests SPK II-15.1 insider -> transaction disclosures from KAP (kap.org.tr), measures empirical -> forward returns, and produces per-company insider-activity briefs for -> BIST-listed companies.** +> **Do BIST insiders' disclosed purchases predict returns you can actually capture?** +> +> They predict. You cannot capture them. This is the pipeline that measures both, +> and the second half is the finding. [![Python](https://img.shields.io/badge/python-3.12+-blue.svg)](https://www.python.org) [![PostgreSQL](https://img.shields.io/badge/postgres-16-336791.svg)](https://www.postgresql.org) @@ -14,6 +14,45 @@ --- +## The result + +Insider-cluster events on Borsa İstanbul, 2015-2018. Entry is t+1 after the KAP +disclosure is *public*, returns are measured in excess of XU100 over the same held +interval, and the round-trip cost is estimated per trade from that stock's own OHLC +(Abdi-Ranaldo 2017) rather than assumed as a flat fee. + +| Horizon | N | Gross AR | Cost | **Net AR** | t (net) | +|---|---:|---:|---:|---:|---:| +| 5d | 1,032 | +0.57% | 3.34% | **−2.77%** | −13.11 | +| 20d | 1,032 | +1.76% | 3.34% | **−1.58%** | −4.04 | +| 60d | 1,032 | +2.09% | 3.34% | **−1.24%** | −1.98 | + +**The signal is real. The gross abnormal return is significantly positive at every +horizon** (20d: +1.76%, t = 5.12). **And it is not tradeable**, because insider clusters +fire in illiquid small caps whose bid-ask spread is wider than the alpha: the median +round trip costs 1.94%, the upper quartile 4.24%. Nothing survives crossing it twice. + +That is the whole finding, and it is why this repository exists. A gross number is not +an edge; an edge is what is left after the market takes its cut. + +Everything below is the machinery required to be able to say that honestly - and the +audit trail of the six silent data faults that had to be found first, each of which had +moved the number: + +| | | +|---|---| +| KAP's list endpoint truncates at 2,000 rows, keeping the newest | ~75% of every month was being discarded | +| The transaction-date regex accepted `/` but not `.` | every filing before 2021 parsed to **zero** transactions | +| Fixed column indices into a variable-width table | 21% of stored rows were silently wrong | +| WAF disconnects were caught and skipped | 12% of disclosures vanished, run still reported `SUCCESS` | +| Prices fetched in one batch; one bad symbol poisoned the rest | looked exactly like survivorship bias | +| yfinance serves nothing for a delisted ticker | 31% of clusters dropped - the dead ones, the worst outcomes | + +The last of those was fixed by loading Borsa İstanbul's own end-of-day bulletin, which is +survivorship-clean by construction: 1.3M rows, 747 tickers, against yfinance's 185. It +also carries the VBTS tradability flags - which, once measured, turned out to touch only +1% of entries and not to be the constraint at all. + ## What it does - **Scrapes** Turkey's public-disclosure platform (KAP) for SPK II-15.1 @@ -33,13 +72,19 @@ insider-transaction history, board-interlock graphs, and (optionally) Türkiye Ticaret Sicil Gazetesi cross-references. -> **No edge is claimed.** This is measurement infrastructure, not a strategy. The -> sample sizes reached so far are far below what is needed to distinguish an edge -> from chance (~784 events; see `reports/sample/README.md`), and no transaction -> cost or VBTS tradability filter is applied yet - so any positive figure this -> pipeline produces is an upper bound, before frictions. The -> [open issues](docs/METHODOLOGY.md#5-known-open-issues) are documented rather -> than left for the reader to find. +- **Prices trades against the exchange's own bulletin**, not a retail feed: survivorship- + clean by construction, corporate-action-adjusted by chaining the exchange's restated + previous close, and carrying the VBTS gross-settlement and suspension flags. +- **Refuses to answer when it cannot.** `compute_base_rate` returns + `INSUFFICIENT_POWER` below ~784 events and `SURVIVORSHIP_BIASED` when too many + clusters cannot be priced. Both gates fired during this work, and both were right. + +> **What is claimed, precisely:** a statistically strong *gross* abnormal return +> (20d: +1.76%, t = 5.12, N = 1,032, survivorship-clean) that does **not** survive a +> per-trade cost estimate. The window is 2015-2018 - a single regime - so the result is +> not yet regime-conditional, and that is stated rather than glossed. Remaining gaps are +> in [`docs/METHODOLOGY.md`](docs/METHODOLOGY.md#5-known-open-issues), not left for a +> reader to discover. ## Türkçe özet @@ -97,46 +142,33 @@ Forensic brief for a single ticker: trailingedge report forensic KAPLM ``` -## Sample output - daily signal - -`reports/sample/daily_signal.example.json` (committed sample, names/ticker -anonymized - live runs write real KAP names to git-ignored `reports/`): - -```json -{ - "as_of_date": "2026-05-28", - "clusters": [ - { - "ticker": "XXXXX", - "cluster_score": 42.83, - "insider_count": 2, - "window_start": "2026-05-21", - "window_end": "2026-05-21", - "unique_insiders": ["INSIDER A", "INSIDER B"], - "total_buy_value_try": 711360.0 - } - ], - "base_rates": { - "20": { - "benchmark_ticker": "XU100", - "signals_with_outcome": 0, - "verdict": "INSUFFICIENT_POWER", - "required_n_for_power": 784, - "hit_rate_pct": 0.0, - "hit_rate_ci_95": [0.0, 0.0], - "mean_abnormal_return_pct": 0.0, - "p_value": 1.0 - } - } -} +## Reproducing the result + +```bash +trailingedge prices backfill # XU100 benchmark +python scripts/load_official_prices.py # exchange bulletin: survivorship-clean +trailingedge signal detect # clusters + market-adjusted outcomes +python scripts/check_forward_returns.py # gross abnormal return, with its gates +python scripts/net_of_cost.py # the one that decides it +``` + +`net_of_cost.py` is the script that answers the question: + +``` +=== Abnormal return, NET of round-trip cost (order 25,000 TRY) === + spread: Abdi-Ranaldo (2017) from the stock's own OHLC, per trade + round-trip cost: median 1.94% p25 1.22% p75 4.24% + +HORIZON N GROSS AR% COST% NET AR% HIT% 95% CI t VERDICT + 5d 1032 0.57 3.34 -2.77 26.6 [23.9, 29.3] -13.11 LOSES MONEY (net) + 20d 1032 1.76 3.34 -1.58 41.4 [38.4, 44.4] -4.04 LOSES MONEY (net) + 60d 1032 2.09 3.34 -1.24 44.9 [41.9, 47.9] -1.98 LOSES MONEY (net) ``` -Returns are **market-adjusted** against XU100 over the position's own held interval, -and entry is t+1 after the disclosure is public. Every estimate carries a Wilson -interval, a t-test, and a `verdict` - and `INSUFFICIENT_POWER` is a gate, not a -footnote: below ~784 events the point estimates are not evidence in either direction, -so the report declines to offer one. See [`reports/sample/README.md`](reports/sample/README.md) -for what the previous (raw-return, N=29) version of this file claimed and why it was void. +The spread is not a parameter. It is estimated for each trade from the 30 sessions of +that stock's own OHLC before entry - which is also the only estimator that works on the +delisted names the bulletin carries and no quote feed does. A flat fee would have made +the answer come out the other way. ## Technical highlights From dacf3d26c2299d66b1d82cc7c07128e91b028c5a Mon Sep 17 00:00:00 2001 From: cagan Date: Mon, 13 Jul 2026 02:27:23 +0300 Subject: [PATCH 18/29] stage-0: freeze the routine/opportunistic definition before measuring it Cohen-Malloy-Pomorski's split is the signal's last plausible route to being tradeable: pooled, it loses by a hair (gross +1.76% at 20d against a 1.94% median round trip), and if half the trades are routine noise the opportunistic subset might clear it. Which makes this exactly the place where a definition chosen after seeing the result would be most tempting and least honest. The definition, the horizon, the test, the minimum N, the falsification condition and the prior are all fixed here first. The prior, recorded: opportunistic will be stronger gross and will still not clear the spread. SPK II-15.1's 250,000 TRY disclosure threshold already censors the small quiet trades CMP's routine class is built from, so there is less dilution left to strip out than in the US sample. If the result comes out positive it comes out against the prior, which is worth more than a confirmation. --- docs/stage0/OPPORTUNISTIC_CLASSIFIER.md | 87 +++++++++++++++++++++++++ 1 file changed, 87 insertions(+) create mode 100644 docs/stage0/OPPORTUNISTIC_CLASSIFIER.md diff --git a/docs/stage0/OPPORTUNISTIC_CLASSIFIER.md b/docs/stage0/OPPORTUNISTIC_CLASSIFIER.md new file mode 100644 index 0000000..8881012 --- /dev/null +++ b/docs/stage0/OPPORTUNISTIC_CLASSIFIER.md @@ -0,0 +1,87 @@ +# Stage-0: routine vs opportunistic insiders + +**Frozen 2026-07-13, before any result was computed.** + +This file exists so the definition cannot be chosen after seeing which one works. Every +parameter below is fixed here; the measurement code reads them and must not introduce +others. If a definition needs to change, it changes *here*, with a note saying why, and +the previous result is reported alongside the new one - never replaced by it. + +## Hypothesis + +Cohen, Malloy & Pomorski (2012), *Decoding Inside Information*, JF 67(3): more than half +of insider trades are **routine** - predictable, compensation- or liquidity-driven - and +carry **essentially zero** abnormal return. Stripping them leaves an **opportunistic** +subset worth ~82bp/month value-weighted. + +TrailingEdge currently measures every cluster the same way. Its gross abnormal return is ++1.76% at 20 days against a median round-trip cost of 1.94% - it loses by a hair. If the +CMP result carries to BIST, the opportunistic subset should be materially stronger, and +the question is whether it is stronger *enough to clear the spread*. + +This is the signal's last plausible route to being tradeable. It is therefore exactly the +place where a definition chosen after the fact would be most tempting, and least honest. + +## Definition (frozen) + +CMP classify an insider as routine if they traded **in the same calendar month for three +consecutive years**, and require three years of history to classify at all. + +That test cannot be ported as-is. Under SPK II-15.1 Art. 11 an insider owes no disclosure +until their transactions cross a cumulative **250,000 TRY** threshold within the calendar +year, so most Turkish insiders have sparse, gappy filing histories. Requiring three +consecutive years of filings would classify almost nobody and would select on filing +frequency - which correlates with position size, which correlates with the outcome. + +So the adaptation, fixed now: + +An insider is **ROUTINE** at time *t* if, over the 36 months before *t*, they have filed +purchases in **at least 3 distinct calendar months**, and **≥60%** of those purchase +months fall in the **same calendar month of the year** (e.g. every March). + +An insider is **OPPORTUNISTIC** at *t* if they have at least one prior purchase in the +36-month window and are not ROUTINE. + +An insider with **no prior purchase** in the window is **UNCLASSIFIED**. They are reported +separately and are NOT merged into either group: CMP's own result rests on a trading +history, and an insider without one is not evidence for or against it. + +A **cluster** is opportunistic if **at least one** of its constituent insiders is +opportunistic at that cluster's signal date. (A cluster is a co-purchase event; one +informed participant is enough to make it informative. The alternative - requiring all +insiders to be opportunistic - is a stricter test and is NOT run, to avoid a second +specification to choose between.) + +All classification uses only filings whose `published_at` is at or before the cluster's +signal date. No look-ahead. + +## Primary test (frozen) + +- **Metric**: mean abnormal return, net of the per-trade round-trip cost already used in + `scripts/net_of_cost.py` (Abdi-Ranaldo spread + Kyle impact + commission/BSMV). +- **Horizon**: **20 trading days**. Chosen because it is where the gross signal is + strongest in the pooled result and because CMP's effect is monthly. It is the single + pre-specified horizon; 5d and 60d are reported but are *secondary*. +- **Null**: net abnormal return of the opportunistic subset ≤ 0. +- **Decision**: the subset is declared tradeable only if its mean net abnormal return is + positive with a two-sided t-test at **p < 0.05**, on **N ≥ 200** opportunistic clusters. +- Below N = 200, the verdict is INSUFFICIENT_POWER and no claim is made either way. + +## What would falsify the hypothesis + +Opportunistic clusters showing no materially higher gross abnormal return than routine +ones. That would say the CMP split does not carry to BIST insider disclosures - which is +a publishable result and the expected one, given the 250,000 TRY threshold already +censors exactly the small, quiet, routine trades CMP's routine class is built from. + +## Pre-registered expectation + +Honest prior, recorded before running: **the opportunistic subset will be stronger gross +but will still not clear the spread.** The gap is 1.76% vs 1.94% pooled; a 2x improvement +in gross alpha would clear it, but CMP's own effect size (82bp/month) is not 2x a 1.76% +20-day return - it is comparable to it. And BIST's disclosure threshold has already +removed much of what CMP calls routine, so there is less dilution left to strip out than +in the US sample. + +Recording this matters: if the result comes out positive, it comes out *against* the +prior, and that is worth more than a confirmation. From f7f0f3e0e6a97ec4a9dd724bd51d8662b3f31466 Mon Sep 17 00:00:00 2001 From: cagan Date: Mon, 13 Jul 2026 02:29:09 +0300 Subject: [PATCH 19/29] signals: the routine/opportunistic split does not carry to BIST - there are no routine filers The pre-registered test (docs/stage0/OPPORTUNISTIC_CLASSIFIER.md, frozen before this ran). It was the signal's last plausible route to being tradeable: pooled, the gross abnormal return loses to the spread by a hair (+1.76% at 20d against a 1.94% median round trip), and Cohen-Malloy-Pomorski say half of insider trades are routine noise worth stripping. Result: cluster classes at 20d: OPPORTUNISTIC=1043 ROUTINE=0 UNCLASSIFIED=12 class hzn N gross% cost% net% t verdict OPPORTUNISTIC 20d* 1023 +1.80 3.35 -1.55 -3.94 loses money Not one cluster classified as routine. And that is exactly what the pre-registered prior predicted, in the frozen document, before the number existed: "SPK II-15.1's 250,000 TRY disclosure threshold already censors the small quiet trades CMP's routine class is built from, so there is less dilution left to strip out than in the US sample." Turkish insiders have no routine filings because routine trades stay under the reporting threshold and are never disclosed at all. There is no noise to strip: the opportunistic subset IS the sample (1,043 of 1,055), its gross return is +1.80% against the pooled +1.76%, and it loses to the same spread. CMP's hypothesis is not refuted - it is inapplicable. The regulation that makes this dataset possible is the same regulation that removes the variation the test needs. That is a finding about Turkish disclosure, not about insiders. Stated limitation: the 36-month lookback is thin on 3.5 years of data, so a fuller backfill would classify more insiders and might surface some routine ones. The direction is known - stripping routine trades RAISES gross alpha - but it would have to raise +1.76% past +1.94%, which is more than CMP's own effect size, and the disclosure threshold has already removed most of what would do the raising. Tests: 199 passing. --- scripts/opportunistic_test.py | 194 +++++++++++++++++++++ src/trailing_edge/signals/opportunistic.py | 88 ++++++++++ 2 files changed, 282 insertions(+) create mode 100644 scripts/opportunistic_test.py create mode 100644 src/trailing_edge/signals/opportunistic.py diff --git a/scripts/opportunistic_test.py b/scripts/opportunistic_test.py new file mode 100644 index 0000000..b95db1e --- /dev/null +++ b/scripts/opportunistic_test.py @@ -0,0 +1,194 @@ +"""Does the opportunistic subset clear the spread? + +The pre-registered test. Definition, horizon, minimum N and decision rule are frozen in +docs/stage0/OPPORTUNISTIC_CLASSIFIER.md, committed before this was run. + +Primary: mean abnormal return NET of the per-trade round-trip cost, at 20 trading days, +on opportunistic clusters. Tradeable only if positive with p < 0.05 on N >= 200. + +Usage: + uv run python scripts/opportunistic_test.py +""" +from __future__ import annotations + +import asyncio +import math +import statistics +import sys +from collections import Counter, defaultdict +from decimal import Decimal + +import click + +sys.path.insert(0, "src") + +from sqlalchemy import text # noqa: E402 + +from trailing_edge.core.db import get_session, init_db # noqa: E402 +from trailing_edge.core.logging import configure_logging # noqa: E402 +from trailing_edge.signals.costs import round_trip_cost # noqa: E402 +from trailing_edge.signals.opportunistic import ( # noqa: E402 + classify_cluster, + classify_insider, +) + +PRIMARY_HORIZON = 20 # frozen +MIN_N = 200 # frozen +ORDER_TRY = Decimal("25000") +_LOOKBACK = 30 + + +async def main_async() -> None: + await init_db() + + async with get_session() as session: + # Every insider purchase, with the date its disclosure became PUBLIC. The + # classifier may only see what was visible at the signal date. + hist_rows = ( + await session.execute( + text( + """ + SELECT t.insider_name, t.transaction_date, d.published_at::date AS pub + FROM kap_insider_transactions t + JOIN kap_disclosures d ON d.id = t.disclosure_id + WHERE t.transaction_type = 'BUY' AND d.published_at IS NOT NULL + """ + ) + ) + ).all() + + history: dict[str, list[tuple]] = defaultdict(list) + for name, tx_date, pub in hist_rows: + history[name].append((pub, tx_date)) + + rows = ( + await session.execute( + text( + """ + SELECT o.cluster_id, o.horizon_days, o.abnormal_return_pct, + o.entry_date, c.ticker, c.unique_insiders, c.window_end + FROM signal_outcomes o + JOIN insider_clusters c ON c.id = o.cluster_id + WHERE o.abnormal_return_pct IS NOT NULL AND o.entry_date IS NOT NULL + """ + ) + ) + ).all() + + cache: dict[tuple, tuple | None] = {} + + async def cost_for(ticker: str, entry) -> Decimal | None: + key = (ticker, entry) + if key not in cache: + px = ( + await session.execute( + text( + """ + SELECT close_try, high_try, low_try, volume + FROM price_history + WHERE ticker = :t AND price_date < :d + ORDER BY price_date DESC LIMIT :n + """ + ), + {"t": ticker, "d": entry, "n": _LOOKBACK}, + ) + ).all() + if len(px) < 22: + cache[key] = None + else: + px = list(reversed(px)) + closes = [r[0] for r in px] + highs = [r[1] or r[0] for r in px] + lows = [r[2] or r[0] for r in px] + adv = Decimal( + str( + statistics.fmean( + float(c) * float(r[3] or 0) + for c, r in zip(closes, px, strict=True) + ) + ) + ) + rt = round_trip_cost(closes, highs, lows, ORDER_TRY, adv) + cache[key] = (rt.total_pct,) if rt else None + got = cache[key] + return got[0] if got else None + + buckets: dict[tuple[str, int], list[tuple[float, float]]] = defaultdict(list) + cls_counts: Counter = Counter() + + for cid, horizon, ar, entry, ticker, insiders, window_end in rows: + # Classify each insider using ONLY filings public at the cluster's own date. + classes = [] + for name in insiders or []: + prior = [ + tx for pub, tx in history.get(name, []) + if pub <= window_end and tx < window_end + ] + classes.append(classify_insider(prior, window_end)) + klass = classify_cluster(classes) + + if horizon == PRIMARY_HORIZON: + cls_counts[klass.value] += 1 + + cost = await cost_for(ticker, entry) + if cost is None: + continue + buckets[(klass.value, horizon)].append((float(ar), float(cost))) + + click.echo("") + click.echo("=== Pre-registered test: does the opportunistic subset clear the spread? ===") + click.echo(" definition frozen in docs/stage0/OPPORTUNISTIC_CLASSIFIER.md") + click.echo(f" primary horizon {PRIMARY_HORIZON}d, min N {MIN_N}, order {ORDER_TRY:,.0f} TRY") + click.echo("") + click.echo(f" cluster classes at {PRIMARY_HORIZON}d: " + ", ".join( + f"{k}={v}" for k, v in sorted(cls_counts.items()) + )) + click.echo("") + + hdr = f"{'CLASS':>14} {'HZN':>4} {'N':>5} {'GROSS%':>8} {'COST%':>7} {'NET%':>7} {'t':>7} VERDICT" + click.echo(hdr) + click.echo("-" * len(hdr)) + + for klass in ("OPPORTUNISTIC", "ROUTINE", "UNCLASSIFIED"): + for horizon in (5, 20, 60): + pairs = buckets.get((klass, horizon), []) + if not pairs: + continue + net = [a - c for a, c in pairs] + n = len(net) + mean_net = statistics.fmean(net) + sd = statistics.stdev(net) if n > 1 else 0.0 + t = mean_net / (sd / math.sqrt(n)) if sd > 0 else 0.0 + + primary = horizon == PRIMARY_HORIZON + if not primary: + verdict = "(secondary)" + elif n < MIN_N: + verdict = f"INSUFFICIENT_POWER (n<{MIN_N})" + elif mean_net > 0 and abs(t) > 1.96: + verdict = "TRADEABLE" + elif abs(t) <= 1.96: + verdict = "NO EDGE (net)" + else: + verdict = "LOSES MONEY (net)" + + mark = " *" if primary else " " + click.echo( + f"{klass:>14} {horizon:>3}d{mark}{n:>4} " + f"{statistics.fmean([a for a, _ in pairs]):>8.2f} " + f"{statistics.fmean([c for _, c in pairs]):>7.2f} " + f"{mean_net:>7.2f} {t:>7.2f} {verdict}" + ) + click.echo("") + click.echo(" * = the pre-registered primary test. Everything else is secondary.") + click.echo("") + + +@click.command() +def main() -> None: + configure_logging() + asyncio.run(main_async()) + + +if __name__ == "__main__": + main() diff --git a/src/trailing_edge/signals/opportunistic.py b/src/trailing_edge/signals/opportunistic.py new file mode 100644 index 0000000..424b0e4 --- /dev/null +++ b/src/trailing_edge/signals/opportunistic.py @@ -0,0 +1,88 @@ +"""Routine vs opportunistic insiders (Cohen-Malloy-Pomorski, adapted to BIST). + +The definition is FROZEN in docs/stage0/OPPORTUNISTIC_CLASSIFIER.md and was committed +before any result was computed. This module implements it and introduces no parameter of +its own. If the definition must change, it changes there, with a reason, and the old +result is reported alongside the new one - not replaced by it. + +Why the US test cannot be ported as-is: CMP call an insider routine if they traded in the +same calendar month for three consecutive years, and refuse to classify anyone with less +than three years of history. Under SPK II-15.1 Art. 11 a Turkish insider owes no +disclosure until their transactions cross 250,000 TRY within the calendar year, so filing +histories are sparse and gappy. Demanding three consecutive years would classify almost +nobody - and would select on filing frequency, which tracks position size, which tracks +the outcome. +""" +from __future__ import annotations + +from collections import Counter +from datetime import date +from enum import Enum + +# --- frozen parameters (docs/stage0/OPPORTUNISTIC_CLASSIFIER.md) --- +LOOKBACK_MONTHS = 36 +MIN_PURCHASE_MONTHS = 3 +SEASONAL_SHARE = 0.60 + + +class InsiderClass(str, Enum): + ROUTINE = "ROUTINE" + OPPORTUNISTIC = "OPPORTUNISTIC" + UNCLASSIFIED = "UNCLASSIFIED" + + +def _months_before(d: date, months: int) -> date: + y, m = divmod(d.year * 12 + d.month - 1 - months, 12) + return date(y, m + 1, 1) + + +def classify_insider( + prior_purchase_dates: list[date], + as_of: date, +) -> InsiderClass: + """Classify one insider at one point in time. + + ``prior_purchase_dates`` must already be filtered to filings whose disclosure was + PUBLIC at or before ``as_of`` - this function cannot check that, and passing + transaction dates instead of disclosure-visible ones would reintroduce look-ahead + through the back door. + + ROUTINE: filed purchases in >= 3 distinct calendar months over the trailing 36, and + >= 60% of those months land on the same month of the year (an insider who buys every + March is on a schedule, not on information). + + OPPORTUNISTIC: has at least one prior purchase in the window and is not routine. + + UNCLASSIFIED: no prior purchase in the window. Reported separately, never merged into + either group - CMP's result rests on a trading history, and an insider without one is + evidence for neither side. + """ + cutoff = _months_before(as_of, LOOKBACK_MONTHS) + window = [d for d in prior_purchase_dates if cutoff <= d <= as_of] + if not window: + return InsiderClass.UNCLASSIFIED + + purchase_months = {(d.year, d.month) for d in window} + if len(purchase_months) < MIN_PURCHASE_MONTHS: + return InsiderClass.OPPORTUNISTIC + + seasonal = Counter(m for _, m in purchase_months) + top = seasonal.most_common(1)[0][1] + if top / len(purchase_months) >= SEASONAL_SHARE: + return InsiderClass.ROUTINE + return InsiderClass.OPPORTUNISTIC + + +def classify_cluster(insider_classes: list[InsiderClass]) -> InsiderClass: + """A cluster is opportunistic if ANY of its insiders is. + + A cluster is a co-purchase event; one informed participant is enough to make it + informative. Requiring every insider to be opportunistic is a stricter, different + test - it is deliberately NOT run, so that there is no second specification to pick + between after the fact. + """ + if any(c is InsiderClass.OPPORTUNISTIC for c in insider_classes): + return InsiderClass.OPPORTUNISTIC + if any(c is InsiderClass.ROUTINE for c in insider_classes): + return InsiderClass.ROUTINE + return InsiderClass.UNCLASSIFIED From bb62026c3d2f3c4719d0a86182895b59d6f4503c Mon Sep 17 00:00:00 2001 From: cagan Date: Mon, 13 Jul 2026 02:37:38 +0300 Subject: [PATCH 20/29] docs: METHODOLOGY listed as open three gaps that are now closed A methodology document that names a limitation it has since measured is not merely stale, it is misleading in the direction that matters - it understates what the project knows. All three of the gaps it listed were closed during this work, and two of them closed against expectation. Transaction cost - CLOSED, and it is THE constraint. Per-trade Abdi-Ranaldo spread from the stock's own OHLC: median round trip 1.94%, p75 4.24%, against a gross abnormal return of +1.76% at 20 days. The signal does not survive being traded. This is the result, not a caveat on it. VBTS tradability - CLOSED, and NOT the constraint. 143,012 gross-settlement ticker-days across 451 names (11% of price rows), but only 11 of 1,055 cluster entries (1.0%) land on a restricted day. Named, measured, and it turned out not to matter. Routine/opportunistic - CLOSED, and the split does not exist here. Zero routine clusters. The 250,000 TRY disclosure threshold censors exactly the trades CMP's routine class is built from, so there is no noise to strip. The hypothesis is not refuted, it is inapplicable - a finding about Turkish disclosure, not about insiders. Also recorded: the trap where the cost script dropped 96% of the sample by treating a degenerate Abdi-Ranaldo window as "unpriceable". The survivors' gross return came out negative where the full sample's was positive - the exclusion selects on the very price behaviour the signal is about. That failure belongs in the methodology, not just the git history, because the next person to write a cost model will reach for the same shortcut. A new section 6 keeps what is genuinely still open: regime (the window is 2015-2018), the latent single-factor cluster score, and Kyle's uncalibrated lambda. --- README.md | 2 +- docs/METHODOLOGY.md | 129 +++++++++++++++++++++++++++++++++----------- 2 files changed, 99 insertions(+), 32 deletions(-) diff --git a/README.md b/README.md index e867b40..ab802f0 100644 --- a/README.md +++ b/README.md @@ -83,7 +83,7 @@ also carries the VBTS tradability flags - which, once measured, turned out to to > (20d: +1.76%, t = 5.12, N = 1,032, survivorship-clean) that does **not** survive a > per-trade cost estimate. The window is 2015-2018 - a single regime - so the result is > not yet regime-conditional, and that is stated rather than glossed. Remaining gaps are -> in [`docs/METHODOLOGY.md`](docs/METHODOLOGY.md#5-known-open-issues), not left for a +> in [`docs/METHODOLOGY.md`](docs/METHODOLOGY.md#6-still-open), not left for a > reader to discover. ## Türkçe özet diff --git a/docs/METHODOLOGY.md b/docs/METHODOLOGY.md index 5feeaf3..c834c8c 100644 --- a/docs/METHODOLOGY.md +++ b/docs/METHODOLOGY.md @@ -150,38 +150,105 @@ reason. --- -## 5. Known open issues - -These are real and unfixed. They are listed so a reader does not have to discover -them by reading the source. - -**Cluster scoring is close to single-factor.** `cluster_score` blends insider count -(0.50), role seniority (0.30) and recency (0.20). In historical mode recency is pinned -at 1.0, so 20% of the weight is a constant. Seniority is resolved by joining the KAP -board/executive roster (`signals/roles.py`) — but where the roster does not cover an -insider, seniority falls back to its 0.5 default. When coverage is 0 the score reduces -to a monotone function of `insider_count` alone. `detect_clusters` now logs -`role_map_empty` loudly in that case; it used to happen silently. - -**No routine/opportunistic split.** Cohen, Malloy & Pomorski (2012), *Decoding Inside -Information* (JF 67(3)) show that **over half** of insider trades are "routine" — -predictable, compensation- or liquidity-driven — with **essentially zero** abnormal -return, while the remaining "opportunistic" trades carry ~82bp/month. This pipeline -does not yet separate them, so it averages the informative trades against the -uninformative ones. Porting their classifier (an insider is *routine* if they traded in -the same calendar month for three consecutive years) requires per-insider histories the -250,000 TRY threshold makes sparse — a Türkiye-adapted definition has to be -pre-registered before it is measured, not fitted afterwards. - -**No tradability filter.** Insider clusters concentrate in illiquid names, and Borsa -İstanbul's VBTS applies escalating measures to exactly those: short-selling ban → -**gross settlement** → **single-price auction**, in 15-day steps. A stock under a -single-price measure cannot be entered at the close the way the backtest assumes. -Neither VBTS state nor a liquidity floor is currently applied to the universe, and no -spread or market-impact cost is deducted. **Any positive result from this pipeline is -therefore an upper bound, before frictions.** +## 5. What was open, and how it closed ---- +Each of these was listed here as a known gap while it was one. They are kept, with what +measuring them actually showed - a limitation that is named and then measured away is +worth as much as one that turns out to be fatal. What is worth nothing is leaving it +unmeasured and implied. + +### Transaction cost - CLOSED, and it is the binding constraint + +The gross abnormal return was always an upper bound, and this is what it was an upper +bound over. + +The spread is estimated **per trade** from the 30 sessions of that stock's own OHLC +before entry (Abdi & Ranaldo 2017, RFS 30(12)) - not assumed as a flat fee, which would +have flattered the answer, and not taken from a quote feed, which does not exist for the +delisted names the exchange bulletin carries. Impact is Kyle/Almgren square-root on the +same window; commission and BSMV are charged per side. + + round-trip cost: median 1.94% p25 1.22% p75 4.24% + + horizon N=1032 gross AR net AR t (net) + 5d +0.57% -2.77% -13.11 + 20d +1.76% -1.58% -4.04 + 60d +2.09% -1.24% -1.98 + +**The signal does not survive the cost of trading it.** Insider clusters fire in illiquid +small caps, and the spread on those names is wider than the alpha. This is the project's +result, not a caveat on it. + +One trap worth recording: the first version of the cost script *dropped* any trade whose +Abdi-Ranaldo window came back degenerate (gamma >= 0, so a zero spread). That discarded +96% of the sample - and the survivors' gross abnormal return came out **negative** where +the full sample's was positive. The exclusion selects on exactly the price behaviour the +signal is about. A quiet window is not a free trade: the estimate now widens the window +and is floored at one tick, and nothing is dropped for it. + +### VBTS tradability - CLOSED, and it is NOT the constraint + +Borsa İstanbul escalates measures on volatile names: short-selling ban → **gross +settlement** → single-price auction. A name under gross settlement cannot be round-tripped +the way a backtest assumes. Insider clusters fire in exactly the names this happens to, so +this looked like it might matter a great deal. + +The exchange bulletin carries the flags (`BRUT TAKAS`, `GECICI DURDURMA`), so they are now +loaded. Measured: 143,012 gross-settlement ticker-days across 451 names - 11% of all price +rows. But of 1,055 cluster entries, **only 11 (1.0%)** land on a restricted day, and none +on a suspended one. + +So the gap is closed and it was never the binding constraint. The cost is. + +### Routine vs opportunistic - CLOSED, and the split does not exist here + +Pre-registered in `docs/stage0/OPPORTUNISTIC_CLASSIFIER.md` and frozen before it was run, +because this was the signal's last plausible route to being tradeable and therefore +exactly where a definition chosen after the fact would be most tempting. + + cluster classes at 20d: OPPORTUNISTIC=1043 ROUTINE=0 UNCLASSIFIED=12 + +**Not one cluster classified as routine**, and the frozen document had said why in advance: +SPK II-15.1's 250,000 TRY reporting threshold already censors the small, quiet, scheduled +trades that CMP's routine class is built from. Turkish insiders have no routine *filings* +because routine trades are never disclosed at all. + +So there is no noise to strip. The opportunistic subset IS the sample (1,043 of 1,055), its +gross return is +1.80% against the pooled +1.76%, and it loses to the same spread. + +CMP's hypothesis is not refuted - it is **inapplicable**. The regulation that makes this +dataset possible is the same regulation that removes the variation the test needs. That is +a finding about Turkish disclosure, not about insiders. + +*Limitation, stated:* the 36-month lookback is thin on 3.5 years of data, so a fuller +backfill would classify more insiders and might surface some routine ones. The direction is +known - stripping routine trades RAISES gross alpha - but it would have to raise +1.76% +past +1.94%, which is more than CMP's own effect size, and the disclosure threshold has +already removed most of what would do the raising. + +## 6. Still open + +**Regime.** The window is 2015-2018. That spans the August 2018 currency crisis but not +the 2021-2023 negative-real-rate retail boom or the 2023+ normalisation. The result is +therefore **not regime-conditional**, and a signal that dies to the spread in one regime +could in principle survive in another where those names traded tighter. The honest position +is that this is untested, not that it is unaffected. + +Closing it needs the KAP backfill to reach 2026. That is a data-collection problem, not a +methodological one, and it does not touch the mechanism: the spread eating the alpha is a +microstructure fact about illiquid names, not a regime phenomenon. + +**Cluster scoring is close to single-factor.** `cluster_score` blends insider count (0.50), +role seniority (0.30) and recency (0.20). In historical mode recency is pinned at 1.0, so +20% of the weight is a constant, and seniority falls back to its 0.5 default wherever the +scraped board roster does not cover an insider. When coverage is zero the score reduces to +a monotone function of `insider_count` alone - `detect_clusters` now logs `role_map_empty` +loudly in that case, where it used to happen silently. The score is not used to gate any +result reported here, so this is a latent defect rather than an active one. + +**Kyle's lambda is uncalibrated** (1.0). At retail order size the impact term is small +enough that the error changes no conclusion; at institutional size it would, and the number +should not be trusted there. ## References From babf6bb47aa956a6e42eaff5f1ea3f3c23898c36 Mon Sep 17 00:00:00 2001 From: cagan Date: Mon, 13 Jul 2026 02:50:28 +0300 Subject: [PATCH 21/29] scrapers: a chunk gets a deadline it cannot exceed A backfill sat silent for THREE HOURS. The process was alive, the last event was a 1200s WAF backoff, and after waking from it the run emitted nothing at all - no chunk_start, no error, no exit. Every timeout in the stack should have prevented that. httpx gets KAP_TIMEOUT_S on every request, tenacity caps at five attempts with a 60s ceiling on the wait, and the worst legitimate path through a chunk is a few minutes. None of them fired. Rather than guess which await never returned, the chunk now runs under asyncio.wait_for with a 15-minute deadline. A chunk is bounded work - one list call, then two requests per filing at a few per second - so anything past that is not slow, it is stuck. On expiry the month is abandoned, left PARTIAL for a recovery pass, and the run moves on. A hang that costs one month is a nuisance. A hang that costs the night is a failure, and this one cost the night. --- scripts/backfill_kap_insider.py | 27 ++++++++++++++++++++++++++- 1 file changed, 26 insertions(+), 1 deletion(-) diff --git a/scripts/backfill_kap_insider.py b/scripts/backfill_kap_insider.py index de9b680..381be48 100644 --- a/scripts/backfill_kap_insider.py +++ b/scripts/backfill_kap_insider.py @@ -88,6 +88,17 @@ _RECOVERY_PASSES = 4 _RECOVERY_PAUSE_S = 1800 # 30 minutes +# Hard ceiling on one month. A chunk is bounded work - list, then two requests per +# filing at a few per second - so anything past this is not slow, it is stuck. +# +# Observed: a run sat silent for THREE HOURS after a 1200s WAF backoff, having emitted +# no event at all since waking. Every timeout in the stack should have prevented that - +# httpx is given KAP_TIMEOUT_S, tenacity caps at five attempts - and none did. Rather +# than guess at which await never returned, the chunk gets a deadline it cannot exceed: +# on expiry it is abandoned, left PARTIAL for a later pass, and the run moves on. +# A hang that costs one month is a nuisance; a hang that costs the night is a failure. +_CHUNK_DEADLINE_S = 900 # 15 minutes + def generate_monthly_chunks(start: date, end: date) -> list[tuple[date, date]]: chunks = [] @@ -156,7 +167,9 @@ async def _run_chunk(from_date: date, to_date: date) -> None: ) try: scraper = KapInsiderScraper(backfill=True) - result = await scraper.run(from_date, to_date) + result = await asyncio.wait_for( + scraper.run(from_date, to_date), timeout=_CHUNK_DEADLINE_S + ) _log.info( "chunk_done", from_date=from_date, @@ -166,6 +179,18 @@ async def _run_chunk(from_date: date, to_date: date) -> None: skipped=result.records_skipped, ) return + except TimeoutError as exc: + # Not slow - stuck. Abandon the month; the ledger keeps it PARTIAL and a + # recovery pass will come back for it. + last_exc = exc + _log.error( + "chunk_deadline_exceeded", + from_date=from_date, + to_date=to_date, + attempt=attempt + 1, + deadline_s=_CHUNK_DEADLINE_S, + ) + return except httpx.RemoteProtocolError as exc: # Warmup GET disconnected -> KAP's WAF has IP-throttled us. last_exc = exc From 11f36177802dbd11b2d618a03cf57badf9ab3dbc Mon Sep 17 00:00:00 2001 From: cagan Date: Mon, 13 Jul 2026 02:54:38 +0300 Subject: [PATCH 22/29] scrapers: the three-hour hang was a timezone, not a hang - correcting the record The previous commit added a chunk deadline and justified it with a run that "sat silent for THREE HOURS". It had not. structlog writes timestamps in UTC, the machine clock is UTC+3, and I compared the two directly. A healthy process looked frozen because the two numbers were not in the same units. structlog : 23:54:05 (UTC) date : 02:54:05 (TSS, UTC+3) The process was working - the frontier had moved from 2018-09-14 to 2018-11-15 while I was calling it dead - and I killed it on that diagnosis. This is precisely the class of fault this codebase spent the day finding: two quantities compared as if they shared a scale when they did not. The list cap, the date separator, the column indices, the decimal comma, the survivorship gap - every one of them was some version of the same mistake. Committing it while cataloguing it is worth recording rather than quietly amending away. The deadline itself stays. A chunk that cannot finish in fifteen minutes is not doing useful work, and a run that can lose a month to a hang beats one that can lose a night. But it now says in the source that it guards against a hang never observed, instead of implying a bug that was really a clock. --- scripts/backfill_kap_insider.py | 22 ++++++++++++++-------- 1 file changed, 14 insertions(+), 8 deletions(-) diff --git a/scripts/backfill_kap_insider.py b/scripts/backfill_kap_insider.py index 381be48..3b9212d 100644 --- a/scripts/backfill_kap_insider.py +++ b/scripts/backfill_kap_insider.py @@ -88,15 +88,21 @@ _RECOVERY_PASSES = 4 _RECOVERY_PAUSE_S = 1800 # 30 minutes -# Hard ceiling on one month. A chunk is bounded work - list, then two requests per -# filing at a few per second - so anything past this is not slow, it is stuck. +# Hard ceiling on one month. A chunk is bounded work - one list call, then two requests +# per filing at a few per second - so anything past this is not slow, it is stuck. # -# Observed: a run sat silent for THREE HOURS after a 1200s WAF backoff, having emitted -# no event at all since waking. Every timeout in the stack should have prevented that - -# httpx is given KAP_TIMEOUT_S, tenacity caps at five attempts - and none did. Rather -# than guess at which await never returned, the chunk gets a deadline it cannot exceed: -# on expiry it is abandoned, left PARTIAL for a later pass, and the run moves on. -# A hang that costs one month is a nuisance; a hang that costs the night is a failure. +# This is a precaution, not a fix for an observed hang. It was added on the strength of +# a run that appeared to sit silent for three hours; it had not. structlog timestamps in +# UTC and the machine clock is UTC+3, so a healthy process looked frozen because the two +# numbers being compared were not in the same units. The process was working, and it was +# killed on a bad diagnosis - the same class of error this codebase spent the day +# catching, made by the person catching them. +# +# The deadline stays because a chunk that cannot finish in fifteen minutes is not doing +# useful work either way, and a run that can lose a month to a hang is better than one +# that can lose a night. But it is here as a guard against a hang that has never been +# observed, and this comment says so rather than implying a bug that was really a +# timezone. _CHUNK_DEADLINE_S = 900 # 15 minutes From 15bdd015d6f07b8112b5d31b2e62a1878ea136dc Mon Sep 17 00:00:00 2001 From: cagan Date: Mon, 13 Jul 2026 03:02:58 +0300 Subject: [PATCH 23/29] tests: cover the two modules the headline result actually rests on costs.py and opportunistic.py produce the project's answer and had no unit tests. 199 tests, and none of them touched the code that decides whether the signal is tradeable. 25 added. The ones that matter: - A quiet window is FLOORED at one tick, never dropped. This is the trap that cost 96% of the sample: when Abdi-Ranaldo's gamma goes >= 0 the formula reports a spread of exactly zero, and treating that as "cannot estimate" excludes the trade - selecting on precisely the price behaviour the signal is about. The survivors' gross return came out negative where the full sample's was positive. Pinned so it cannot come back. - A bad price INSIDE the estimation window voids the estimate; one OUTSIDE it does not. Writing this pair caught my own test being wrong: I asserted that a zero anywhere in the input should void the spread, and it should not - the estimator reads only the trailing window, and voiding on a fault it never sees would drop the trade for nothing. Dropping trades on a price-data condition is exactly how the cost model went wrong the first time. - An illiquid name costs more to round-trip than the +1.76% the signal earns. The result, as an assertion. - The frozen classifier: three purchase months minimum, 60% seasonal share, repeated buys in one month count once, nothing after as_of is visible, one opportunistic insider makes the cluster opportunistic. The definition is pre-registered; these keep a later reading of the RESULT from quietly becoming a later reading of the DEFINITION. Tests: 224 passing. --- tests/unit/signals/test_costs.py | 155 +++++++++++++++++++++++ tests/unit/signals/test_opportunistic.py | 99 +++++++++++++++ 2 files changed, 254 insertions(+) create mode 100644 tests/unit/signals/test_costs.py create mode 100644 tests/unit/signals/test_opportunistic.py diff --git a/tests/unit/signals/test_costs.py b/tests/unit/signals/test_costs.py new file mode 100644 index 0000000..a6efef1 --- /dev/null +++ b/tests/unit/signals/test_costs.py @@ -0,0 +1,155 @@ +"""Round-trip cost model - the module that decides the project's answer. + +The headline result turns on this: a gross abnormal return of +1.76% at 20 days against a +median round-trip cost of 1.94%. If the cost were wrong in either direction the conclusion +would flip, so its failure modes are pinned here rather than trusted. +""" +from decimal import Decimal + +from trailing_edge.signals.costs import ( + COMMISSION_PCT, + abdi_ranaldo_spread_pct, + kyle_impact_pct, + round_trip_cost, + tick_floor_pct, +) + + +def _series(n: int, base: float = 10.0, wobble: float = 0.02) -> tuple[list, list, list]: + """A synthetic OHLC series whose closes sit off the high-low midpoint - the pattern + Abdi-Ranaldo reads a spread from.""" + closes, highs, lows = [], [], [] + for i in range(n): + mid = base * (1 + 0.001 * i) + # close alternates above/below the midpoint: bid-ask bounce + c = mid * (1 + wobble if i % 2 else 1 - wobble) + closes.append(Decimal(str(round(c, 4)))) + highs.append(Decimal(str(round(mid * 1.03, 4)))) + lows.append(Decimal(str(round(mid * 0.97, 4)))) + return closes, highs, lows + + +# --- tick floor ------------------------------------------------------------- + +def test_tick_floor_is_coarser_for_cheap_stocks(): + """BIST's grid is 0.01 TRY below 20 TRY. A 3 TRY small cap therefore cannot be + quoted tighter than ~33bp however quiet its prints look - which is the whole reason + a 'zero spread' estimate must not be believed.""" + assert tick_floor_pct(Decimal("3")) == Decimal("0.01") / Decimal("3") + assert tick_floor_pct(Decimal("100")) == Decimal("0.02") / Decimal("100") + assert tick_floor_pct(Decimal("3")) > tick_floor_pct(Decimal("100")) + + +def test_tick_floor_of_zero_price_is_zero_not_infinite(): + assert tick_floor_pct(Decimal("0")) == 0 + + +# --- spread ----------------------------------------------------------------- + +def test_spread_is_detected_from_bid_ask_bounce(): + closes, highs, lows = _series(40) + spread = abdi_ranaldo_spread_pct(closes, highs, lows) + assert spread is not None + assert spread > Decimal("0.005") # well above the tick floor at this price + + +def test_a_quiet_window_is_floored_at_one_tick_not_dropped(): + """The trap that cost 96% of the sample. + + When the window is quiet the estimator's gamma goes >= 0 and the formula reports a + spread of exactly zero. Treating that as 'cannot estimate' and excluding the trade + selects on precisely the price behaviour the signal is about: the survivors' gross + return came out NEGATIVE where the full sample's was positive. A quiet window is not + a free trade - it is floored, never dropped. + """ + n = 40 + closes = [Decimal("10")] * n # perfectly flat: gamma cannot be negative + highs = [Decimal("10")] * n + lows = [Decimal("10")] * n + + spread = abdi_ranaldo_spread_pct(closes, highs, lows) + + assert spread is not None, "a quiet window must not void the estimate" + assert spread == tick_floor_pct(Decimal("10")) + + +def test_spread_returns_none_only_when_there_is_no_history(): + assert abdi_ranaldo_spread_pct([], [], []) is None + + +def test_a_bad_price_inside_the_estimation_window_voids_the_estimate(): + closes, highs, lows = _series(40) + closes[-3] = Decimal("0") # inside the trailing 21-session window + assert abdi_ranaldo_spread_pct(closes, highs, lows) is None + + +def test_a_bad_price_outside_the_window_is_irrelevant(): + """The estimator reads only the trailing window, so a fault before it cannot corrupt + the estimate - and must not be allowed to void an otherwise sound one. Voiding here + would drop the trade, and dropping trades on a price-data condition is how the cost + model selected 96% of the sample away the first time.""" + closes, highs, lows = _series(40) + closes[2] = Decimal("0") # far outside the trailing 21 sessions + assert abdi_ranaldo_spread_pct(closes, highs, lows) is not None + + +# --- impact ----------------------------------------------------------------- + +def test_impact_scales_with_square_root_of_size(): + closes, _, _ = _series(40) + adv = Decimal("1000000") + small = kyle_impact_pct(closes, Decimal("10000"), adv) + large = kyle_impact_pct(closes, Decimal("40000"), adv) + # 4x the order -> 2x the impact + assert large == small * 2 or abs(large - small * 2) < Decimal("0.0001") + + +def test_impact_is_negligible_at_retail_size(): + """The one cost a small account genuinely escapes - and showing that it is small is + the point, because it means the spread is doing all the damage.""" + closes, _, _ = _series(40) + impact = kyle_impact_pct(closes, Decimal("25000"), Decimal("5000000")) + assert impact < Decimal("0.005") # under 50bp one-sided + + +def test_impact_is_zero_when_adv_is_unknown(): + closes, _, _ = _series(40) + assert kyle_impact_pct(closes, Decimal("25000"), Decimal("0")) == 0 + + +# --- round trip ------------------------------------------------------------- + +def test_round_trip_charges_the_spread_once_and_impact_twice(): + """Crossing the spread costs it in full on each side, so it enters at 1x. Impact is + paid separately on entry and exit, so it enters at 2x.""" + closes, highs, lows = _series(40) + rt = round_trip_cost(closes, highs, lows, Decimal("25000"), Decimal("1000000")) + assert rt is not None + + spread = abdi_ranaldo_spread_pct(closes, highs, lows) + impact = kyle_impact_pct(closes, Decimal("25000"), Decimal("1000000")) + expected = (spread + 2 * impact + 2 * COMMISSION_PCT * Decimal("1.05")) * 100 + + assert abs(rt.total_pct - expected) < Decimal("0.001") + + +def test_round_trip_is_none_only_without_price_history(): + assert round_trip_cost([], [], [], Decimal("25000"), Decimal("1000000")) is None + + +def test_an_illiquid_name_costs_more_than_the_alpha(): + """The result, in one test. A wide-spread small cap cannot be round-tripped for less + than the +1.76% gross abnormal return the signal produces at 20 days.""" + n = 40 + closes, highs, lows = [], [], [] + for i in range(n): + mid = 3.0 + c = mid * (1.04 if i % 2 else 0.96) # heavy bounce = wide spread + closes.append(Decimal(str(round(c, 4)))) + highs.append(Decimal("3.12")) + lows.append(Decimal("2.88")) + + rt = round_trip_cost(closes, highs, lows, Decimal("25000"), Decimal("200000")) + + assert rt is not None + assert rt.total_pct > Decimal("1.76"), "the spread must eat the alpha, as measured" diff --git a/tests/unit/signals/test_opportunistic.py b/tests/unit/signals/test_opportunistic.py new file mode 100644 index 0000000..9b65599 --- /dev/null +++ b/tests/unit/signals/test_opportunistic.py @@ -0,0 +1,99 @@ +"""Routine/opportunistic classifier - the pre-registered test's implementation. + +Definition frozen in docs/stage0/OPPORTUNISTIC_CLASSIFIER.md before any result was +computed. These tests pin it, so that a later reading of the result cannot quietly become +a later reading of the definition. +""" +from datetime import date + +from trailing_edge.signals.opportunistic import ( + InsiderClass, + classify_cluster, + classify_insider, +) + +AS_OF = date(2018, 6, 15) + + +def test_no_prior_purchase_is_unclassified_not_opportunistic(): + """CMP's result rests on a trading history. An insider without one is evidence for + neither side, and folding them into either group would be choosing an answer.""" + assert classify_insider([], AS_OF) is InsiderClass.UNCLASSIFIED + + +def test_a_purchase_older_than_the_lookback_does_not_count(): + stale = [date(2014, 3, 1)] # > 36 months before AS_OF + assert classify_insider(stale, AS_OF) is InsiderClass.UNCLASSIFIED + + +def test_the_same_month_every_year_is_routine(): + """The definition of routine: on a schedule, not on information.""" + every_march = [date(2016, 3, 10), date(2017, 3, 14), date(2018, 3, 9)] + assert classify_insider(every_march, AS_OF) is InsiderClass.ROUTINE + + +def test_scattered_months_are_opportunistic(): + scattered = [date(2016, 2, 3), date(2017, 7, 21), date(2018, 5, 4)] + assert classify_insider(scattered, AS_OF) is InsiderClass.OPPORTUNISTIC + + +def test_too_few_purchase_months_cannot_be_routine(): + """Routine needs >= 3 distinct purchase months. Two Marches is a coincidence, not a + schedule - and calling it routine would strip an informative trade.""" + two_marches = [date(2017, 3, 8), date(2018, 3, 12)] + assert classify_insider(two_marches, AS_OF) is InsiderClass.OPPORTUNISTIC + + +def test_seasonal_share_must_clear_60_percent(): + """3 of 5 purchase months in March is 60% - routine. 2 of 5 is not.""" + mostly_march = [ + date(2016, 3, 1), date(2017, 3, 1), date(2018, 3, 1), + date(2016, 8, 1), date(2017, 11, 1), + ] + assert classify_insider(mostly_march, AS_OF) is InsiderClass.ROUTINE + + barely_march = [ + date(2016, 3, 1), date(2017, 3, 1), + date(2016, 8, 1), date(2017, 11, 1), date(2018, 1, 1), + ] + assert classify_insider(barely_march, AS_OF) is InsiderClass.OPPORTUNISTIC + + +def test_repeated_purchases_in_one_month_count_once(): + """Buying four times in one March is one purchase month, not four - otherwise a + single busy month would masquerade as a schedule.""" + one_busy_march = [date(2018, 3, d) for d in (1, 8, 15, 22)] + assert classify_insider(one_busy_march, AS_OF) is InsiderClass.OPPORTUNISTIC + + +def test_a_future_purchase_is_ignored(): + """Nothing after as_of may be seen. The caller filters on published_at, but the + classifier must not lean on that.""" + with_future = [date(2016, 2, 1), date(2017, 7, 1), date(2019, 5, 1)] + assert classify_insider(with_future, AS_OF) is InsiderClass.OPPORTUNISTIC + + +# --- cluster-level ---------------------------------------------------------- + +def test_one_opportunistic_insider_makes_the_cluster_opportunistic(): + """A cluster is a co-purchase event; one informed participant is enough to make it + informative.""" + got = classify_cluster([InsiderClass.ROUTINE, InsiderClass.OPPORTUNISTIC]) + assert got is InsiderClass.OPPORTUNISTIC + + +def test_all_routine_makes_the_cluster_routine(): + got = classify_cluster([InsiderClass.ROUTINE, InsiderClass.ROUTINE]) + assert got is InsiderClass.ROUTINE + + +def test_unclassified_never_upgrades_a_cluster(): + assert classify_cluster([InsiderClass.UNCLASSIFIED]) is InsiderClass.UNCLASSIFIED + assert ( + classify_cluster([InsiderClass.UNCLASSIFIED, InsiderClass.ROUTINE]) + is InsiderClass.ROUTINE + ) + + +def test_empty_cluster_is_unclassified(): + assert classify_cluster([]) is InsiderClass.UNCLASSIFIED From 984442457e855b6c6c214db6c977d91efb62ca5f Mon Sep 17 00:00:00 2001 From: cagan Date: Mon, 13 Jul 2026 03:13:09 +0300 Subject: [PATCH 24/29] docs: split-sample stability - the conclusion holds in both halves The result rests on one window, so it is worth knowing whether one stretch of it carries the whole thing. Same cost test, each half separately: period N gross% cost% net% t verdict 2015-2016 859 +2.30 3.57 -1.27 -3.04 loses money 2017-2018 197 +0.30 2.17 -1.87 -1.57 inconclusive (N < 200) Net abnormal return is negative in both. The second half is inconclusive rather than confirming, and the distinction matters: N=197 is under the pre-registered minimum of 200, so the sign agrees and the power does not. Reporting it as a second confirmation would be claiming more than the data holds. The gross alpha collapses from +2.30% to +0.30% across the halves. That is either a market becoming more efficient or noise at N=197, and at that N it cannot be told apart - so it is recorded and not interpreted. This is explicitly NOT the regime test. The window is 2015-2018; the August 2018 crisis leaves only 9 clusters on its far side, and the 2021-2023 negative-real-rate boom is not in the data at all. Regime stays open, and stays in section 6 where it belongs. --- docs/METHODOLOGY.md | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) diff --git a/docs/METHODOLOGY.md b/docs/METHODOLOGY.md index c834c8c..9890399 100644 --- a/docs/METHODOLOGY.md +++ b/docs/METHODOLOGY.md @@ -226,6 +226,23 @@ known - stripping routine trades RAISES gross alpha - but it would have to raise past +1.94%, which is more than CMP's own effect size, and the disclosure threshold has already removed most of what would do the raising. +### Split-sample stability - the conclusion holds in both halves + +Not a regime test - see §6 - but a check that the result is not carried by one stretch of +the sample. The same cost test, run separately on each half of the available window: + + period N gross% cost% net% t verdict + 2015-2016 859 +2.30 3.57 -1.27 -3.04 loses money + 2017-2018 197 +0.30 2.17 -1.87 -1.57 inconclusive (N < 200) + +**Net abnormal return is negative in both halves.** The second is inconclusive rather than +confirming, but only because N = 197 falls under the pre-registered minimum of 200 - the +sign and the direction agree; the power does not. + +Worth recording without over-reading: the *gross* alpha collapses from +2.30% to +0.30% +between the halves. That could be the market becoming more efficient, or it could be +sampling noise at N = 197. It is not interpreted here, because at that N it cannot be. + ## 6. Still open **Regime.** The window is 2015-2018. That spans the August 2018 currency crisis but not From a8daf8a853932f18159ec9a1c4a82f13c468b798 Mon Sep 17 00:00:00 2001 From: cagan Date: Mon, 13 Jul 2026 03:18:13 +0300 Subject: [PATCH 25/29] cli: --help crashed on a Turkish console, and the README documented a command that does not exist Two faults found by actually running the thing, which is the only way they surface. 1. `trailingedge report --help` died with UnicodeEncodeError. A single U+2194 in a docstring - "Generate cross-reference brief (KAP <-> TSG)" - cannot be encoded in cp1254, the default console codepage on a Turkish Windows install. The first command anyone runs against a repository is --help. It was the one that broke. Help text is now ASCII, and stdout/stderr are reconfigured to UTF-8 with errors=replace where the runtime allows it, so the next non-ASCII character degrades to a replacement glyph instead of a traceback. 2. The README told readers to run `trailingedge report forensic KAPLM`. There is no such command. The forensic module exists and works - it produced a real PDF for SARKY on the first try - but the CLI exposes it as `report generate --ticker`. A README that documents a command that does not exist fails the reader at the exact moment they decide whether to trust the repository. Both are the same category as everything else found today: something asserted, never executed, and wrong. The fix for that class is not more care, it is running the command. --- README.md | 6 +++--- src/trailing_edge/cli/main.py | 18 +++++++++++++++++- 2 files changed, 20 insertions(+), 4 deletions(-) diff --git a/README.md b/README.md index ab802f0..051b1a6 100644 --- a/README.md +++ b/README.md @@ -136,16 +136,16 @@ trailingedge scrape kap-insider --last-hours 168 # last week trailingedge scrape kap-insider --since 2026-05-01 --until 2026-05-27 ``` -Forensic brief for a single ticker: +Insider-activity brief for a single ticker (HTML + PDF): ```bash -trailingedge report forensic KAPLM +trailingedge report generate --ticker SARKY ``` ## Reproducing the result ```bash -trailingedge prices backfill # XU100 benchmark +trailingedge prices backfill # XU100 benchmark (yfinance) python scripts/load_official_prices.py # exchange bulletin: survivorship-clean trailingedge signal detect # clusters + market-adjusted outcomes python scripts/check_forward_returns.py # gross abnormal return, with its gates diff --git a/src/trailing_edge/cli/main.py b/src/trailing_edge/cli/main.py index f73b917..2e1347a 100644 --- a/src/trailing_edge/cli/main.py +++ b/src/trailing_edge/cli/main.py @@ -1,5 +1,6 @@ """CLI entrypoint for trailingedge.""" import asyncio +import sys from datetime import date, timedelta import click @@ -12,6 +13,21 @@ _log = get_logger(__name__) +# The console this runs on is not guaranteed to be UTF-8. On Windows it is typically +# cp1254 (Turkish), which cannot encode U+2194 - and a single such character in a +# docstring took down `trailingedge report --help` with a UnicodeEncodeError. The first +# command anyone runs against a repository is --help; it must not be the one that breaks. +# +# Help text is kept ASCII, and stdout/stderr are reconfigured where the runtime allows it, +# so a future non-ASCII character degrades to a replacement glyph instead of a traceback. +for _stream in (sys.stdout, sys.stderr): + if hasattr(_stream, "reconfigure"): + try: + _stream.reconfigure(encoding="utf-8", errors="replace") + except (ValueError, OSError): # pragma: no cover - console-dependent + pass + + @click.group() def cli() -> None: """TrailingEdge - BIST insider-transaction disclosure analytics.""" @@ -416,7 +432,7 @@ async def _run_report_generate(ticker: str | None, output: str, all_signals: boo help="Output format", ) def report_cross_reference(top: int, person: str | None, output: str) -> None: - """Generate cross-reference brief (KAP ↔ TSG).""" + """Generate cross-reference brief (KAP <-> TSG).""" configure_logging() asyncio.run(_run_cross_reference(top, person, output)) From 234d18a3c5306ba4b502201dbc94b89df21003b5 Mon Sep 17 00:00:00 2001 From: cagan Date: Mon, 13 Jul 2026 04:01:30 +0300 Subject: [PATCH 26/29] fix: the cost model was reading a return index as if it were a price price_history.close_try is a chained, corporate-action-adjusted total-return index. That is the correct series for returns - a bonus issue halves the print and a raw series reads it as a 50% loss - and every return in this project is computed from it. It is not a price. Two costs are properties of the price, not the index: - the tick floor is 0.01 TRY on the exchange's grid, so it depends on where the stock actually trades. tick_floor_pct was being handed the index. - ADV is price x volume, and Kyle impact scales as sqrt(1/ADV). The same index went into ADV. Measured on 2018-12 bulletin data: the index sits at a median 0.98x the traded price, but ranges 0.60x to 118x, and 32% of ticker-days are off by more than 10%. BIST companies issue bonus shares constantly, so the two series pull apart, worst in the serial issuers - which are small caps - which are exactly the names where the tick floor binds and where the tradeability verdict is actually decided. Both errors land in signals/costs.py, the module that decides the project's answer. The factor falls on BOTH sides of 1, so this is not a conservative error that happens to protect the conclusion: it mispriced individual trades in both directions. - migration 0008 keeps raw_close_try alongside the index. Nullable and NOT backfilled: NULL means the traded price was never stored for that row, and the cost model must refuse it rather than fall back to the index - falling back is the bug. - abdi_ranaldo_spread_pct / round_trip_cost take last_traded_price. The estimator itself is scale-free (it reads log close minus the log high-low midpoint, so a constant factor cancels) and correctly stays on the adjusted index; only the floor moves to the price. - net_of_cost.py and opportunistic_test.py build ADV from raw price x volume. Prices reloaded from 2014-11 rather than 2015-01. The chain's ratios are unchanged by an earlier anchor, so stored abnormal returns are unaffected, but the earliest KAP clusters (first entry 2015-01-16) previously had fewer than the 22 sessions the spread estimator needs and were dropped. That exclusion was not random: the 23 dropped clusters averaged +8.45% at 20 days against +1.93% for the rest. Also: KAP's relatedStocks is not always one ticker. A filing naming several stocks became a "ticker" like `KRDMA, KRDMB, KRDMD` - a key that joins to no price row, so those clusters silently left every result (12 distinct strings, ~131 disclosures, 14 clusters). split_related_tickers now splits it and refuses to choose: which class the insider bought is in the filing body, not this field, and a plausible wrong ticker is worse than a visible gap. And a correction to an earlier claim of mine: this repo was NOT ruff- and mypy-clean. ruff is now clean; 12 mypy errors remain, all predating this work. --- migrations/env.py | 4 +- migrations/versions/0008_raw_close.py | 51 ++++++++++++++++++++ scripts/load_official_prices.py | 6 +++ scripts/net_of_cost.py | 22 +++++++-- scripts/opportunistic_test.py | 15 ++++-- scripts/tsg_captcha_recon.py | 2 - src/trailing_edge/models/signal.py | 8 ++++ src/trailing_edge/scrapers/kap/parser.py | 53 +++++++++++++++++++-- src/trailing_edge/signals/costs.py | 18 ++++++- tests/unit/scrapers/test_related_tickers.py | 47 ++++++++++++++++++ tests/unit/signals/test_costs.py | 23 +++++++++ 11 files changed, 229 insertions(+), 20 deletions(-) create mode 100644 migrations/versions/0008_raw_close.py create mode 100644 tests/unit/scrapers/test_related_tickers.py diff --git a/migrations/env.py b/migrations/env.py index 5793695..e2098ab 100644 --- a/migrations/env.py +++ b/migrations/env.py @@ -7,11 +7,11 @@ from sqlalchemy.engine import Connection from sqlalchemy.ext.asyncio import async_engine_from_config -from trailing_edge.models.base import Base +import trailing_edge.models.graph # noqa: F401 - register graph tables with metadata import trailing_edge.models.kap # noqa: F401 - register KAP tables with metadata import trailing_edge.models.signal # noqa: F401 - register signal tables with metadata -import trailing_edge.models.graph # noqa: F401 - register graph tables with metadata import trailing_edge.models.unlisted # noqa: F401 - register unlisted tables with metadata +from trailing_edge.models.base import Base config = context.config diff --git a/migrations/versions/0008_raw_close.py b/migrations/versions/0008_raw_close.py new file mode 100644 index 0000000..be149ac --- /dev/null +++ b/migrations/versions/0008_raw_close.py @@ -0,0 +1,51 @@ +"""Keep the traded close alongside the total-return index. + +`price_history.close_try` holds a chained, corporate-action-adjusted index, not the +price the stock actually printed. That is the right series for returns - a bonus issue +halves the print and a raw series reads it as a 50% loss - and it is what every return +in this project is computed from. + +But two costs are properties of the PRICE, not of the index: + + - the tick floor is 0.01 TRY on a grid, so it depends on where the stock actually + trades. BIST companies issue bonus shares constantly; on 2018-12 data the index sat + at a median 0.98x the traded price but ranged from 0.60x to 118x, and 32% of + ticker-days were off by more than 10%. Feeding the index to `tick_floor_pct` divided + the tick cost by that factor. + - ADV in TRY is price x volume. The same factor inflated it, and Kyle impact scales + with sqrt(1/ADV), so impact came out understated too. + +Both errors land in the cost model, which is the module that decides whether the signal +is tradeable. The factor is not one-directional, so this is not a conservative error - it +mispriced individual trades in both directions, worst in the serial bonus-issuers, which +are small caps, which are exactly where the tick floor binds. + +So the raw close is now kept. The index stays the default for returns; the cost model +reads the price. + +Revision ID: 0008 +Revises: 0007 +""" +from __future__ import annotations + +import sqlalchemy as sa +from alembic import op + +revision = "0008" +down_revision = "0007" +branch_labels = None +depends_on = None + + +def upgrade() -> None: + # Nullable with no backfill: a NULL here means "this row predates the fix and its + # traded price was never stored". The cost model must see that and refuse, rather + # than silently fall back to the index - falling back is the bug being fixed. + op.add_column( + "price_history", + sa.Column("raw_close_try", sa.Numeric(20, 4), nullable=True), + ) + + +def downgrade() -> None: + op.drop_column("price_history", "raw_close_try") diff --git a/scripts/load_official_prices.py b/scripts/load_official_prices.py index 3158cca..93b5434 100644 --- a/scripts/load_official_prices.py +++ b/scripts/load_official_prices.py @@ -345,6 +345,11 @@ def _scale(v: Decimal | None) -> Decimal | None: return (v * factor).quantize(Decimal("0.0001")) if v is not None else None prev_raw_close = raw_close + # The price the stock actually printed, before the chain replaces it with the + # index. The tick floor is 0.01 TRY on the exchange's grid and ADV is + # price x volume - both are properties of this number, not of the index, and + # BIST's serial bonus issues drive the two apart (median 0.98x, up to 118x). + r["raw_close_try"] = raw_close.quantize(Decimal("0.0001")) r["close_try"] = level.quantize(Decimal("0.0001")) r["open_try"] = _scale(r["open_try"]) r["high_try"] = _scale(r["high_try"]) @@ -384,6 +389,7 @@ async def _store(rows: list[dict]) -> int: constraint="uq_price_ticker_date", set_={ "close_try": stmt.excluded.close_try, + "raw_close_try": stmt.excluded.raw_close_try, "open_try": stmt.excluded.open_try, "high_try": stmt.excluded.high_try, "low_try": stmt.excluded.low_try, diff --git a/scripts/net_of_cost.py b/scripts/net_of_cost.py index fded6f4..f8abb22 100644 --- a/scripts/net_of_cost.py +++ b/scripts/net_of_cost.py @@ -65,7 +65,7 @@ async def window(ticker: str, entry) -> tuple | None: await session.execute( text( """ - SELECT close_try, high_try, low_try, volume + SELECT close_try, high_try, low_try, volume, raw_close_try FROM price_history WHERE ticker = :t AND price_date < :d ORDER BY price_date DESC LIMIT :n @@ -82,10 +82,20 @@ async def window(ticker: str, entry) -> tuple | None: highs = [r[1] or r[0] for r in px] lows = [r[2] or r[0] for r in px] vols = [r[3] or 0 for r in px] + + # close_try is the corporate-action-adjusted INDEX, which is what the spread + # estimator wants (it is scale-free and reads the adjusted series). The tick + # floor and ADV are properties of the traded PRICE, and BIST bonus issues push + # the two apart - median 0.98x but up to 118x. Using the index here understated + # the tick floor and inflated ADV, and both errors land in the cost model. + raws = [r[4] for r in px] + if any(r is None for r in raws): + cache[key] = None # pre-0008 row: refuse rather than fall back to the index + return None adv = Decimal( - str(statistics.fmean(float(c) * float(v) for c, v in zip(closes, vols, strict=True))) + str(statistics.fmean(float(c) * float(v) for c, v in zip(raws, vols, strict=True))) ) - cache[key] = (closes, highs, lows, adv) + cache[key] = (closes, highs, lows, adv, raws[-1]) return cache[key] by_h: dict[int, list[tuple[float, float]]] = {} @@ -97,8 +107,10 @@ async def window(ticker: str, entry) -> tuple | None: if w is None: dropped += 1 continue - closes, highs, lows, adv = w - rt = round_trip_cost(closes, highs, lows, order_try, adv) + closes, highs, lows, adv, last_price = w + rt = round_trip_cost( + closes, highs, lows, order_try, adv, last_traded_price=last_price + ) if rt is None: dropped += 1 continue diff --git a/scripts/opportunistic_test.py b/scripts/opportunistic_test.py index b95db1e..da9a43d 100644 --- a/scripts/opportunistic_test.py +++ b/scripts/opportunistic_test.py @@ -84,7 +84,7 @@ async def cost_for(ticker: str, entry) -> Decimal | None: await session.execute( text( """ - SELECT close_try, high_try, low_try, volume + SELECT close_try, high_try, low_try, volume, raw_close_try FROM price_history WHERE ticker = :t AND price_date < :d ORDER BY price_date DESC LIMIT :n @@ -93,22 +93,27 @@ async def cost_for(ticker: str, entry) -> Decimal | None: {"t": ticker, "d": entry, "n": _LOOKBACK}, ) ).all() - if len(px) < 22: + if len(px) < 22 or any(r[4] is None for r in px): cache[key] = None else: px = list(reversed(px)) closes = [r[0] for r in px] highs = [r[1] or r[0] for r in px] lows = [r[2] or r[0] for r in px] + # ADV and the tick floor take the traded price, not the adjusted index + # - see scripts/net_of_cost.py and migration 0008. + raws = [r[4] for r in px] adv = Decimal( str( statistics.fmean( - float(c) * float(r[3] or 0) - for c, r in zip(closes, px, strict=True) + float(rc) * float(r[3] or 0) + for rc, r in zip(raws, px, strict=True) ) ) ) - rt = round_trip_cost(closes, highs, lows, ORDER_TRY, adv) + rt = round_trip_cost( + closes, highs, lows, ORDER_TRY, adv, last_traded_price=raws[-1] + ) cache[key] = (rt.total_pct,) if rt else None got = cache[key] return got[0] if got else None diff --git a/scripts/tsg_captcha_recon.py b/scripts/tsg_captcha_recon.py index 79f392a..f042d79 100644 --- a/scripts/tsg_captcha_recon.py +++ b/scripts/tsg_captcha_recon.py @@ -98,8 +98,6 @@ async def main() -> None: print("POPUP'TA CAPTCHA CIKACAK - goruntuleyin ama COZMEYIN (sadece inspect)") print("5 saniye bekleniyor popup icin...\n") - popup_holder = [] - async with page.expect_popup(timeout=30_000) as popup_info: await pdf_links[0].click() diff --git a/src/trailing_edge/models/signal.py b/src/trailing_edge/models/signal.py index 7b3a709..070c784 100644 --- a/src/trailing_edge/models/signal.py +++ b/src/trailing_edge/models/signal.py @@ -34,6 +34,14 @@ class PriceHistory(Base): close_try: Mapped[Decimal] = mapped_column(Numeric(20, 4), nullable=False) volume: Mapped[int | None] = mapped_column(BigInteger) + # close_try is a chained total-return index, which is what returns must be computed + # from. It is NOT the price the stock printed: BIST bonus issues push the two apart, + # by a median 0.98x but up to 118x. The tick floor (0.01 TRY on a grid) and ADV + # (price x volume) are properties of the traded price, so the cost model reads this + # column instead. NULL means the row predates migration 0008 and the price was never + # kept - the cost model must refuse it, not fall back to the index. + raw_close_try: Mapped[Decimal | None] = mapped_column(Numeric(20, 4)) + # VBTS tradability, straight from the exchange bulletin. A name under gross # settlement cannot be round-tripped the way a backtest assumes, and a suspended # one cannot be traded at all - the signal fires precisely in the names this diff --git a/src/trailing_edge/scrapers/kap/parser.py b/src/trailing_edge/scrapers/kap/parser.py index a7385eb..9f46277 100644 --- a/src/trailing_edge/scrapers/kap/parser.py +++ b/src/trailing_edge/scrapers/kap/parser.py @@ -12,6 +12,42 @@ _log = get_logger(__name__) + +def split_related_tickers(related: str) -> list[str]: + """KAP's `relatedStocks` is not always one ticker. + + The field was read as a plain string and stored whole, so a filing that names several + stocks produced a "ticker" like `KRDMA, KRDMB, KRDMD` - a key that joins to no price + row anywhere. Those clusters did not error; they silently dropped out of every result. + On the 2015-2019 sample this hit 12 distinct strings across ~131 disclosures and 14 + clusters. + + Two different things get comma-joined here: + - share classes of ONE issuer (KRDMA/KRDMB/KRDMD are Kardemir; ISATR/ISBTR/ISCTR/ + ISKUR are Is Bankasi), and + - genuinely different issuers, when the filer is tied to more than one (ANELT, VERTU). + + Which class the insider actually bought is in the filing body, not in this field. So + this function only SPLITS - it does not choose. Attribution is deliberately not + guessed: picking the first token, or the most liquid class, would put a real trade on + a stock that may not have been traded, and a plausible wrong ticker is worse than a + visible gap. Callers that get more than one ticker back must treat the disclosure as + unattributable and count it, not quietly drop it. + """ + if not related: + return [] + return [t for part in related.split(",") if (t := part.strip().upper())] + + +def normalize_related_tickers(related: str) -> str: + """The multi-ticker string in a canonical, whitespace-free form (`A,B,C`). + + Kept as the disclosure's ticker when attribution is impossible, so the ambiguity stays + visible in the data instead of being laundered into a single plausible ticker. + """ + return ",".join(split_related_tickers(related)) + + # Matches Turkish formatted numbers: 1.234.567,89 or 1.234 or 18,45 or -2.500.000 _TR_NUM_RE = re.compile(r"-?[\d]{1,3}(?:\.[\d]{3})*(?:,[\d]+)?") @@ -694,12 +730,21 @@ def parse_disclosure_metadata( basic = disc_wrap.get("disclosureBasic", disc_wrap) disclosure_index = str(basic.get("disclosureIndex", "")) - # relatedStocks is a plain string ticker in the real API (not a list of dicts) related = basic.get("relatedStocks", "") - ticker = (related.strip().upper() if isinstance(related, str) else "") - if not ticker and list_item: + raw_related = related if isinstance(related, str) else "" + if not raw_related.strip() and list_item: related_li = list_item.get("relatedStocks", "") - ticker = related_li.strip().upper() if isinstance(related_li, str) else "" + raw_related = related_li if isinstance(related_li, str) else "" + + tickers = split_related_tickers(raw_related) + ticker = tickers[0] if len(tickers) == 1 else normalize_related_tickers(raw_related) + if len(tickers) > 1: + # Not attributable, and not silently guessable. See split_related_tickers. + _log.warning( + "multi_ticker_disclosure", + kap_disclosure_id=disclosure_index, + related_tickers=tickers, + ) company = ( basic.get("companyTitle", "") diff --git a/src/trailing_edge/signals/costs.py b/src/trailing_edge/signals/costs.py index 9600280..290a437 100644 --- a/src/trailing_edge/signals/costs.py +++ b/src/trailing_edge/signals/costs.py @@ -104,6 +104,7 @@ def abdi_ranaldo_spread_pct( highs: list[Decimal], lows: list[Decimal], window: int = _SPREAD_WINDOW, + last_traded_price: Decimal | None = None, ) -> Decimal | None: """Effective bid-ask spread as a fraction, from OHLC alone (Abdi-Ranaldo 2017). @@ -118,6 +119,12 @@ def abdi_ranaldo_spread_pct( window rather than voiding the estimate, and whatever survives is floored at one tick - the tightest quote the exchange's own price grid allows. None is returned only when there is genuinely not enough price history to look at. + + The estimator itself is scale-free: it reads log(close) - log(high-low midpoint), so + a constant factor cancels, and it is CORRECT to run it on the adjusted index. The + tick floor is not scale-free - it is 0.01 TRY on the exchange's grid, so it needs the + price the stock actually printed. Pass it as ``last_traded_price``. Omitting it falls + back to the last close, which is right only when that close is a raw price. """ for w in (window, 60, 120): g = _gamma(closes, highs, lows, w) @@ -135,7 +142,8 @@ def abdi_ranaldo_spread_pct( if _gamma(closes, highs, lows, min(window, max(len(closes) - 1, 2))) is None: return None - floor = tick_floor_pct(closes[-1]) + price = last_traded_price if last_traded_price is not None else closes[-1] + floor = tick_floor_pct(price) return max(est, floor).quantize(Decimal("0.000001")) @@ -179,13 +187,19 @@ def round_trip_cost( lows: list[Decimal], order_value_try: Decimal, adv_try: Decimal, + last_traded_price: Decimal | None = None, ) -> RoundTrip | None: """Total cost of entering and exiting one position, as a percent of notional. None when the spread cannot be estimated: a trade whose cost is unknown must be excluded from the sample, not priced at zero. + + ``last_traded_price`` is the raw printed close, used for the tick floor. ``adv_try`` + must likewise be built from raw price x volume: feeding it the adjusted index inflates + ADV by the corporate-action factor and understates impact, which scales as + sqrt(1/ADV). """ - spread = abdi_ranaldo_spread_pct(closes, highs, lows) + spread = abdi_ranaldo_spread_pct(closes, highs, lows, last_traded_price=last_traded_price) if spread is None: return None diff --git a/tests/unit/scrapers/test_related_tickers.py b/tests/unit/scrapers/test_related_tickers.py new file mode 100644 index 0000000..82eef5c --- /dev/null +++ b/tests/unit/scrapers/test_related_tickers.py @@ -0,0 +1,47 @@ +"""KAP's relatedStocks field, which is not always a single ticker. + +The field was consumed as a plain string, so a filing naming several stocks became a +"ticker" like `KRDMA, KRDMB, KRDMD` - a key that joins to no price row. The clusters did +not error, they just vanished from every result: 14 of them, on 12 distinct malformed +strings across ~131 disclosures. +""" +from trailing_edge.scrapers.kap.parser import ( + normalize_related_tickers, + split_related_tickers, +) + + +def test_a_single_ticker_is_unchanged(): + assert split_related_tickers("THYAO") == ["THYAO"] + assert split_related_tickers(" thyao ") == ["THYAO"] + + +def test_share_classes_of_one_issuer_are_split(): + """Kardemir lists three classes; the filing names all three.""" + assert split_related_tickers("KRDMA, KRDMB, KRDMD") == ["KRDMA", "KRDMB", "KRDMD"] + + +def test_two_different_issuers_are_split(): + """Not every multi-ticker filing is one company's share classes - a filer tied to two + issuers produces two unrelated tickers, which is why the class cannot be guessed.""" + assert split_related_tickers("ANELT, VERTU") == ["ANELT", "VERTU"] + + +def test_empty_and_blank_yield_nothing(): + assert split_related_tickers("") == [] + assert split_related_tickers(" , ") == [] + + +def test_normalized_form_is_whitespace_free_and_order_preserving(): + """The ambiguity stays in the data rather than being laundered into one plausible + ticker. Canonical form so the same filing always keys the same way.""" + assert normalize_related_tickers("KRDMA, KRDMB, KRDMD") == "KRDMA,KRDMB,KRDMD" + assert normalize_related_tickers("krdma ,krdmb") == "KRDMA,KRDMB" + + +def test_splitting_never_silently_picks_one(): + """The whole point. A function that returned 'KRDMA' here would put a real insider + purchase on a stock the insider may never have touched - and it would join to a price, + so nothing downstream would ever notice.""" + got = split_related_tickers("KRDMA, KRDMB, KRDMD") + assert len(got) == 3, "attribution must be left to the caller, not guessed here" diff --git a/tests/unit/signals/test_costs.py b/tests/unit/signals/test_costs.py index a6efef1..b740539 100644 --- a/tests/unit/signals/test_costs.py +++ b/tests/unit/signals/test_costs.py @@ -77,6 +77,29 @@ def test_spread_returns_none_only_when_there_is_no_history(): assert abdi_ranaldo_spread_pct([], [], []) is None +def test_tick_floor_keys_off_the_traded_price_not_the_adjusted_index(): + """price_history.close_try is a chained total-return index, not a price. A serial + bonus-issuer's index can sit far above its actual print - 118x at the extreme on + 2018-12 data. The tick floor is 0.01 TRY on the exchange's grid, so keying it off the + index divides the floor by that factor and hands the trade a spread it could never + get. The estimator is scale-free and stays on the index; the floor takes the price. + """ + n = 40 + # a quiet window, so the estimate IS the floor and nothing else can mask the error + closes = [Decimal("100")] * n # index level + highs = [Decimal("100")] * n + lows = [Decimal("100")] * n + + traded = Decimal("2.50") # what the stock actually prints after its bonus issues + + on_index = abdi_ranaldo_spread_pct(closes, highs, lows) + on_price = abdi_ranaldo_spread_pct(closes, highs, lows, last_traded_price=traded) + + assert on_index == tick_floor_pct(Decimal("100")) # 0.02/100 = 2bp - fiction + assert on_price == tick_floor_pct(traded) # 0.01/2.50 = 40bp - the real grid + assert on_price > on_index * 10 + + def test_a_bad_price_inside_the_estimation_window_voids_the_estimate(): closes, highs, lows = _series(40) closes[-3] = Decimal("0") # inside the trailing 21-session window From 3c646048ffea5ad91b911b8bc27b11c47908d3a3 Mon Sep 17 00:00:00 2001 From: cagan Date: Mon, 13 Jul 2026 04:03:29 +0300 Subject: [PATCH 27/29] fix: make the exchange bulletin authoritative over yfinance in price_history Two writers touch price_history and they do not mean the same thing by close_try. load_official_prices.py stores a chained corporate-action-adjusted total-return index with the matching raw print alongside it; data/prices.py stores yfinance's own adjusted close and knows nothing about raw_close_try. `prices backfill` pulls EVERY distinct KAP insider ticker from yfinance, so it would have walked over the whole bulletin-derived series: replacing a survivorship-clean chain (yfinance serves nothing at all for a delisted BIST ticker, not even the years it traded) with a biased one, dropping the VBTS flags, and leaving close_try and raw_close_try sourced from two different providers while the cost model reads both. The upsert now refuses any row that already carries raw_close_try - i.e. any row the bulletin wrote. yfinance keeps the job it is genuinely needed for: XU100, which is an index and appears in no equity bulletin. Pinned by an integration test in both directions, because this regression would raise no error, only a quietly worse number. --- src/trailing_edge/data/prices.py | 16 +++ .../signals/test_bulletin_is_authoritative.py | 105 ++++++++++++++++++ 2 files changed, 121 insertions(+) create mode 100644 tests/integration/signals/test_bulletin_is_authoritative.py diff --git a/src/trailing_edge/data/prices.py b/src/trailing_edge/data/prices.py index 06fbce5..84c55b6 100644 --- a/src/trailing_edge/data/prices.py +++ b/src/trailing_edge/data/prices.py @@ -192,6 +192,21 @@ def _to_dec(v) -> Decimal | None: continue insert_stmt = pg_insert(PriceHistory.__table__).values(values_list) + # The exchange bulletin is authoritative and yfinance must not overwrite + # it. They are not interchangeable sources of the same number: + # + # - the bulletin is survivorship-clean; yfinance serves NOTHING for a + # delisted BIST ticker, not even the years it traded. + # - the bulletin carries the VBTS flags (gross settlement, suspension). + # - scripts/load_official_prices.py stores close_try as a chained + # total-return index with the matching raw print in raw_close_try. + # Letting yfinance replace close_try alone would leave the pair + # inconsistent - an index from one source, a price from another - and + # the cost model reads both. + # + # So a row that already has raw_close_try came from the bulletin and is + # left alone. yfinance stays what it is genuinely needed for: the XU100 + # benchmark, which is an index and appears in no equity bulletin. upsert_stmt = insert_stmt.on_conflict_do_update( constraint="uq_price_ticker_date", set_={ @@ -201,6 +216,7 @@ def _to_dec(v) -> Decimal | None: "low_try": insert_stmt.excluded.low_try, "volume": insert_stmt.excluded.volume, }, + where=PriceHistory.__table__.c.raw_close_try.is_(None), ) await session.execute(upsert_stmt) results[ticker] = len(values_list) diff --git a/tests/integration/signals/test_bulletin_is_authoritative.py b/tests/integration/signals/test_bulletin_is_authoritative.py new file mode 100644 index 0000000..57e7409 --- /dev/null +++ b/tests/integration/signals/test_bulletin_is_authoritative.py @@ -0,0 +1,105 @@ +"""yfinance must not overwrite an exchange-bulletin price row. + +Two writers touch price_history and they do NOT mean the same thing by close_try: + + - scripts/load_official_prices.py stores a chained corporate-action-adjusted + total-return index, with the matching raw print in raw_close_try. + - data/prices.py (yfinance) stores yfinance's own adjusted close, and knows nothing + about raw_close_try. + +`trailing-edge prices backfill` pulls every KAP insider ticker from yfinance, so without +a guard it would walk over the entire bulletin-derived series - replacing a +survivorship-clean chain (yfinance serves NOTHING for a delisted BIST ticker) with a +biased one, dropping the VBTS flags, and leaving close_try and raw_close_try sourced from +two different providers while the cost model reads both. + +This pins the guard. It is the kind of regression that produces no error, only a quietly +worse number. +""" +from datetime import date +from decimal import Decimal + +import pytest +from sqlalchemy import delete, select +from sqlalchemy.dialects.postgresql import insert as pg_insert + +from trailing_edge.models.signal import PriceHistory + +_TICKER = "ZZTEST" +_DAY = date(2019, 3, 14) + + +async def _upsert_like_yfinance(session, close: Decimal) -> None: + """Exactly the statement data/prices.py issues, guard included.""" + stmt = pg_insert(PriceHistory.__table__).values( + [{"ticker": _TICKER, "price_date": _DAY, "close_try": close, "volume": 1}] + ) + await session.execute( + stmt.on_conflict_do_update( + constraint="uq_price_ticker_date", + set_={ + "close_try": stmt.excluded.close_try, + "open_try": stmt.excluded.open_try, + "high_try": stmt.excluded.high_try, + "low_try": stmt.excluded.low_try, + "volume": stmt.excluded.volume, + }, + where=PriceHistory.__table__.c.raw_close_try.is_(None), + ) + ) + + +@pytest.mark.asyncio +async def test_yfinance_cannot_overwrite_a_bulletin_row(db_session): + await db_session.execute(delete(PriceHistory).where(PriceHistory.ticker == _TICKER)) + + # a bulletin row: chained index 100, the stock actually printed 2.50 + await db_session.execute( + pg_insert(PriceHistory.__table__).values( + [ + { + "ticker": _TICKER, + "price_date": _DAY, + "close_try": Decimal("100.0000"), + "raw_close_try": Decimal("2.5000"), + "volume": 1, + } + ] + ) + ) + + await _upsert_like_yfinance(db_session, Decimal("2.5000")) + + row = ( + await db_session.execute( + select(PriceHistory.close_try, PriceHistory.raw_close_try).where( + PriceHistory.ticker == _TICKER + ) + ) + ).one() + + assert row.close_try == Decimal("100.0000"), "the bulletin's chained index must survive" + assert row.raw_close_try == Decimal("2.5000") + + await db_session.execute(delete(PriceHistory).where(PriceHistory.ticker == _TICKER)) + + +@pytest.mark.asyncio +async def test_yfinance_still_fills_a_row_the_bulletin_never_covered(db_session): + """The guard must not become a blanket refusal: XU100 is an index, appears in no + equity bulletin, and yfinance is the only source for it. A row with no raw_close_try + did not come from the bulletin and is still yfinance's to write.""" + await db_session.execute(delete(PriceHistory).where(PriceHistory.ticker == _TICKER)) + + await _upsert_like_yfinance(db_session, Decimal("11.0000")) + await _upsert_like_yfinance(db_session, Decimal("12.0000")) # a later correction + + row = ( + await db_session.execute( + select(PriceHistory.close_try).where(PriceHistory.ticker == _TICKER) + ) + ).one() + + assert row.close_try == Decimal("12.0000") + + await db_session.execute(delete(PriceHistory).where(PriceHistory.ticker == _TICKER)) From 148aa8b2b22ed979ebf31ba2aa608aee8d9ec7ca Mon Sep 17 00:00:00 2001 From: cagan Date: Mon, 13 Jul 2026 04:09:55 +0300 Subject: [PATCH 28/29] report: emit the net-of-cost verdict alongside the gross one compute_base_rate measures the GROSS abnormal return and knows nothing about the bid-ask spread. On this sample it correctly returns EDGE_DETECTED: +2.07% at 20 days, t = 5.36, N = 1,079, survivorship-clean. It is a real signal. It is also uncapturable, which is this project's entire finding - and the daily report was printing "EDGE_DETECTED" and a 54.7% hit rate and stopping there. The tool contradicted its own conclusion on its own screen. Anyone running it would trade a signal measured to lose 1.35% net at 20 days. So the net result now travels with the gross one, in both channels: - stdout prints the cost warning under the table - the JSON carries a `cost_adjusted` block, because a machine consumer reading `verdict: EDGE_DETECTED` never opens the README Numbers throughout refreshed onto the corrected cost basis (raw price rather than the total-return index, migration 0008): gross 20d +2.07% t = +5.36 N = 1,079 net 20d -1.35% t = -3.31 N = 1,070 cost median 1.93% mean 3.37% p75 4.34% VBTS-restricted entries re-measured: 16 of 1,079 (1.5%). Still not the constraint. Pre-registered opportunistic test re-run on the corrected data, decision rule NOT re-chosen: OPPORTUNISTIC N=1,059, gross +2.08%, net -1.31%, t = -3.18. ROUTINE = 0. The recorded prior said the subset would be "stronger gross but still not clear the spread" - it was not even stronger. docs/stage0/OPPORTUNISTIC_CLASSIFIER.md is APPENDED to, never edited. Its frozen baseline (+1.76% / 1.94%) is left standing even though both numbers have since moved: a pre-registration that gets rewritten once the answer is known is worth less than none. reports/sample regenerated from a real run and re-anonymized per the repo's convention. Its README no longer claims "nothing measured yet" - that was two findings out of date. --- README.md | 45 ++++-- docs/METHODOLOGY.md | 49 ++++-- docs/stage0/OPPORTUNISTIC_CLASSIFIER.md | 30 ++++ reports/sample/README.md | 78 ++++++---- reports/sample/daily_signal.example.json | 140 +++++++++++------- src/trailing_edge/reports/daily_signal.py | 40 +++++ .../signals/test_cluster_score_seniority.py | 10 +- 7 files changed, 280 insertions(+), 112 deletions(-) diff --git a/README.md b/README.md index 051b1a6..c8a33e4 100644 --- a/README.md +++ b/README.md @@ -23,20 +23,23 @@ interval, and the round-trip cost is estimated per trade from that stock's own O | Horizon | N | Gross AR | Cost | **Net AR** | t (net) | |---|---:|---:|---:|---:|---:| -| 5d | 1,032 | +0.57% | 3.34% | **−2.77%** | −13.11 | -| 20d | 1,032 | +1.76% | 3.34% | **−1.58%** | −4.04 | -| 60d | 1,032 | +2.09% | 3.34% | **−1.24%** | −1.98 | +| 5d | 1,070 | +0.66% | 3.37% | **−2.71%** | −12.91 | +| 20d | 1,070 | +2.02% | 3.37% | **−1.35%** | −3.31 | +| 60d | 1,071 | +2.18% | 3.37% | **−1.20%** | −1.86 | **The signal is real. The gross abnormal return is significantly positive at every -horizon** (20d: +1.76%, t = 5.12). **And it is not tradeable**, because insider clusters -fire in illiquid small caps whose bid-ask spread is wider than the alpha: the median -round trip costs 1.94%, the upper quartile 4.24%. Nothing survives crossing it twice. +horizon** (20d: +2.07%, t = 5.36, N = 1,079). **And it is not tradeable**, because insider +clusters fire in illiquid small caps whose bid-ask spread is wider than the alpha: the +median round trip costs 1.93%, the upper quartile 4.34%. Nothing survives crossing it +twice. At 60 days the net loss is no longer statistically distinguishable from zero +(t = −1.86) - which buys nothing: the point estimate is still negative, and "you might +merely break even after three months" is not an edge either. That is the whole finding, and it is why this repository exists. A gross number is not an edge; an edge is what is left after the market takes its cut. Everything below is the machinery required to be able to say that honestly - and the -audit trail of the six silent data faults that had to be found first, each of which had +audit trail of the silent data faults that had to be found first, each of which had moved the number: | | | @@ -47,12 +50,25 @@ moved the number: | WAF disconnects were caught and skipped | 12% of disclosures vanished, run still reported `SUCCESS` | | Prices fetched in one batch; one bad symbol poisoned the rest | looked exactly like survivorship bias | | yfinance serves nothing for a delisted ticker | 31% of clusters dropped - the dead ones, the worst outcomes | +| **The cost model read the total-return index as if it were a price** | the tick floor and ADV were wrong on 32% of ticker-days | +| KAP's `relatedStocks` is not always one ticker | `KRDMA, KRDMB, KRDMD` joined to no price row; 14 clusters left every result in silence | -The last of those was fixed by loading Borsa İstanbul's own end-of-day bulletin, which is -survivorship-clean by construction: 1.3M rows, 747 tickers, against yfinance's 185. It +The delisting fault was fixed by loading Borsa İstanbul's own end-of-day bulletin, which +is survivorship-clean by construction: 1.3M rows, 749 tickers, against yfinance's 185. It also carries the VBTS tradability flags - which, once measured, turned out to touch only 1% of entries and not to be the constraint at all. +The cost fault is the one worth dwelling on, because it sat directly under the number that +decides the answer. `close_try` is a *chained total-return index* - correct for returns, +since a bonus issue halves the print and a raw series would read it as a 50% loss. It is +not a price. But the tick floor is 0.01 TRY on the exchange's grid and ADV is price × +volume, and both were being fed the index. BIST companies issue bonus shares constantly, +so the two series pull apart: a median 0.98× but ranging 0.60× to 118×. The error was not +one-directional, so it was not conservative - it simply mispriced trades, worst in the +serial bonus-issuers, which are small caps, which are precisely where the tick floor binds +and where tradeability is decided. Fixing it *raised* N from 1,032 to 1,070 and left the +verdict standing. + ## What it does - **Scrapes** Turkey's public-disclosure platform (KAP) for SPK II-15.1 @@ -80,7 +96,7 @@ also carries the VBTS tradability flags - which, once measured, turned out to to clusters cannot be priced. Both gates fired during this work, and both were right. > **What is claimed, precisely:** a statistically strong *gross* abnormal return -> (20d: +1.76%, t = 5.12, N = 1,032, survivorship-clean) that does **not** survive a +> (20d: +2.07%, t = 5.36, N = 1,079, survivorship-clean) that does **not** survive a > per-trade cost estimate. The window is 2015-2018 - a single regime - so the result is > not yet regime-conditional, and that is stated rather than glossed. Remaining gaps are > in [`docs/METHODOLOGY.md`](docs/METHODOLOGY.md#6-still-open), not left for a @@ -157,12 +173,13 @@ python scripts/net_of_cost.py # the one that decides it ``` === Abnormal return, NET of round-trip cost (order 25,000 TRY) === spread: Abdi-Ranaldo (2017) from the stock's own OHLC, per trade - round-trip cost: median 1.94% p25 1.22% p75 4.24% + dropped (no cost estimate): 27 + round-trip cost: median 1.93% p25 1.19% p75 4.34% HORIZON N GROSS AR% COST% NET AR% HIT% 95% CI t VERDICT - 5d 1032 0.57 3.34 -2.77 26.6 [23.9, 29.3] -13.11 LOSES MONEY (net) - 20d 1032 1.76 3.34 -1.58 41.4 [38.4, 44.4] -4.04 LOSES MONEY (net) - 60d 1032 2.09 3.34 -1.24 44.9 [41.9, 47.9] -1.98 LOSES MONEY (net) + 5d 1070 0.66 3.37 -2.71 26.9 [24.3, 29.7] -12.91 LOSES MONEY (net) + 20d 1070 2.02 3.37 -1.35 41.3 [38.4, 44.3] -3.31 LOSES MONEY (net) + 60d 1071 2.18 3.37 -1.20 44.4 [41.5, 47.4] -1.86 NO EDGE (net) ``` The spread is not a parameter. It is estimated for each trade from the 30 sessions of diff --git a/docs/METHODOLOGY.md b/docs/METHODOLOGY.md index 9890399..3cff0a6 100644 --- a/docs/METHODOLOGY.md +++ b/docs/METHODOLOGY.md @@ -168,16 +168,33 @@ have flattered the answer, and not taken from a quote feed, which does not exist delisted names the exchange bulletin carries. Impact is Kyle/Almgren square-root on the same window; commission and BSMV are charged per side. - round-trip cost: median 1.94% p25 1.22% p75 4.24% + round-trip cost: median 1.93% p25 1.19% p75 4.34% - horizon N=1032 gross AR net AR t (net) - 5d +0.57% -2.77% -13.11 - 20d +1.76% -1.58% -4.04 - 60d +2.09% -1.24% -1.98 + horizon N=1070 gross AR net AR t (net) + 5d +0.66% -2.71% -12.91 + 20d +2.02% -1.35% -3.31 + 60d +2.18% -1.20% -1.86 (not significant) **The signal does not survive the cost of trading it.** Insider clusters fire in illiquid small caps, and the spread on those names is wider than the alpha. This is the project's -result, not a caveat on it. +result, not a caveat on it. At 60 days the net loss stops being statistically +distinguishable from zero, which is not a reprieve: the point estimate is still negative, +and a signal that *may* break even over three months is not an edge either. + +A second trap, found later and more serious than the first, because it sat under the +number that decides the answer. `price_history.close_try` is a **chained total-return +index**, not a price - correct for returns, and every return here is computed from it. But +the tick floor is 0.01 TRY on the exchange's grid, and ADV is price x volume, and both +were being handed the index. BIST companies issue bonus shares constantly, so the index +and the print pull apart: measured on 2018-12 bulletin data, a median 0.98x but a range of +0.60x to 118x, with 32% of ticker-days off by more than 10%. Because the factor falls on +both sides of 1, the error was not conservative - it mispriced trades in both directions, +worst in the serial bonus-issuers, which are small caps, which is exactly where the tick +floor binds. Migration 0008 keeps the raw print alongside the index; the spread estimator +is scale-free and correctly stays on the index, while the floor and ADV moved to the +price. Correcting it *raised* N from 1,032 to 1,070 (the earliest 2015 clusters had been +silently dropped for want of a 22-session lookback, and those clusters averaged +8.45% at +20 days against +1.93% for the rest) and left the verdict standing. One trap worth recording: the first version of the cost script *dropped* any trade whose Abdi-Ranaldo window came back degenerate (gamma >= 0, so a zero spread). That discarded @@ -195,8 +212,7 @@ this looked like it might matter a great deal. The exchange bulletin carries the flags (`BRUT TAKAS`, `GECICI DURDURMA`), so they are now loaded. Measured: 143,012 gross-settlement ticker-days across 451 names - 11% of all price -rows. But of 1,055 cluster entries, **only 11 (1.0%)** land on a restricted day, and none -on a suspended one. +rows. But of 1,079 cluster entries, **only 16 (1.5%)** land on a restricted day. So the gap is closed and it was never the binding constraint. The cost is. @@ -206,15 +222,20 @@ Pre-registered in `docs/stage0/OPPORTUNISTIC_CLASSIFIER.md` and frozen before it because this was the signal's last plausible route to being tradeable and therefore exactly where a definition chosen after the fact would be most tempting. - cluster classes at 20d: OPPORTUNISTIC=1043 ROUTINE=0 UNCLASSIFIED=12 + cluster classes at 20d: OPPORTUNISTIC=1066 ROUTINE=0 UNCLASSIFIED=13 + + OPPORTUNISTIC 20d N=1059 gross +2.08% cost 3.38% net -1.31% t = -3.18 + LOSES MONEY (net) **Not one cluster classified as routine**, and the frozen document had said why in advance: SPK II-15.1's 250,000 TRY reporting threshold already censors the small, quiet, scheduled trades that CMP's routine class is built from. Turkish insiders have no routine *filings* because routine trades are never disclosed at all. -So there is no noise to strip. The opportunistic subset IS the sample (1,043 of 1,055), its -gross return is +1.80% against the pooled +1.76%, and it loses to the same spread. +So there is no noise to strip. The opportunistic subset IS the sample (1,066 of 1,079), its +gross return is +2.08% against the pooled +2.07% - indistinguishable - and it loses to the +same spread. The pre-registered prior said the subset "will be stronger gross but will +still not clear the spread"; it was not even stronger. CMP's hypothesis is not refuted - it is **inapplicable**. The regulation that makes this dataset possible is the same regulation that removes the variation the test needs. That is @@ -222,9 +243,9 @@ a finding about Turkish disclosure, not about insiders. *Limitation, stated:* the 36-month lookback is thin on 3.5 years of data, so a fuller backfill would classify more insiders and might surface some routine ones. The direction is -known - stripping routine trades RAISES gross alpha - but it would have to raise +1.76% -past +1.94%, which is more than CMP's own effect size, and the disclosure threshold has -already removed most of what would do the raising. +known - stripping routine trades RAISES gross alpha - but it would have to raise +2.08% +past the 3.38% mean cost, which is far more than CMP's own effect size, and the disclosure +threshold has already removed most of what would do the raising. ### Split-sample stability - the conclusion holds in both halves diff --git a/docs/stage0/OPPORTUNISTIC_CLASSIFIER.md b/docs/stage0/OPPORTUNISTIC_CLASSIFIER.md index 8881012..f569fca 100644 --- a/docs/stage0/OPPORTUNISTIC_CLASSIFIER.md +++ b/docs/stage0/OPPORTUNISTIC_CLASSIFIER.md @@ -85,3 +85,33 @@ in the US sample. Recording this matters: if the result comes out positive, it comes out *against* the prior, and that is worth more than a confirmation. + +--- + +## Addendum, 2026-07-13 (after the fact - the text above is unchanged) + +Nothing above this line has been edited. A pre-registration that gets rewritten once the +answer is known is worth less than no pre-registration at all, so the baseline figures in +it (+1.76% gross against a 1.94% median cost) stay exactly as they were written, even +though both numbers have since moved. + +They moved because a fault was found in the cost model *after* this document was frozen: +`price_history.close_try` is a chained total-return index, and the tick floor and ADV were +reading it as if it were a traded price. On the corrected basis the pooled figures are ++2.07% gross (N = 1,079) against a 1.93% median / 3.37% mean round-trip cost. See +`docs/METHODOLOGY.md` §5. + +**The frozen decision rule was applied to the corrected data, not re-chosen for it:** + + OPPORTUNISTIC 20d (primary) N = 1,059 gross +2.08% net -1.31% t = -3.18 + ROUTINE N = 0 + +- N = 1,059 clears the pre-registered MIN_N of 200. +- Net is negative with p < 0.05, so the subset is **not** declared tradeable. +- The recorded prior said the opportunistic subset "will be stronger gross but will still + not clear the spread." Half right, and the wrong half is the interesting one: it was + **not stronger gross at all** (+2.08% against the pooled +2.07%). There was no routine + class to strip, because the 250,000 TRY disclosure threshold means routine trades are + never filed in the first place. + +The hypothesis is not refuted. It is inapplicable. diff --git a/reports/sample/README.md b/reports/sample/README.md index 1d68085..e601539 100644 --- a/reports/sample/README.md +++ b/reports/sample/README.md @@ -1,26 +1,42 @@ # Sample daily signal report -`daily_signal.example.json` is a faithful example of what `trailing-edge report daily` -emits — including, right now, its refusal to report anything. +`daily_signal.example.json` is a faithful example of what `trailingedge signal daily-report` +emits. The cluster is anonymized (ticker and insider names); every statistic in it is real +and reproducible from the pipeline. -## Why every base rate reads `INSUFFICIENT_POWER` with `signals_with_outcome: 0` +## Read the two verdicts together, or you will read it wrong -Migration `0006` made the market-adjusted **abnormal return** the primary metric. -Outcomes written before that migration have `abnormal_return_pct = NULL` and are -deliberately **not** backfilled with a guessed benchmark — a NULL that forces a -recompute is safer than a plausible wrong number. `compute_base_rate` counts only -rows it can market-adjust, so until the pipeline is re-run the honest answer is -"nothing measured yet". +The file carries two, and they disagree on purpose: -To populate it: - -```bash -trailing-edge prices backfill # fetches XU100 alongside the signal universe -trailing-edge signal returns # recomputes outcomes with abnormal returns -trailing-edge report daily +```json +"base_rates": { "20": { "verdict": "EDGE_DETECTED", ... } } +"cost_adjusted": { "verdict": "LOSES_MONEY_NET_OF_COST", ... } ``` -## What the previous version of this file claimed, and why it was void +`compute_base_rate` measures the **gross** abnormal return and knows nothing about the +bid-ask spread. On this sample it is right: +2.07% at 20 days, t = 5.36, N = 1,079, +survivorship-clean. That is a real, statistically strong signal. + +It is also uncapturable. Insider clusters fire in illiquid BIST small caps whose round +trip costs a median 1.93% — wider than the alpha. Net of cost the same sample returns +**−1.35% at 20 days (t = −3.31)**. + +A report that printed `EDGE_DETECTED` and a 54.7% hit rate and stopped there would have +this tool contradict its own project's conclusion, and a reader would trade it. So the net +result is emitted in the same document and printed on the same screen, rather than left in +a methodology file nobody opens. The gross number is not the finding; the gap between the +two numbers is. + +## `cluster_score` is not yet a real score + +The sample scores 46.17 and 45.50. Both are computed with the seniority term pinned at its +0.5 default, because `person_company_roles` is empty — the report logs `role_map_empty` on +every run. Until `graph scrape-management` populates the board roster, `cluster_score` is +`insider_count` with a decimal point on it, and it is documented as such in +[`docs/METHODOLOGY.md`](../../docs/METHODOLOGY.md#6-still-open) rather than presented as a +model. + +## What the first version of this file claimed, and why it was void For the record, since the numbers were public: @@ -30,27 +46,27 @@ For the record, since the numbers were public: | 20d | 29 | 55.17% | −0.48% | | 60d | 23 | 52.17% | +5.91% | -Two things make those figures unusable as evidence, and both are now fixed in code: +Two things made those figures unusable, and both are fixed in code: **They were raw returns, not abnormal returns.** BIST quotes nominal TRY under high -inflation, so a randomly chosen stock rises well over half the time and its mean -return is not zero. A "hit rate" scored against a 50% coin-flip prior and a mean -scored against 0% hand the signal credit for market drift and beta. The 60-day -`+5.91%` in particular is comfortably inside what the index alone returned over the -same span. See `src/trailing_edge/signals/abnormal.py`. +inflation, so a randomly chosen stock rises well over half the time and its mean return is +not zero. A hit rate scored against a 50% coin-flip prior and a mean scored against 0% +hand the signal credit for market drift. The 60-day `+5.91%` in particular is comfortably +inside what the index alone returned over the same span. See +`src/trailing_edge/signals/abnormal.py`. -**N was 29.** The 95% Wilson interval around a 55.17% hit rate is roughly -**[37%, 72%]** — it contains the 50% null with room to spare. Separating a 55% hit -rate from 50% at α=0.05 with 80% power needs +**N was 29.** The 95% Wilson interval around a 55.17% hit rate is roughly **[37%, 72%]** — +it contains the 50% null with room to spare. Separating a 55% hit rate from 50% at α=0.05 +with 80% power needs ``` n = (1.96 + 0.84)² × 0.25 / 0.05² ≈ 784 ``` -observations. At N=29 the sample cannot distinguish this signal from a coin, in -*either* direction. Publishing the point estimate without that interval was the -actual error; the number itself was never the problem. +observations. At N=29 the sample could not distinguish this signal from a coin in *either* +direction. Publishing the point estimate without that interval was the actual error; the +number itself was never the problem. -`compute_base_rate` now returns a `verdict` field, and `INSUFFICIENT_POWER` is a -gate rather than a footnote: below the power threshold the point estimates are not -evidence and downstream reports must not present them as such. +N was never 29 as a fact about the market. It was the sum of the silent data faults listed +in the top-level [README](../../README.md) — each of which had moved it. After they were +fixed, N is 1,079. diff --git a/reports/sample/daily_signal.example.json b/reports/sample/daily_signal.example.json index 604002b..eecbfab 100644 --- a/reports/sample/daily_signal.example.json +++ b/reports/sample/daily_signal.example.json @@ -1,75 +1,115 @@ { - "as_of_date": "2026-05-28", + "as_of_date": "2018-12-14", "clusters": [ { "ticker": "XXXXX", - "cluster_score": 42.8333, + "cluster_score": 46.1667, "insider_count": 2, - "window_start": "2026-05-21", - "window_end": "2026-05-21", - "days_since_last_buy": 7, + "window_start": "2018-12-11", + "window_end": "2018-12-12", + "days_since_last_buy": 2, "unique_insiders": [ "INSIDER A", "INSIDER B" ], - "total_buy_value_try": 711360.0 + "total_buy_value_try": 7563364.61 + }, + { + "ticker": "XXXXX", + "cluster_score": 45.5, + "insider_count": 2, + "window_start": "2018-12-11", + "window_end": "2018-12-11", + "days_since_last_buy": 3, + "unique_insiders": [ + "INSIDER A", + "INSIDER B" + ], + "total_buy_value_try": 2490289.61 } ], "base_rates": { "5": { "horizon_days": 5, - "benchmark_ticker": null, - "total_signals": 30, - "signals_with_outcome": 0, - "verdict": "INSUFFICIENT_POWER", + "benchmark_ticker": "XU100", + "total_signals": 1093, + "signals_with_outcome": 1079, + "verdict": "EDGE_DETECTED", "required_n_for_power": 784, - "hit_rate_pct": 0.0, - "hit_rate_ci_95": [0.0, 0.0], - "mean_abnormal_return_pct": 0.0, - "median_abnormal_return_pct": 0.0, - "t_stat": 0.0, - "p_value": 1.0, - "best_abnormal_return_pct": 0.0, - "worst_abnormal_return_pct": 0.0, - "mean_raw_return_pct": 0.0, - "mean_benchmark_return_pct": 0.0 + "hit_rate_pct": 52.09, + "hit_rate_ci_95": [ + 49.1, + 55.05 + ], + "mean_abnormal_return_pct": 0.68, + "median_abnormal_return_pct": 0.14, + "t_stat": 3.75, + "p_value": 0.0002, + "best_abnormal_return_pct": 35.78, + "worst_abnormal_return_pct": -23.05, + "mean_raw_return_pct": 0.65, + "mean_benchmark_return_pct": -0.03 }, "20": { "horizon_days": 20, - "benchmark_ticker": null, - "total_signals": 30, - "signals_with_outcome": 0, - "verdict": "INSUFFICIENT_POWER", + "benchmark_ticker": "XU100", + "total_signals": 1093, + "signals_with_outcome": 1079, + "verdict": "EDGE_DETECTED", "required_n_for_power": 784, - "hit_rate_pct": 0.0, - "hit_rate_ci_95": [0.0, 0.0], - "mean_abnormal_return_pct": 0.0, - "median_abnormal_return_pct": 0.0, - "t_stat": 0.0, - "p_value": 1.0, - "best_abnormal_return_pct": 0.0, - "worst_abnormal_return_pct": 0.0, - "mean_raw_return_pct": 0.0, - "mean_benchmark_return_pct": 0.0 + "hit_rate_pct": 54.68, + "hit_rate_ci_95": [ + 51.7, + 57.63 + ], + "mean_abnormal_return_pct": 2.07, + "median_abnormal_return_pct": 0.78, + "t_stat": 5.36, + "p_value": 0.0, + "best_abnormal_return_pct": 96.68, + "worst_abnormal_return_pct": -57.35, + "mean_raw_return_pct": 2.33, + "mean_benchmark_return_pct": 0.26 }, "60": { "horizon_days": 60, - "benchmark_ticker": null, - "total_signals": 30, - "signals_with_outcome": 0, - "verdict": "INSUFFICIENT_POWER", + "benchmark_ticker": "XU100", + "total_signals": 1094, + "signals_with_outcome": 1080, + "verdict": "EDGE_DETECTED", "required_n_for_power": 784, - "hit_rate_pct": 0.0, - "hit_rate_ci_95": [0.0, 0.0], - "mean_abnormal_return_pct": 0.0, - "median_abnormal_return_pct": 0.0, - "t_stat": 0.0, - "p_value": 1.0, - "best_abnormal_return_pct": 0.0, - "worst_abnormal_return_pct": 0.0, - "mean_raw_return_pct": 0.0, - "mean_benchmark_return_pct": 0.0 + "hit_rate_pct": 52.13, + "hit_rate_ci_95": [ + 49.15, + 55.1 + ], + "mean_abnormal_return_pct": 2.31, + "median_abnormal_return_pct": 1.02, + "t_stat": 3.6, + "p_value": 0.0003, + "best_abnormal_return_pct": 101.25, + "worst_abnormal_return_pct": -65.82, + "mean_raw_return_pct": 4.25, + "mean_benchmark_return_pct": 1.94 } }, - "generated_at": "2026-05-28" -} + "cost_adjusted": { + "basis": "Abdi-Ranaldo (2017) spread + Kyle/Almgren impact + commission/BSMV, estimated per trade from the stock's own OHLC", + "median_round_trip_cost_pct": 1.93, + "mean_round_trip_cost_pct": 3.37, + "net_abnormal_return_pct": { + "5": -2.71, + "20": -1.35, + "60": -1.2 + }, + "net_t_stat": { + "5": -12.91, + "20": -3.31, + "60": -1.86 + }, + "n": 1070, + "verdict": "LOSES_MONEY_NET_OF_COST", + "note": "The gross verdict above is real and significant. It is not tradeable: insider clusters fire in illiquid small caps whose bid-ask spread is wider than the alpha. See docs/METHODOLOGY.md." + }, + "generated_at": "2018-12-14" +} \ No newline at end of file diff --git a/src/trailing_edge/reports/daily_signal.py b/src/trailing_edge/reports/daily_signal.py index d5e38a3..959f6a8 100644 --- a/src/trailing_edge/reports/daily_signal.py +++ b/src/trailing_edge/reports/daily_signal.py @@ -96,6 +96,29 @@ def _print_table(as_of_date: date, clusters: list[InsiderCluster], base_rates: d ) click.echo(sep_row) + _print_cost_warning() + + +# The base rate is computed GROSS. compute_base_rate knows nothing about the bid-ask +# spread, so on this sample it returns EDGE_DETECTED - a real, significant, and entirely +# uncapturable +2.07% at 20 days. Printing a hit rate and a verdict with no mention of the +# cost that eats them would have this tool contradict its own project's finding, and would +# be the single most misleading thing it could do: a reader sees "EDGE_DETECTED, 54.7%" +# and trades it. The measured net result goes on the same screen as the gross one. +_NET_OF_COST_20D = "-1.35%" +_NET_T_20D = "-3.31" + + +def _print_cost_warning() -> None: + click.echo("") + click.echo(" HIT RATE AND BASE RATE ABOVE ARE GROSS - BEFORE TRANSACTION COSTS.") + click.echo( + f" Net of the measured round-trip cost, this signal has historically LOST money:" + f" 20d net {_NET_OF_COST_20D} (t = {_NET_T_20D}, N = 1,070)." + ) + click.echo(" Insider clusters fire in illiquid small caps whose spread is wider than") + click.echo(" the alpha. See docs/METHODOLOGY.md and scripts/net_of_cost.py.") + click.echo("") def _to_json_safe(obj): @@ -186,6 +209,23 @@ async def generate_daily_report(as_of_date: date | None = None) -> DailyReport: } for h in horizons }, + # Every base_rate above is GROSS. A machine consumer reading `verdict: + # EDGE_DETECTED` and acting on it would be acting on a number this project has + # measured as uncapturable, so the net result travels in the same document rather + # than in a README the consumer never reads. + "cost_adjusted": { + "basis": "Abdi-Ranaldo (2017) spread + Kyle/Almgren impact + commission/BSMV," + " estimated per trade from the stock's own OHLC", + "median_round_trip_cost_pct": 1.93, + "mean_round_trip_cost_pct": 3.37, + "net_abnormal_return_pct": {"5": -2.71, "20": -1.35, "60": -1.20}, + "net_t_stat": {"5": -12.91, "20": -3.31, "60": -1.86}, + "n": 1070, + "verdict": "LOSES_MONEY_NET_OF_COST", + "note": "The gross verdict above is real and significant. It is not tradeable:" + " insider clusters fire in illiquid small caps whose bid-ask spread is wider" + " than the alpha. See docs/METHODOLOGY.md.", + }, "generated_at": str(today), } report_path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8") diff --git a/tests/unit/signals/test_cluster_score_seniority.py b/tests/unit/signals/test_cluster_score_seniority.py index 522cb8f..7fee95b 100644 --- a/tests/unit/signals/test_cluster_score_seniority.py +++ b/tests/unit/signals/test_cluster_score_seniority.py @@ -30,11 +30,15 @@ def _score(roles, count=2, gap=7): def test_reproduces_the_degenerate_score_from_the_committed_sample(): - """reports/sample/daily_signal.example.json recorded cluster_score=42.8333. + """The committed sample still scores with seniority pinned at its 0.5 default, + because person_company_roles is effectively empty in production - the report logs + `role_map_empty` every run. Under that default the score collapses to: - That value is exactly what you get when seniority is the 0.5 default: (0.25 * 0.5) + (0.5 * 0.3) + (0.7667 * 0.2) = 0.428333 - Reproducing it from all-None roles is the proof the map was dead in production. + + Reproducing it from all-None roles is the proof the map is dead. Until + `graph scrape-management` populates the roster, cluster_score is insider_count wearing + a decimal point. """ assert _score([None, None]) == Decimal("42.8333") From 608e733cb418f0c17a88cedc425c74f9decc3072 Mon Sep 17 00:00:00 2001 From: cagan Date: Mon, 13 Jul 2026 04:30:14 +0300 Subject: [PATCH 29/29] chore: mypy clean, and cover the two inserts that the type fix touched 12 mypy errors, all predating this work, and one of them was hiding a real defect. repository.upsert_disclosure guarded `row` for None on one line and then dereferenced it on the next, with a `# type: ignore[return-value]` over the top. INSERT ... ON CONFLICT DO UPDATE ... RETURNING always yields a row so it never fired, but the failure mode it was papering over is handing the caller a None model that reads as a successfully stored disclosure. Both invariants are now checked and raise. The rest were typing, not behaviour: - pg_insert(Model.__table__) -> pg_insert(Model) (5 call sites: prices, cluster, returns, and two in the graph repository) - Result.rowcount -> cast to CursorResult, which is what execute() actually returns - TsgClient's playwright handles were annotated None by inference - BeautifulSoup types an href as str | AttributeValueList Changing five live INSERT statements to satisfy a type checker is exactly the kind of edit that type-checks and then writes nothing, so they are verified rather than assumed. Four were already covered. The fifth - insert(Person)/insert(PersonCompanyRole) in upsert_management_roles - was only reachable through a test that scrapes KAP live, and so could not run at all while the WAF was blocking. It now has an offline test that writes synthetic board members and re-runs them, which also pins the idempotency the method's own docstring claims (a NULL valid_from makes ON CONFLICT never fire, so it DELETEs first). Remaining 4 integration failures are unrelated and pre-existing: they assert on graph and Ticaret Sicil seed data that was never loaded (unlisted_companies is 0 - that scraper needs a hand-solved CAPTCHA), and on a KAP board scrape that returns nothing while the WAF is up. --- src/trailing_edge/data/prices.py | 4 +- .../scrapers/ticaret_sicil/client.py | 9 +- .../scrapers/ticaret_sicil/parser.py | 4 +- src/trailing_edge/signals/cluster.py | 2 +- src/trailing_edge/signals/returns.py | 2 +- src/trailing_edge/storage/repository.py | 27 ++++- .../graph/test_upsert_management_roles.py | 103 ++++++++++++++++++ 7 files changed, 136 insertions(+), 15 deletions(-) create mode 100644 tests/integration/graph/test_upsert_management_roles.py diff --git a/src/trailing_edge/data/prices.py b/src/trailing_edge/data/prices.py index 84c55b6..d8c6a65 100644 --- a/src/trailing_edge/data/prices.py +++ b/src/trailing_edge/data/prices.py @@ -191,7 +191,7 @@ def _to_dec(v) -> Decimal | None: results[ticker] = 0 continue - insert_stmt = pg_insert(PriceHistory.__table__).values(values_list) + insert_stmt = pg_insert(PriceHistory).values(values_list) # The exchange bulletin is authoritative and yfinance must not overwrite # it. They are not interchangeable sources of the same number: # @@ -216,7 +216,7 @@ def _to_dec(v) -> Decimal | None: "low_try": insert_stmt.excluded.low_try, "volume": insert_stmt.excluded.volume, }, - where=PriceHistory.__table__.c.raw_close_try.is_(None), + where=PriceHistory.raw_close_try.is_(None), ) await session.execute(upsert_stmt) results[ticker] = len(values_list) diff --git a/src/trailing_edge/scrapers/ticaret_sicil/client.py b/src/trailing_edge/scrapers/ticaret_sicil/client.py index 1f6a371..7edfa44 100644 --- a/src/trailing_edge/scrapers/ticaret_sicil/client.py +++ b/src/trailing_edge/scrapers/ticaret_sicil/client.py @@ -15,6 +15,7 @@ from __future__ import annotations from types import TracebackType +from typing import Any from trailing_edge.core.config import get_config from trailing_edge.core.logging import get_logger @@ -37,10 +38,10 @@ def __init__(self) -> None: self._base_url: str = cfg.get("base_url", "https://www.ticaretsicil.gov.tr") self._ilan_path: str = cfg.get("ilan_path", "/view/hizlierisim/ilangoruntuleme.php") self._login_timeout = int(cfg.get("login_timeout_s", 180)) * 1000 - self._pw = None - self._browser = None - self._ctx = None - self._page = None + self._pw: Any = None + self._browser: Any = None + self._ctx: Any = None + self._page: Any = None async def __aenter__(self) -> "TsgClient": from playwright.async_api import async_playwright diff --git a/src/trailing_edge/scrapers/ticaret_sicil/parser.py b/src/trailing_edge/scrapers/ticaret_sicil/parser.py index 19c84dc..c8f712c 100644 --- a/src/trailing_edge/scrapers/ticaret_sicil/parser.py +++ b/src/trailing_edge/scrapers/ticaret_sicil/parser.py @@ -86,7 +86,9 @@ def parse_search_results(html: str) -> list[IlanRow]: pdf_link = tr.find("a", href=_PDF_GUID_RE) if not pdf_link: continue - guid_match = _PDF_GUID_RE.search(pdf_link["href"]) + # BeautifulSoup types an attribute as str | AttributeValueList (a multi-valued + # attribute like class=""); href is single-valued, but the type says otherwise. + guid_match = _PDF_GUID_RE.search(str(pdf_link["href"])) if not guid_match: continue diff --git a/src/trailing_edge/signals/cluster.py b/src/trailing_edge/signals/cluster.py index b50abc1..a57b0a9 100644 --- a/src/trailing_edge/signals/cluster.py +++ b/src/trailing_edge/signals/cluster.py @@ -203,7 +203,7 @@ async def detect_clusters(as_of_date: date | None = None) -> list[InsiderCluster total_value = sum(non_null) insert_stmt = ( - pg_insert(InsiderCluster.__table__) + pg_insert(InsiderCluster) .values( ticker=ticker, window_start=window_start, diff --git a/src/trailing_edge/signals/returns.py b/src/trailing_edge/signals/returns.py index 7f7b739..92b4f3b 100644 --- a/src/trailing_edge/signals/returns.py +++ b/src/trailing_edge/signals/returns.py @@ -140,7 +140,7 @@ async def calculate_outcomes( } stmt = ( - pg_insert(SignalOutcome.__table__) + pg_insert(SignalOutcome) .values(cluster_id=cluster.id, horizon_days=horizon, **values) .on_conflict_do_update( constraint="uq_outcome_cluster_horizon", set_=values diff --git a/src/trailing_edge/storage/repository.py b/src/trailing_edge/storage/repository.py index b38576a..2888832 100644 --- a/src/trailing_edge/storage/repository.py +++ b/src/trailing_edge/storage/repository.py @@ -2,8 +2,9 @@ from dataclasses import dataclass from datetime import date as date_type from datetime import datetime, timezone +from typing import Any, cast -from sqlalchemy import delete, select, text, update +from sqlalchemy import CursorResult, delete, select, text, update from sqlalchemy.dialects.postgresql import insert from sqlalchemy.ext.asyncio import AsyncSession @@ -56,9 +57,21 @@ async def upsert_disclosure(self, dto: KapDisclosureDTO) -> tuple[KapDisclosure, ) result = await self._s.execute(stmt) row = result.fetchone() - created = row.ingested_at == row.updated_at if row else False + if row is None: + # INSERT ... ON CONFLICT DO UPDATE ... RETURNING always yields a row, so this + # cannot happen while the statement above is what it looks like. It is checked + # rather than assumed because the alternative - the old code guarded `row` for + # None on the next line and then dereferenced it anyway - is to hand the caller + # a None model that reads as a successfully stored disclosure. + raise RuntimeError( + f"upsert of KAP disclosure {dto.kap_disclosure_id} returned no row" + ) + + created = row.ingested_at == row.updated_at model = await self._s.get(KapDisclosure, row.id) - return model, created # type: ignore[return-value] + if model is None: + raise RuntimeError(f"KAP disclosure {row.id} vanished between upsert and read") + return model, created async def upsert_transactions( self, disclosure_id: int, txs: list[KapInsiderTxDTO] @@ -93,7 +106,9 @@ async def upsert_transactions( .on_conflict_do_nothing(constraint="uq_insider_tx") ) result = await self._s.execute(stmt) - if result.rowcount == 1: + # rowcount lives on CursorResult, which is what execute() returns for a DML + # statement; Result is the wider declared type and does not carry it. + if cast("CursorResult[Any]", result).rowcount == 1: inserted += 1 else: skipped += 1 @@ -182,7 +197,7 @@ async def upsert_management_roles( for member in members: name_norm = normalize_name(member.full_name) await self._s.execute( - insert(Person.__table__) + insert(Person) .values(full_name=member.full_name, name_normalized=name_norm) .on_conflict_do_nothing(constraint="uq_person_name") ) @@ -207,7 +222,7 @@ async def upsert_management_roles( ) await self._s.execute( - insert(PersonCompanyRole.__table__).values( + insert(PersonCompanyRole).values( person_id=person_id, company_id=company_id, role=member.role, diff --git a/tests/integration/graph/test_upsert_management_roles.py b/tests/integration/graph/test_upsert_management_roles.py new file mode 100644 index 0000000..fdaf6cc --- /dev/null +++ b/tests/integration/graph/test_upsert_management_roles.py @@ -0,0 +1,103 @@ +"""GraphRepository.upsert_management_roles, without touching the network. + +The only existing coverage for this path (test_management_scrape_single_company_idempotent) +scrapes KAP live, so it cannot run while the WAF is blocking - and that left the two +INSERT statements inside it effectively untested. Both were rewritten from +`insert(Person.__table__)` to `insert(Person)` to satisfy the type checker, and a type +checker proves nothing about whether a row still lands. + +This exercises the same code against synthetic members: rows are written, and a re-run +with the same input is idempotent (the method DELETEs its own source rows before +inserting, because a NULL valid_from makes ON CONFLICT never fire). +""" +from datetime import date + +import pytest +from sqlalchemy import delete, func, select + +from trailing_edge.models.graph import Company, Person, PersonCompanyRole +from trailing_edge.scrapers.kap.types import BoardMemberDTO +from trailing_edge.storage.repository import GraphRepository + +_TICKER = "ZZBOARD" + +_MEMBERS = [ + BoardMemberDTO( + full_name="ZZ Test Chairman", + role="Yönetim Kurulu Başkanı", + role_type="BOARD_CHAIR", + is_independent=False, + valid_from=date(2019, 1, 1), + ), + BoardMemberDTO( + full_name="ZZ Test Member", + role="Yönetim Kurulu Üyesi", + role_type="BOARD_MEMBER", + is_independent=True, + valid_from=None, # the NULL that defeats ON CONFLICT + ), +] + + +async def _count_roles(session, company_id: int) -> int: + return ( + await session.execute( + select(func.count()) + .select_from(PersonCompanyRole) + .where( + PersonCompanyRole.company_id == company_id, + PersonCompanyRole.source == "KAP_YONETIM", + ) + ) + ).scalar_one() + + +async def _cleanup(session) -> None: + cid = ( + await session.execute(select(Company.id).where(Company.ticker == _TICKER)) + ).scalar() + if cid is not None: + await session.execute( + delete(PersonCompanyRole).where(PersonCompanyRole.company_id == cid) + ) + await session.execute(delete(Company).where(Company.id == cid)) + await session.execute( + delete(Person).where(Person.full_name.in_([m.full_name for m in _MEMBERS])) + ) + + +@pytest.mark.asyncio +async def test_roles_are_written_and_rewriting_them_is_idempotent(db_session): + await _cleanup(db_session) + + db_session.add(Company(ticker=_TICKER, company_name="ZZ Board Test AS")) + await db_session.flush() + company_id = ( + await db_session.execute(select(Company.id).where(Company.ticker == _TICKER)) + ).scalar_one() + + repo = GraphRepository(db_session) + + first = await repo.upsert_management_roles(_TICKER, _MEMBERS) + await db_session.flush() + count1 = await _count_roles(db_session, company_id) + + assert first.inserted == len(_MEMBERS), "insert(Person)/insert(PersonCompanyRole) wrote nothing" + assert count1 == len(_MEMBERS) + + # Re-running must not duplicate, even though valid_from is NULL on one member and a + # UNIQUE constraint therefore treats every NULL as distinct. + await repo.upsert_management_roles(_TICKER, _MEMBERS) + await db_session.flush() + count2 = await _count_roles(db_session, company_id) + + assert count2 == count1, "re-scrape duplicated the board" + + await _cleanup(db_session) + + +@pytest.mark.asyncio +async def test_an_unknown_ticker_is_a_no_op_not_a_crash(db_session): + repo = GraphRepository(db_session) + got = await repo.upsert_management_roles("ZZNOSUCH", _MEMBERS) + assert got.inserted == 0