diff --git a/Makefile b/Makefile index 2426a66f1..6b5c8bdb6 100644 --- a/Makefile +++ b/Makefile @@ -1300,6 +1300,22 @@ ifndef EXPECTED_CANONICAL_SHA256 endif @PYTHONDONTWRITEBYTECODE=1 python3 -m src.sec_fundamentals_patch_preview --sec-preview-path "$(SEC_PREVIEW)" --canonical-path "$(or $(CANONICAL_PATH),data/fundamentals.csv)" --expected-sec-preview-sha256 "$(EXPECTED_SEC_PREVIEW_SHA256)" --expected-canonical-sha256 "$(EXPECTED_CANONICAL_SHA256)" --repository-head "$(shell git rev-parse HEAD)" +.PHONY: sec-fundamentals-patch-apply +sec-fundamentals-patch-apply: +ifndef PATCH_PREVIEW + $(error PATCH_PREVIEW is required, for example: make sec-fundamentals-patch-apply PATCH_PREVIEW=/tmp/reviewed-patch.json) +endif +ifndef EXPECTED_PATCH_PREVIEW_SHA256 + $(error EXPECTED_PATCH_PREVIEW_SHA256 is required) +endif +ifndef EXPECTED_CANONICAL_SHA256 + $(error EXPECTED_CANONICAL_SHA256 is required) +endif +ifneq ($(CONFIRM_EXACT_FOUR_CELL_APPLY),1) + $(error CONFIRM_EXACT_FOUR_CELL_APPLY=1 is required) +endif + @PYTHONDONTWRITEBYTECODE=1 python3 -m src.sec_fundamentals_patch_apply --patch-preview-path "$(PATCH_PREVIEW)" --canonical-path "$(or $(CANONICAL_PATH),data/fundamentals.csv)" --expected-patch-preview-sha256 "$(EXPECTED_PATCH_PREVIEW_SHA256)" --expected-canonical-sha256 "$(EXPECTED_CANONICAL_SHA256)" --repository-head "$(shell git rev-parse HEAD)" --authorize-exact-four-cell-apply + demo-dashboard-render-smoke: @STOCK_RESEARCH_DATA_PROFILE=demo python3 -m src.dashboard_render_smoke diff --git a/data/fundamentals.csv b/data/fundamentals.csv index ef69020dd..5c5f375ab 100644 --- a/data/fundamentals.csv +++ b/data/fundamentals.csv @@ -5,7 +5,7 @@ TSLA,EV,Consumer Discretionary,48.0,-0.0293069915037363,0.040009701878157,0.2,94 SPY,Broad Market,ETF,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,, QQQ,Nasdaq Growth,ETF,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,, XLF,Financials,ETF,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,, -AMD,,,,2.2434570906877664,0.8134734471758304,,5329000000.0,2.65,6735000000.0,,1.2638393694877088,0.6931882154250328,,,5585000000.0,4098000000.0,,1630600639.0,,,,,,sec_companyfacts,2017-12-30,2488,10-K,2018-02-27,0000002488-18-000042,EBITDA was not staged because no direct SEC EBITDA fact was available.,ADVANCED MICRO DEVICES INC,,, +AMD,,,,2.2434570906877664,0.8134734471758304,,34639000000.0,2.65,6735000000.0,,1.2638393694877088,0.6931882154250328,,,5585000000.0,4098000000.0,,1630600639.0,,,,,,sec_companyfacts,2017-12-30,2488,10-K,2026-02-04,0000002488-18-000042,EBITDA was not staged because no direct SEC EBITDA fact was available.,ADVANCED MICRO DEVICES INC,,, AVGO,,,,3.117716768714201,0.2827609363008442,,20848000000.0,4.77,26914000000.0,,1.2909631619339983,1.2223714504988488,,,14174000000.0,66057000000.0,,4734668184.0,,,,,,sec_companyfacts,2018-11-04,1730168,10-K,2018-12-21,0001730168-18-000084,EBITDA was not staged because no direct SEC EBITDA fact was available.,Broadcom Inc.,,, COHR,,,,2.931500342667924,0.0425994611639342,,1158794000.0,-0.52,192764000.0,,0.1663488074670735,0.0829491695676712,,,1592730000.0,3193781000.0,,195639321.0,,,,,,sec_companyfacts,2018-06-30,820318,10-K,2018-08-28,0001564590-18-022409,EBITDA was not staged because no direct SEC EBITDA fact was available.,COHERENT CORP.,,, CRDO,,,,1.263434730787169,0.1194734130845401,,436775000.0,0.29,29022000.0,,0.0664461106977276,0.0849957071718848,,,1220464000.0,,,184449940.0,,,,,,sec_companyfacts,2025-05-03,1807794,10-K,2025-07-02,0001628280-25-033813,Debt was unavailable from SEC Companyfacts. | EBITDA was not staged because no direct SEC EBITDA fact was available.,Credo Technology Group Holding Ltd,,, @@ -13,7 +13,7 @@ HOOD,,,,0.515757370382921,0.4209702660406886,,4473000000.0,2.05,1610000000.0,,0. LITE,,,,-0.0653730653730653,0.0640138408304498,,404600000.0,0.37,-104700000.0,,-0.2587740978744439,-0.4451309935739001,,,2617800000.0,6520400000.0,,77800000.0,,,,,,sec_companyfacts,2019-06-29,1633978,10-K,2019-08-27,0001633978-19-000069,EBITDA was not staged because no direct SEC EBITDA fact was available.,Lumentum Holdings Inc.,,, META,,,,0.2216703849824621,0.3008369574952977,,200966000000.0,23.49,46109000000.0,,0.2294368201586338,0.4143785515957923,,,23426000000.0,58748000000.0,,2196045588.0,,,,,,sec_companyfacts; sec_filing_document,2026-03-09,1326801,10-Q,2026-04-30,0001628280-26-028526,EBITDA was not staged because no direct SEC EBITDA fact was available. | Shares outstanding was unavailable from SEC Companyfacts. | Shares outstanding staged from explicit SEC filing document fact dei:EntityCommonStockSharesOutstanding.,"Meta Platforms, Inc.",,,2026-06-26T01:41:27+00:00 MU,,,,2.897781197896627,0.2809713401994011,,30391000000.0,7.59,1668000000.0,,0.0548846698035602,0.3214767529860814,,,13908000000.0,3624000000.0,,1127734051.0,,,,,,sec_companyfacts,2018-08-30,723125,10-K,2018-10-15,0000723125-18-000092,EBITDA was not staged because no direct SEC EBITDA fact was available.,MICRON TECHNOLOGY INC,,, -AAPL,,,,3.9862949403923778,0.42173233682863,,265595000000.0,7.46,98767000000.0,,0.3718707053973155,0.5009506956079746,,,45572000000.0,82714000000.0,,14687356000.0,,,,,,sec_companyfacts,2018-09-29,320193,10-K,2018-11-05,0000320193-18-000145,EBITDA was not staged because no direct SEC EBITDA fact was available.,Apple Inc.,,, +AAPL,,,,3.9862949403923778,0.42173233682863,,416161000000.0,7.46,98767000000.0,,0.3718707053973155,0.5009506956079746,,,45572000000.0,82714000000.0,,14687356000.0,,,,,,sec_companyfacts,2018-09-29,320193,10-K,2025-10-31,0000320193-18-000145,EBITDA was not staged because no direct SEC EBITDA fact was available.,Apple Inc.,,, GOOGL,,,,0.1509008108154437,0.3280987796522654,,402836000000.0,10.81,73266000000.0,,0.1818755026859565,0.3203263859238002,,,38063000000.0,79499000000.0,,5824000000.0,,,,,,sec_companyfacts; sec_filing_document,2026-03-16,1652044,10-Q,2026-04-30,0001652044-26-000048,EBITDA was not staged because no direct SEC EBITDA fact was available. | Shares outstanding was unavailable from SEC Companyfacts. | Shares outstanding staged from explicit SEC filing document fact dei:EntityCommonStockSharesOutstanding.,Alphabet Inc.,,,2026-06-26T01:36:19+00:00 A,,,,0.0642551166111375,0.2913685152057245,,4472000000.0,4.57,1152000000.0,,0.257602862254025,0.3307245080500894,,,1758000000.0,3034000000.0,,282602317.0,,,,,,sec_companyfacts,2017-10-31,1090872,10-K,2017-12-21,0001090872-17-000018,EBITDA was not staged because no direct SEC EBITDA fact was available.,"AGILENT TECHNOLOGIES, INC.",,, ABBV,,,,0.0856676252352043,0.0690974493132766,,61160000000.0,2.36,17816000000.0,,0.2913015042511445,0.2464846304774362,,,9391000000.0,0.0,,1766792821.0,,,,,,sec_companyfacts,2025-12-31,1551152,10-K,2026-02-20,0001551152-26-000008,EBITDA was not staged because no direct SEC EBITDA fact was available.,AbbVie Inc.,,, diff --git a/data/reviewed_batch_proofs.csv b/data/reviewed_batch_proofs.csv index 0a1b76892..249fd80ea 100644 --- a/data/reviewed_batch_proofs.csv +++ b/data/reviewed_batch_proofs.csv @@ -1,4 +1,5 @@ batch_id,review_date,reviewer,lane,scope,tickers,command_run,validation_result,preview_result,apply_result,pre_run_readiness_snapshot,post_run_readiness_snapshot,changed_readiness_counts,changed_tickers,source_files,generated_artifacts_reviewed,final_outcome,notes +RB-20260820-SEC-DIRECT-001,2026-08-20,owner-authorized Codex review,fundamentals,exact SEC direct-field canonical apply,"AAPL,AMD",hash-guarded exact four-cell apply,make validate-data passed; affected tests passed; Research render smoke passed,inspection-only patch preview sha256=8d0da6bdd2dd52a938adcd1e882031fd3f278d3fc245151943f6f0d2c0e91553; projection identity=4749c18918220d53ca2d2dc27492bd81cec3f83dc11fb85d71848e9b899b616a,applied exact four cells; receipt sha256=af7d95169fdf568bfee93cc2f90dc94ce15b28326dbb3ab1c372f69715aaa944,saved readiness stale; no materialization,saved readiness stale; unchanged,no readiness counts changed,"AAPL,AMD",data/fundamentals.csv,only the exact reviewed canonical fundamentals artifact retained; all other generated churn excluded,human_reviewed_supported,"{""apply_receipt_sha256"":""af7d95169fdf568bfee93cc2f90dc94ce15b28326dbb3ab1c372f69715aaa944"",""canonical_after_sha256"":""6cd354f2cf2a2733224ec3a1ac2cbd70a0374cee5c55d18f7e680716c5aefc99"",""canonical_before_sha256"":""1b27ac9b31f9316177dfe060e21a90111d32c632104801ba4decae741ec0c418"",""changed_cells"":[""AMD:revenue"",""AMD:sec_filed_date"",""AAPL:revenue"",""AAPL:sec_filed_date""],""patch_preview_sha256"":""8d0da6bdd2dd52a938adcd1e882031fd3f278d3fc245151943f6f0d2c0e91553"",""readiness_mutated"":false}" RB-20260613T012355Z,2026-06-12,local reviewer,peers,top 10 peer reviewed batch packet,"AAPL,MSFT,CRDO,QQQ,SMH,AACB,AACI,AACO,AAPG,AARD",make coverage-frontier TOP_N=10 && make reviewed-batch LANE=peers TOP_N=10 && make peer-mapping-queue TOP_N=25,not_run_read_only,not_run_read_only,not_run_no_source_proof,peer-ready 9/3538; peer valuation inputs blocked,peer queue still shows 3512 missing source-backed mapping rows and 15 mapped rows waiting on peer valuation inputs,none; no data applied,none,outputs/reviewed_batch_packet.md; outputs/reviewed_batch_packet.csv; data/reports/peer_unlock_worklist.csv,kept reviewed packet and batch proof ledger; broad generated CSV churn excluded,skipped,Read-only peer lane pilot. Source proof was not applied; peer mapping and peer valuation inputs remain review-gated. RB-20260613T014434Z,2026-06-12,local reviewer,prices,top 10 price reviewed batch packet plus 3500-candidate dry run,"ARCT,ARDX,ARE,ARQT,ARRY,ARVN,ARW,ARWR,ASAN,ASB",make reviewed-batch LANE=prices TOP_N=10 && make price-refresh-loop DRY_RUN=1 MAX_CANDIDATES=3500 TOP_N=100 PROVIDER=yahoo && make reviewed-batch-compare LANE=prices BATCH_ID=RB-20260613T014434Z REVIEW_DATE=2026-06-12 TOP_N=10,not_run_dry_run_only,dry_run_reviewed_35_capped_batches_no_provider_call,not_run_no_data_changed,missing prior snapshot; make reviewed-batch-compare reported 0 prior rows,current readiness report has 3538 rows; no real refresh was run,not available; comparison blocked by missing prior snapshot,none,outputs/reviewed_batch_packet.md; outputs/reviewed_batch_packet.csv; dry-run terminal output; data/reports/ticker_readiness_report.csv,kept reviewed packet and batch proof ledger; broad generated CSV churn excluded,skipped,"Dry-run-only price pilot. Planned 35 capped 100-ticker batches for up to 3500 missing-price candidates; no provider call, import, apply, or local data refresh was run." RB-20260614-PRICE-COVERAGE,2026-06-14,local reviewer,prices,capped Yahoo missing-price refresh across remaining broad-universe price queue,capped_missing_price_queue; 3289 changed tickers; BF.B and BRK.B still blocked,make readiness-snapshot && make price-refresh-loop MAX_CANDIDATES=3500 TOP_N=100 PROVIDER=yahoo SLEEP_SECONDS=30 && make readiness && make status,make price-validate returned no_staged_file because no manual import file was used,make price-preview returned no_staged_file because direct capped provider refresh was used,price-refresh-loop completed capped provider refresh; no imports-apply path used,3538 rows from data/reports/ticker_readiness_report.previous.csv,3538 rows from data/reports/ticker_readiness_report.csv,overall_readiness_state blocked 3273->2 partial 265->3536; price_ready not_ready 3273->2 ready 265->3536; dcf_ready ready 23->27; peer_ready ready 9->26,"3289 changed; sample A,AAL,AAME,AAON,ABBV,ABEO,ABSI,ABT,ABTC,ABUS",data/prices.csv; outputs/price_update_status.csv; data/price_coverage_report.csv; data/reports/ticker_readiness_report.previous.csv; data/reports/ticker_readiness_report.csv,make diff-hygiene reviewed; broad generated CSV churn excluded from public commit by default,supported,"Real capped research-grade Yahoo price refresh. Price coverage now 3536/3538; BF.B and BRK.B remain provider/symbol blocked. No broker integration, order routing, auto-trading, or investment advice." diff --git a/scripts/diff_hygiene.py b/scripts/diff_hygiene.py index 72baba450..511c737ce 100644 --- a/scripts/diff_hygiene.py +++ b/scripts/diff_hygiene.py @@ -4,6 +4,7 @@ import argparse import csv +import hashlib import json import subprocess from shlex import quote @@ -173,6 +174,21 @@ def resolve_commit(repo_root: Path, revision: str, *, label: str) -> str: return result.stdout.strip() +def resolve_merge_base( + repo_root: Path, base_revision: str, head_revision: str +) -> str: + base_sha = resolve_commit(repo_root, base_revision, label="range base") + head_sha = resolve_commit(repo_root, head_revision, label="range head") + result = subprocess.run( + ["git", "merge-base", base_sha, head_sha], + cwd=repo_root, + check=True, + capture_output=True, + text=True, + ) + return result.stdout.strip() + + def load_range_status( repo_root: Path, base_revision: str, head_revision: str ) -> list[StatusEntry]: @@ -210,6 +226,43 @@ def load_staged_added_lines(repo_root: Path, path: str) -> list[str]: return rows +def load_range_added_lines( + repo_root: Path, base_revision: str, head_revision: str, path: str +) -> list[str]: + base_sha = resolve_commit(repo_root, base_revision, label="range base") + head_sha = resolve_commit(repo_root, head_revision, label="range head") + result = subprocess.run( + [ + "git", + "diff", + "--unified=0", + f"{base_sha}...{head_sha}", + "--", + path, + ], + cwd=repo_root, + check=True, + capture_output=True, + text=True, + ) + return [ + line[1:] + for line in result.stdout.splitlines() + if line.startswith("+") and not line.startswith("+++") + ] + + +def load_revision_file(repo_root: Path, revision: str, path: str) -> bytes: + commit = resolve_commit(repo_root, revision, label="range revision") + result = subprocess.run( + ["git", "show", f"{commit}:{path}"], + cwd=repo_root, + check=True, + capture_output=True, + ) + return result.stdout + + def load_branch_status(repo_root: Path) -> str: result = subprocess.run( ["git", "status", "--short", "--branch"], @@ -564,10 +617,178 @@ def range_hygiene_has_blockers(entries: list[StatusEntry]) -> bool: return bool(group_entries(entries)["generated_csv_churn"]) +def _csv_changed_cells(before: bytes, after: bytes) -> tuple[str, ...] | None: + try: + before_rows = list(csv.reader(before.decode("utf-8").splitlines())) + after_rows = list(csv.reader(after.decode("utf-8").splitlines())) + except (UnicodeDecodeError, csv.Error): + return None + if not before_rows or not after_rows or before_rows[0] != after_rows[0]: + return None + header = before_rows[0] + if not header or header[0].strip().lower() != "ticker": + return None + if len(before_rows) != len(after_rows): + return None + changed: list[str] = [] + for before_row, after_row in zip(before_rows[1:], after_rows[1:]): + if ( + len(before_row) != len(header) + or len(after_row) != len(header) + or before_row[0] != after_row[0] + ): + return None + ticker = before_row[0].strip().upper() + if not ticker: + return None + for index, column in enumerate(header[1:], start=1): + if before_row[index] != after_row[index]: + changed.append(f"{ticker}:{column.strip()}") + return tuple(changed) + + +def _is_sha256(value: object) -> bool: + text = str(value or "").strip().lower() + return len(text) == 64 and all(character in "0123456789abcdef" for character in text) + + +def _range_fundamentals_proof_matches( + repo_root: Path, + *, + base_revision: str, + head_revision: str, + path: str, +) -> bool: + try: + merge_base = resolve_merge_base(repo_root, base_revision, head_revision) + before = load_revision_file(repo_root, merge_base, path) + after = load_revision_file(repo_root, head_revision, path) + except subprocess.CalledProcessError: + return False + changed_cells = _csv_changed_cells(before, after) + if not changed_cells: + return False + changed_tickers = {cell.split(":", 1)[0] for cell in changed_cells} + before_sha256 = hashlib.sha256(before).hexdigest() + after_sha256 = hashlib.sha256(after).hexdigest() + proof_lines = load_range_added_lines( + repo_root, + base_revision, + head_revision, + "data/reviewed_batch_proofs.csv", + ) + for line in proof_lines: + try: + parts = next(csv.reader([line])) + except csv.Error: + continue + if len(parts) < 18: + continue + lane = parts[3].strip().lower() + proof_ticker_tokens = [ + token.strip().upper() + for token in parts[5].replace(";", ",").split(",") + if token.strip() + ] + applied_ticker_tokens = [ + token.strip().upper() + for token in parts[13].replace(";", ",").split(",") + if token.strip() + ] + proof_tickers = set(proof_ticker_tokens) + applied_tickers = set(applied_ticker_tokens) + source_paths = { + token.strip() for token in parts[14].split(";") if token.strip() + } + outcome = parts[-2].strip().lower() + if ( + lane != "fundamentals" + or outcome not in SUPPORTED_REVIEW_OUTCOMES + or len(proof_ticker_tokens) != len(proof_tickers) + or len(applied_ticker_tokens) != len(applied_tickers) + or proof_tickers != changed_tickers + or applied_tickers != changed_tickers + or path not in source_paths + or not parts[9].strip().lower().startswith("applied") + ): + continue + try: + binding = json.loads(parts[-1]) + except (json.JSONDecodeError, TypeError): + continue + if not isinstance(binding, dict): + continue + if ( + binding.get("canonical_before_sha256") != before_sha256 + or binding.get("canonical_after_sha256") != after_sha256 + or binding.get("readiness_mutated") is not False + or tuple(binding.get("changed_cells", ())) != changed_cells + or not _is_sha256(binding.get("patch_preview_sha256")) + or not _is_sha256(binding.get("apply_receipt_sha256")) + ): + continue + return True + return False + + +def range_hygiene_blockers_for_repo( + entries: list[StatusEntry], + repo_root: Path, + base_revision: str, + head_revision: str, +) -> dict[str, list[StatusEntry]]: + groups = group_entries(entries) + reviewed: list[StatusEntry] = [] + for entry in groups["generated_csv_churn"]: + if entry.path == "data/fundamentals.csv" and _range_fundamentals_proof_matches( + repo_root, + base_revision=base_revision, + head_revision=head_revision, + path=entry.path, + ): + reviewed.append(entry) + reviewed_paths = {entry.path for entry in reviewed} + return { + "generated_csv_churn": [ + entry + for entry in groups["generated_csv_churn"] + if entry.path not in reviewed_paths + ], + "reviewed_canonical_data": reviewed, + } + + +def range_hygiene_has_blockers_for_repo( + entries: list[StatusEntry], + repo_root: Path, + base_revision: str, + head_revision: str, +) -> bool: + return bool( + range_hygiene_blockers_for_repo( + entries, repo_root, base_revision, head_revision + )["generated_csv_churn"] + ) + + def build_range_check_report( - entries: list[StatusEntry], base_revision: str, head_revision: str + entries: list[StatusEntry], + base_revision: str, + head_revision: str, + *, + repo_root: Path | None = None, ) -> str: groups = group_entries(entries) + blockers = ( + range_hygiene_blockers_for_repo( + entries, repo_root, base_revision, head_revision + ) + if repo_root is not None + else { + "generated_csv_churn": groups["generated_csv_churn"], + "reviewed_canonical_data": [], + } + ) lines = [ "Pull Request Range Hygiene Check", "Read-only: this command inspects committed paths and does not modify the working tree.", @@ -578,13 +799,13 @@ def build_range_check_report( format_count_line("Generated CSV/JSON churn", groups["generated_csv_churn"]), format_count_line("Manual-review paths", groups["review_manually"]), ] - if range_hygiene_has_blockers(entries): + if blockers["generated_csv_churn"]: lines.extend( [ "", "Pull-request range hygiene failed.", "Generated CSV/JSON churn is present in the explicit base/head range:", - *format_paths(groups["generated_csv_churn"], limit=80), + *format_paths(blockers["generated_csv_churn"], limit=80), ] ) else: @@ -592,7 +813,15 @@ def build_range_check_report( [ "", "Pull-request range hygiene passed.", - "No generated CSV/JSON churn is present in the explicit base/head range.", + "No unreviewed generated CSV/JSON churn is present in the explicit base/head range.", + ] + ) + if blockers["reviewed_canonical_data"]: + lines.extend( + [ + "", + "Reviewed canonical data accepted by proof ledger:", + *format_paths(blockers["reviewed_canonical_data"], limit=80), ] ) if groups["review_manually"]: @@ -1275,8 +1504,21 @@ def main() -> int: entries = load_range_status(repo_root, args.range_base, args.range_head) except ValueError as exc: parser.error(str(exc)) - print(build_range_check_report(entries, args.range_base, args.range_head)) - return 1 if range_hygiene_has_blockers(entries) else 0 + print( + build_range_check_report( + entries, + args.range_base, + args.range_head, + repo_root=repo_root, + ) + ) + return ( + 1 + if range_hygiene_has_blockers_for_repo( + entries, repo_root, args.range_base, args.range_head + ) + else 0 + ) if args.staged_check: entries = load_staged_status(repo_root) print(build_staged_check_report(entries, repo_root)) diff --git a/src/company_workbench_html_browser_gate.py b/src/company_workbench_html_browser_gate.py index 42285d665..fe494e302 100644 --- a/src/company_workbench_html_browser_gate.py +++ b/src/company_workbench_html_browser_gate.py @@ -1078,16 +1078,80 @@ def repository_fingerprint(repo_root: Path) -> str: def _visible(page, selector: str, *, scroll_vertical: bool = True) -> bool: + if scroll_vertical: + expected_scroll_y = page.evaluate( + """selector => { + const node = document.querySelector(selector); + if (!node) return null; + delete document.documentElement.__srccVisibleSettlement; + const initial = node.getBoundingClientRect(); + const requested = Math.max( + 0, + window.scrollY + initial.top - + (window.innerHeight - initial.height) / 2 + ); + const maximum = Math.max( + 0, + document.documentElement.scrollHeight - window.innerHeight + ); + const expected = Math.min(requested, maximum); + window.scrollTo(window.scrollX, expected); + return expected; + }""", + selector, + ) + if expected_scroll_y is None: + return False + try: + page.wait_for_function( + """({selector, expected}) => { + const node = document.querySelector(selector); + if (!node) return false; + void node.offsetWidth; + const rect = node.getBoundingClientRect(); + const scrollSettled = Math.abs(window.scrollY - expected) <= 1; + const intersectsViewport = rect.right > 0 && + rect.left < window.innerWidth && + rect.bottom > 0 && + rect.top < window.innerHeight; + if (!scrollSettled || !intersectsViewport) { + delete document.documentElement.__srccVisibleSettlement; + return false; + } + const fingerprint = JSON.stringify([ + window.scrollX, + window.scrollY, + rect.left, + rect.right, + rect.top, + rect.bottom, + ]); + const previous = document.documentElement.__srccVisibleSettlement; + if (!previous || previous.fingerprint !== fingerprint) { + document.documentElement.__srccVisibleSettlement = { + fingerprint, + stableFrames: 0, + }; + return false; + } + previous.stableFrames += 1; + return previous.stableFrames >= 2; + }""", + arg={"selector": selector, "expected": expected_scroll_y}, + polling="raf", + timeout=2_000, + ) + except Exception: + return False + finally: + page.evaluate( + "delete document.documentElement.__srccVisibleSettlement" + ) return bool( page.evaluate( - """({selector, scrollVertical}) => { + """selector => { const node = document.querySelector(selector); if (!node) return false; - if (scrollVertical) { - const initial = node.getBoundingClientRect(); - const targetY = Math.max(0, window.scrollY + initial.top - (window.innerHeight - initial.height) / 2); - window.scrollTo(window.scrollX, targetY); - } let rect = node.getBoundingClientRect(); let left = Math.max(0, rect.left); let right = Math.min(window.innerWidth, rect.right); @@ -1120,7 +1184,7 @@ def _visible(page, selector: str, *, scroll_vertical: bool = True) -> bool: candidate => candidate === node || node.contains(candidate) )); }""", - {"selector": selector, "scrollVertical": scroll_vertical}, + selector, ) ) diff --git a/src/sec_fundamentals_patch_apply.py b/src/sec_fundamentals_patch_apply.py new file mode 100644 index 000000000..b7dec73bf --- /dev/null +++ b/src/sec_fundamentals_patch_apply.py @@ -0,0 +1,466 @@ +"""Hash-guarded apply path for exactly four reviewed SEC canonical cells.""" + +from __future__ import annotations + +import argparse +import csv +import hashlib +import io +import json +import os +import re +import subprocess +import tempfile +from dataclasses import dataclass +from decimal import Decimal, InvalidOperation +from pathlib import Path +from typing import Any, Mapping + + +_SHA256_PATTERN = re.compile(r"^[0-9a-f]{64}$") +_GIT_HEAD_PATTERN = re.compile(r"^[0-9a-f]{40}$") +_COORDINATES = ( + ("AAPL", "revenue", "revenue"), + ("AAPL", "filing_dates", "sec_filed_date"), + ("AMD", "revenue", "revenue"), + ("AMD", "filing_dates", "sec_filed_date"), +) +_CIKS = {"AAPL": "0000320193", "AMD": "0000002488"} +_SEC_URL_PREFIX = "https://data.sec.gov/api/xbrl/companyfacts/CIK" + + +@dataclass(frozen=True) +class SecFundamentalsPatchApply: + canonical_csv_bytes: bytes + receipt: dict[str, Any] + canonical_path: Path + repository_root: Path + authorized_repository_head: str + + +def _sha256(value: bytes) -> str: + return hashlib.sha256(value).hexdigest() + + +def _json_bytes(value: Any) -> bytes: + return json.dumps( + value, + sort_keys=True, + separators=(",", ":"), + ensure_ascii=True, + allow_nan=False, + ).encode("utf-8") + + +def _parse_packet(value: bytes) -> Mapping[str, Any]: + try: + packet = json.loads(value) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise ValueError("patch preview must be valid JSON") from exc + if not isinstance(packet, Mapping): + raise ValueError("patch preview must be a JSON object") + return packet + + +def _live_repository_head(repository_root: Path) -> str: + try: + return subprocess.run( + ["git", "rev-parse", "HEAD"], + cwd=repository_root, + check=True, + capture_output=True, + text=True, + ).stdout.strip() + except (OSError, subprocess.CalledProcessError) as exc: + raise ValueError("live repository HEAD could not be resolved") from exc + + +def _canonical_lines(value: bytes) -> tuple[list[str], list[bytes], list[list[str]]]: + try: + physical_lines = value.splitlines(keepends=True) + decoded = value.decode("utf-8") + except UnicodeDecodeError as exc: + raise ValueError("canonical CSV must be UTF-8") from exc + if not physical_lines or not decoded.endswith("\n"): + raise ValueError("canonical CSV must be newline-terminated") + if any(not line.endswith(b"\n") or b"\r" in line for line in physical_lines): + raise ValueError("canonical CSV must use one physical line per record") + rows: list[list[str]] = [] + for line in physical_lines: + try: + parsed = list(csv.reader(io.StringIO(line.decode("utf-8"), newline=""), strict=True)) + except (UnicodeDecodeError, csv.Error) as exc: + raise ValueError("canonical CSV must use one physical line per record") from exc + if len(parsed) != 1: + raise ValueError("canonical CSV must use one physical line per record") + rows.append(parsed[0]) + header = rows[0] + if not header or len(header) != len(set(header)) or "ticker" not in header: + raise ValueError("canonical CSV header is invalid") + if "currency" in header: + raise ValueError("canonical schema expansion is not authorized") + for column in ("revenue", "sec_filed_date"): + if column not in header: + raise ValueError(f"canonical schema is missing reviewed column: {column}") + if any(len(row) != len(header) for row in rows[1:]): + raise ValueError("canonical CSV must use one physical line per record") + return header, physical_lines, rows + + +def _values_equal(current: str, expected: Any) -> bool: + if isinstance(expected, bool): + return False + if isinstance(expected, (int, float)): + try: + return Decimal(current) == Decimal(str(expected)) + except InvalidOperation: + return False + return current == str(expected) + + +def _candidate_text(current: str, candidate: Any) -> str: + if isinstance(candidate, bool) or candidate is None: + raise ValueError("candidate value type is not authorized") + if isinstance(candidate, (int, float)): + try: + number = Decimal(str(candidate)) + except InvalidOperation as exc: + raise ValueError("candidate numeric value is invalid") from exc + if not number.is_finite(): + raise ValueError("candidate numeric value is invalid") + if current.endswith(".0") and number == number.to_integral_value(): + return f"{number.quantize(Decimal('1'))}.0" + return str(candidate) + if not isinstance(candidate, str) or "\n" in candidate or "\r" in candidate: + raise ValueError("candidate value type is not authorized") + return candidate + + +def _validate_proof(packet: Mapping[str, Any]) -> None: + if packet.get("status") != "inspection_only": + raise ValueError("patch preview must remain inspection-only") + if packet.get("canonical_apply_authorized") is not False: + raise ValueError("patch preview must remain inspection-only") + if packet.get("repository_writes") != []: + raise ValueError("patch preview must remain inspection-only") + if packet.get("source_rights_mutated") is not False: + raise ValueError("patch preview source rights contract is invalid") + if packet.get("readiness_mutated") is not False: + raise ValueError("patch preview readiness contract is invalid") + if packet.get("materialization_performed") is not False: + raise ValueError("patch preview materialization contract is invalid") + proof = packet.get("in_memory_projection_proof") + if not isinstance(proof, Mapping): + raise ValueError("patch preview projection proof is required") + exact = { + "schema_added_columns": [], + "schema_removed_columns": [], + "row_order_unchanged": True, + "full_row_replacement": False, + "staged_input_used": False, + "untouched_cells_unchanged": True, + "untouched_rows_unchanged": True, + "untouched_columns_unchanged": True, + } + if any(proof.get(key) != expected for key, expected in exact.items()): + raise ValueError("patch preview projection proof is unsafe") + for prefix in ("untouched_cells", "untouched_rows", "untouched_columns"): + before = proof.get(f"{prefix}_sha256_before") + after = proof.get(f"{prefix}_sha256_after") + if not isinstance(before, str) or before != after or not _SHA256_PATTERN.fullmatch(before): + raise ValueError("patch preview projection proof is unsafe") + if proof.get("column_count_before") != proof.get("column_count_after"): + raise ValueError("patch preview projection proof is unsafe") + if proof.get("row_count_before") != proof.get("row_count_after"): + raise ValueError("patch preview projection proof is unsafe") + + +def _validate_cell(cell: Mapping[str, Any], expected: tuple[str, str, str]) -> None: + ticker, field, column = expected + if (cell.get("ticker"), cell.get("field"), cell.get("canonical_column")) != expected: + raise ValueError("patch cells do not match the exact reviewed coordinates") + if cell.get("commercial_rights_approved") is not True or cell.get("source_rights_status") != "approved": + raise ValueError("patch cell rights are not approved") + if cell.get("field_scope_status") != "approved": + raise ValueError("patch cell field scope is not approved") + if cell.get("schema_status") != "existing_canonical": + raise ValueError("patch cell schema is not existing canonical") + unit = "date" if field == "filing_dates" else "USD" + if cell.get("unit") != unit: + raise ValueError("patch cell unit is invalid") + cik = _CIKS[ticker] + source_url = f"{_SEC_URL_PREFIX}{cik}.json" + if cell.get("source_url") != source_url: + raise ValueError("patch cell must use the reviewed official SEC source") + refs = cell.get("source_refs") + if not isinstance(refs, list) or not refs: + raise ValueError("patch cell provenance is required") + if cell.get("provenance_sha256") != _sha256(_json_bytes(refs)): + raise ValueError("patch cell provenance hash mismatch") + for ref in refs: + if not isinstance(ref, Mapping) or ref.get("source_url") != source_url: + raise ValueError("patch cell provenance source is invalid") + if ref.get("unit") != unit: + raise ValueError("patch cell provenance unit is invalid") + if ref.get("retrieval_timestamp") != cell.get("retrieval_timestamp"): + raise ValueError("patch cell provenance timestamp mismatch") + for key in ("accession", "concept", "filed", "form", "period_end", "taxonomy"): + if not ref.get(key): + raise ValueError(f"patch cell provenance is missing {key}") + if field == "filing_dates" and ref.get("underlying_fact_unit") != "USD": + raise ValueError("filing date must preserve underlying fact unit") + primary = refs[0] + projected = { + "period_start": primary.get("period_start"), + "period_end": primary.get("period_end"), + "filing_date": primary.get("filed"), + "accession": primary.get("accession"), + "concept": primary.get("concept"), + "taxonomy": primary.get("taxonomy"), + "form": primary.get("form"), + } + if any(cell.get(key) != value for key, value in projected.items()): + raise ValueError("patch cell provenance projection mismatch") + if column == "sec_filed_date" and cell.get("candidate_value") != cell.get("filing_date"): + raise ValueError("filing date candidate does not match SEC provenance") + + +def build_sec_fundamentals_patch_apply( + patch_preview_bytes: bytes, + canonical_csv_bytes: bytes, + *, + canonical_path: str, + expected_patch_preview_sha256: str, + expected_canonical_sha256: str, + repository_head: str, + repository_root: Path, + authorization_confirmed: bool, +) -> SecFundamentalsPatchApply: + if authorization_confirmed is not True: + raise ValueError("explicit authorization for the exact four-cell apply is required") + if not _SHA256_PATTERN.fullmatch(expected_patch_preview_sha256): + raise ValueError("expected patch preview SHA-256 is invalid") + if not _SHA256_PATTERN.fullmatch(expected_canonical_sha256): + raise ValueError("expected canonical SHA-256 is invalid") + if not _GIT_HEAD_PATTERN.fullmatch(repository_head): + raise ValueError("repository HEAD must be a full lowercase Git hash") + repository_root = repository_root.resolve() + live_head = _live_repository_head(repository_root) + if live_head != repository_head: + raise ValueError("live repository HEAD does not match the authorized preview") + patch_sha = _sha256(patch_preview_bytes) + canonical_sha = _sha256(canonical_csv_bytes) + if patch_sha != expected_patch_preview_sha256: + raise ValueError("patch preview hash precondition mismatch") + if canonical_sha != expected_canonical_sha256: + raise ValueError("canonical hash precondition mismatch") + packet = _parse_packet(patch_preview_bytes) + _validate_proof(packet) + preconditions = packet.get("preconditions") + if not isinstance(preconditions, Mapping): + raise ValueError("patch preview preconditions are required") + packet_canonical_path = preconditions.get("canonical_path") + if not isinstance(packet_canonical_path, str) or not packet_canonical_path: + raise ValueError("patch preview canonical path is not authorized") + if packet_canonical_path != canonical_path: + raise ValueError("canonical path precondition mismatch") + authorized_canonical_path = Path(canonical_path) + if not authorized_canonical_path.is_absolute(): + authorized_canonical_path = repository_root / authorized_canonical_path + authorized_canonical_path = authorized_canonical_path.resolve() + if preconditions.get("canonical_sha256") != canonical_sha: + raise ValueError("patch preview canonical hash precondition mismatch") + if preconditions.get("repository_head") != repository_head: + raise ValueError("patch preview repository HEAD precondition mismatch") + sec_sha = preconditions.get("sec_preview_sha256") + if not isinstance(sec_sha, str) or not _SHA256_PATTERN.fullmatch(sec_sha): + raise ValueError("patch preview SEC evidence hash is invalid") + identity = _sha256( + _json_bytes( + { + "canonical_sha256": canonical_sha, + "sec_preview_sha256": sec_sha, + "repository_head": repository_head, + "patch_coordinates": [[ticker, column] for ticker, _, column in _COORDINATES], + } + ) + ) + if packet.get("projection_identity") != identity: + raise ValueError("patch preview projection identity mismatch") + cells = packet.get("patch_cells") + if packet.get("changed_cell_count") != 4 or not isinstance(cells, list) or len(cells) != 4: + raise ValueError("patch preview must contain exactly four cells") + for cell, expected in zip(cells, _COORDINATES, strict=True): + if not isinstance(cell, Mapping): + raise ValueError("patch cell must be an object") + _validate_cell(cell, expected) + + header, physical_lines, parsed_rows = _canonical_lines(canonical_csv_bytes) + proof = packet["in_memory_projection_proof"] + if proof.get("column_count_before") != len(header) or proof.get("row_count_before") != len(parsed_rows) - 1: + raise ValueError("patch preview projection proof does not match canonical bytes") + ticker_index = header.index("ticker") + by_ticker: dict[str, tuple[int, list[str]]] = {} + for index, row in enumerate(parsed_rows[1:], start=1): + ticker = row[ticker_index].strip().upper() + if ticker in by_ticker: + raise ValueError(f"duplicate canonical ticker row: {ticker}") + by_ticker[ticker] = (index, row) + + changed_cells: list[dict[str, Any]] = [] + touched_lines: set[int] = set() + after_lines = list(physical_lines) + grouped: dict[str, list[Mapping[str, Any]]] = {} + for cell in cells: + grouped.setdefault(str(cell["ticker"]), []).append(cell) + for ticker in ("AAPL", "AMD"): + if ticker not in by_ticker: + raise ValueError(f"canonical ticker row is missing: {ticker}") + line_index, row = by_ticker[ticker] + projected = list(row) + for cell in grouped[ticker]: + column = str(cell["canonical_column"]) + column_index = header.index(column) + current = row[column_index] + if not _values_equal(current, cell.get("canonical_precondition")): + raise ValueError(f"canonical cell precondition mismatch: {ticker}:{column}") + candidate = _candidate_text(current, cell.get("candidate_value")) + if _values_equal(current, cell.get("candidate_value")): + raise ValueError(f"candidate does not change canonical cell: {ticker}:{column}") + projected[column_index] = candidate + changed_cells.append( + { + "ticker": ticker, + "canonical_column": column, + "before": current, + "after": candidate, + "provenance_sha256": cell["provenance_sha256"], + } + ) + output = io.StringIO(newline="") + csv.writer(output, lineterminator="\n").writerow(projected) + original = io.StringIO(newline="") + csv.writer(original, lineterminator="\n").writerow(row) + if original.getvalue().encode("utf-8") != physical_lines[line_index]: + raise ValueError("canonical target row is not in byte-preserving CSV form") + after_lines[line_index] = output.getvalue().encode("utf-8") + touched_lines.add(line_index) + + if [ + (cell["ticker"], cell["canonical_column"]) + for cell in changed_cells + ] != [(ticker, column) for ticker, _, column in _COORDINATES]: + raise ValueError("applied cells do not match the exact reviewed coordinates") + canonical_after = b"".join(after_lines) + for index, (before, after) in enumerate(zip(physical_lines, after_lines, strict=True)): + if index not in touched_lines and before != after: + raise ValueError("unrelated canonical bytes changed") + after_header, _, after_rows = _canonical_lines(canonical_after) + if after_header != header or [row[ticker_index] for row in after_rows[1:]] != [ + row[ticker_index] for row in parsed_rows[1:] + ]: + raise ValueError("canonical schema or row order changed") + receipt = { + "status": "authorized_exact_four_cell_projection", + "applied": False, + "authorization_confirmed": True, + "canonical_path": canonical_path, + "canonical_sha256_before": canonical_sha, + "canonical_sha256_after": _sha256(canonical_after), + "patch_preview_sha256": patch_sha, + "projection_identity": identity, + "repository_head": repository_head, + "changed_cell_count": 4, + "changed_cells": changed_cells, + "schema_unchanged": True, + "row_order_unchanged": True, + "untouched_bytes_unchanged": True, + "source_rights_mutated": False, + "readiness_mutated": False, + "readiness_materialized": False, + "repository_writes": [], + } + return SecFundamentalsPatchApply( + canonical_csv_bytes=canonical_after, + receipt=receipt, + canonical_path=authorized_canonical_path, + repository_root=repository_root, + authorized_repository_head=repository_head, + ) + + +def apply_sec_fundamentals_patch( + result: SecFundamentalsPatchApply, +) -> dict[str, Any]: + canonical_path = result.canonical_path + if _live_repository_head(result.repository_root) != result.authorized_repository_head: + raise ValueError("repository HEAD changed before materialization") + current = canonical_path.read_bytes() + if _sha256(current) != result.receipt["canonical_sha256_before"]: + raise ValueError("canonical hash changed before materialization") + temporary_path: Path | None = None + try: + with tempfile.NamedTemporaryFile( + mode="wb", + prefix=f".{canonical_path.name}.", + suffix=".tmp", + dir=canonical_path.parent, + delete=False, + ) as handle: + temporary_path = Path(handle.name) + handle.write(result.canonical_csv_bytes) + handle.flush() + os.fsync(handle.fileno()) + os.chmod(temporary_path, canonical_path.stat().st_mode & 0o777) + if _live_repository_head(result.repository_root) != result.authorized_repository_head: + raise ValueError("repository HEAD changed before materialization") + os.replace(temporary_path, canonical_path) + temporary_path = None + finally: + if temporary_path is not None: + temporary_path.unlink(missing_ok=True) + receipt = dict(result.receipt) + receipt["status"] = "authorized_exact_four_cell_apply_complete" + receipt["applied"] = True + receipt["repository_writes"] = [str(canonical_path)] + return receipt + + +def render_sec_fundamentals_patch_apply(result: Mapping[str, Any]) -> str: + return json.dumps(result, indent=2, sort_keys=True, allow_nan=False) + + +def _parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser(description="Apply exactly four reviewed SEC canonical cells.") + parser.add_argument("--patch-preview-path", type=Path, required=True) + parser.add_argument("--canonical-path", type=Path, default=Path("data/fundamentals.csv")) + parser.add_argument("--expected-patch-preview-sha256", required=True) + parser.add_argument("--expected-canonical-sha256", required=True) + parser.add_argument("--repository-head", required=True) + parser.add_argument("--authorize-exact-four-cell-apply", action="store_true") + return parser + + +def main(argv: list[str] | None = None) -> int: + parser = _parser() + args = parser.parse_args(argv) + try: + result = build_sec_fundamentals_patch_apply( + args.patch_preview_path.read_bytes(), + args.canonical_path.read_bytes(), + canonical_path=str(args.canonical_path), + expected_patch_preview_sha256=args.expected_patch_preview_sha256, + expected_canonical_sha256=args.expected_canonical_sha256, + repository_head=args.repository_head, + repository_root=Path.cwd(), + authorization_confirmed=args.authorize_exact_four_cell_apply, + ) + receipt = apply_sec_fundamentals_patch(result) + except (OSError, ValueError) as exc: + parser.error(str(exc)) + print(render_sec_fundamentals_patch_apply(receipt)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/test_company_workbench_html_browser_gate.py b/tests/test_company_workbench_html_browser_gate.py index 3ec31d745..0cb1cdaca 100644 --- a/tests/test_company_workbench_html_browser_gate.py +++ b/tests/test_company_workbench_html_browser_gate.py @@ -2519,6 +2519,56 @@ def observe(page): assert all(item["browser_version"] for item in evidence) +def test_visible_waits_for_post_scroll_geometry_to_settle(): + document = b""" + +
+
Target
+""" + + def observe(page): + page.evaluate( + """() => { + const target = document.querySelector('#target'); + const nativeRect = target.getBoundingClientRect.bind(target); + const nativeScrollTo = window.scrollTo.bind(window); + let geometrySettled = true; + target.getBoundingClientRect = () => { + const rect = nativeRect(); + if (geometrySettled) return rect; + return { + x: rect.x, + y: -2000, + left: rect.left, + right: rect.right, + top: -2000, + bottom: -1900, + width: rect.width, + height: rect.height, + toJSON: () => ({}), + }; + }; + window.scrollTo = (...args) => { + geometrySettled = false; + nativeScrollTo(...args); + let remainingFrames = 6; + const settleGeometry = () => { + remainingFrames -= 1; + if (remainingFrames === 0) { + geometrySettled = true; + return; + } + requestAnimationFrame(settleGeometry); + }; + requestAnimationFrame(settleGeometry); + }; + }""" + ) + return browser_gate._visible(page, "#target") + + assert _run_actual_browser_page(document, observe) is True + + def test_media_css_settlement_timeout_fails_closed_with_diagnostics_and_cleanup(): sabotaged = _append_test_css( _synthetic_brief("complete"), diff --git a/tests/test_diff_hygiene.py b/tests/test_diff_hygiene.py index 047e3b126..e193b7628 100644 --- a/tests/test_diff_hygiene.py +++ b/tests/test_diff_hygiene.py @@ -1,5 +1,6 @@ from __future__ import annotations +import csv import importlib.util import json import subprocess @@ -64,6 +65,214 @@ def git(*args: str) -> str: assert "failed" in report.lower() +def _reviewed_fundamentals_range( + tmp_path: Path, + *, + proof_tickers: str = "AAPL,AMD", + canonical_after_sha256: str = ( + "8cb2f4024aadd0a2171c1d0754093fc6efed82b922f99643daeca57472a4a1c6" + ), + add_unreviewed_generated_file: bool = False, + diverged_base: bool = False, +) -> tuple[object, Path, str, str]: + module = load_diff_hygiene_module() + repo = tmp_path / "repo" + repo.mkdir() + + def git(*args: str) -> str: + result = subprocess.run( + ["git", *args], cwd=repo, check=True, capture_output=True, text=True + ) + return result.stdout.strip() + + git("init", "-b", "main") + git("config", "user.name", "Range Test") + git("config", "user.email", "range@example.invalid") + data_dir = repo / "data" + data_dir.mkdir() + fundamentals = data_dir / "fundamentals.csv" + fundamentals.write_text( + "ticker,revenue,sec_filed_date\n" + "AAPL,265595000000.0,2018-11-05\n" + "AMD,5329000000.0,2018-02-27\n", + encoding="utf-8", + ) + proof = data_dir / "reviewed_batch_proofs.csv" + proof.write_text( + "batch_id,review_date,reviewer,lane,scope,tickers,command_run," + "validation_result,preview_result,apply_result," + "pre_run_readiness_snapshot,post_run_readiness_snapshot," + "changed_readiness_counts,changed_tickers,source_files," + "generated_artifacts_reviewed,final_outcome,notes\n", + encoding="utf-8", + ) + git("add", "data/fundamentals.csv", "data/reviewed_batch_proofs.csv") + git("commit", "-m", "base") + base_sha = git("rev-parse", "HEAD") + git("switch", "-c", "feature") + + fundamentals.write_text( + "ticker,revenue,sec_filed_date\n" + "AAPL,416161000000.0,2025-10-31\n" + "AMD,34639000000.0,2026-02-04\n", + encoding="utf-8", + ) + proof_binding = { + "apply_receipt_sha256": "2" * 64, + "canonical_after_sha256": canonical_after_sha256, + "canonical_before_sha256": ( + "a20d5d39714b519d9580871d4a1287744f8ce57506e6f085b384518ba5e73624" + ), + "changed_cells": [ + "AAPL:revenue", + "AAPL:sec_filed_date", + "AMD:revenue", + "AMD:sec_filed_date", + ], + "patch_preview_sha256": "1" * 64, + "readiness_mutated": False, + } + with proof.open("a", encoding="utf-8", newline="") as handle: + csv.writer(handle).writerow( + [ + "RB-SEC-4", + "2026-08-20", + "Owner", + "fundamentals", + "Exact four-cell SEC apply", + proof_tickers, + "hash-guarded apply", + "valid", + "reviewed preview", + "applied", + "stale", + "stale", + "none", + "AAPL,AMD", + "data/fundamentals.csv", + "exact reviewed artifact", + "human_reviewed_supported", + json.dumps(proof_binding, sort_keys=True, separators=(",", ":")), + ] + ) + git("add", "data/fundamentals.csv", "data/reviewed_batch_proofs.csv") + if add_unreviewed_generated_file: + readiness = data_dir / "reports" / "readiness.json" + readiness.parent.mkdir() + readiness.write_text("{}\n", encoding="utf-8") + git("add", "data/reports/readiness.json") + git("commit", "-m", "reviewed fundamentals") + head_sha = git("rev-parse", "HEAD") + if diverged_base: + git("switch", "main") + fundamentals.write_text( + "ticker,revenue,sec_filed_date\n" + "AAPL,265595000000.0,2018-11-05\n" + "AMD,5329000000.0,2018-02-27\n" + "GOOG,1000000000.0,2026-08-20\n", + encoding="utf-8", + ) + git("add", "data/fundamentals.csv") + git("commit", "-m", "advance base") + base_sha = git("rev-parse", "HEAD") + git("switch", "feature") + return module, repo, base_sha, head_sha + + +def test_pr_range_hygiene_accepts_proof_backed_fundamentals(tmp_path: Path): + module, repo, base_sha, head_sha = _reviewed_fundamentals_range(tmp_path) + entries = module.load_range_status(repo, base_sha, head_sha) + + report = module.build_range_check_report( + entries, base_sha, head_sha, repo_root=repo + ) + + assert module.range_hygiene_has_blockers_for_repo( + entries, repo, base_sha, head_sha + ) is False + assert "Pull-request range hygiene passed." in report + assert "Reviewed canonical data accepted by proof ledger:" in report + assert "data/fundamentals.csv" in report + + +def test_pr_range_hygiene_uses_merge_base_for_proof_binding(tmp_path: Path): + module, repo, base_sha, head_sha = _reviewed_fundamentals_range( + tmp_path, diverged_base=True + ) + entries = module.load_range_status(repo, base_sha, head_sha) + + assert module.range_hygiene_has_blockers_for_repo( + entries, repo, base_sha, head_sha + ) is False + + +def test_pr_range_hygiene_blocks_fundamentals_with_incomplete_ticker_proof( + tmp_path: Path, +): + module, repo, base_sha, head_sha = _reviewed_fundamentals_range( + tmp_path, proof_tickers="AAPL" + ) + entries = module.load_range_status(repo, base_sha, head_sha) + + assert module.range_hygiene_has_blockers_for_repo( + entries, repo, base_sha, head_sha + ) is True + + +def test_pr_range_hygiene_blocks_fundamentals_with_extra_ticker_proof( + tmp_path: Path, +): + module, repo, base_sha, head_sha = _reviewed_fundamentals_range( + tmp_path, proof_tickers="AAPL,AMD,NVDA" + ) + entries = module.load_range_status(repo, base_sha, head_sha) + + assert module.range_hygiene_has_blockers_for_repo( + entries, repo, base_sha, head_sha + ) is True + + +def test_pr_range_hygiene_blocks_duplicate_ticker_proof(tmp_path: Path): + module, repo, base_sha, head_sha = _reviewed_fundamentals_range( + tmp_path, proof_tickers="AAPL,AAPL,AMD" + ) + entries = module.load_range_status(repo, base_sha, head_sha) + + assert module.range_hygiene_has_blockers_for_repo( + entries, repo, base_sha, head_sha + ) is True + + +def test_pr_range_hygiene_blocks_fundamentals_with_wrong_canonical_hash( + tmp_path: Path, +): + module, repo, base_sha, head_sha = _reviewed_fundamentals_range( + tmp_path, canonical_after_sha256="0" * 64 + ) + entries = module.load_range_status(repo, base_sha, head_sha) + + assert module.range_hygiene_has_blockers_for_repo( + entries, repo, base_sha, head_sha + ) is True + + +def test_pr_range_hygiene_keeps_unreviewed_generated_files_blocked( + tmp_path: Path, +): + module, repo, base_sha, head_sha = _reviewed_fundamentals_range( + tmp_path, add_unreviewed_generated_file=True + ) + entries = module.load_range_status(repo, base_sha, head_sha) + report = module.build_range_check_report( + entries, base_sha, head_sha, repo_root=repo + ) + + assert module.range_hygiene_has_blockers_for_repo( + entries, repo, base_sha, head_sha + ) is True + assert "data/reports/readiness.json" in report + + def test_diff_hygiene_keeps_markdown_reports_as_reviewable_examples(): module = load_diff_hygiene_module() diff --git a/tests/test_launchers.py b/tests/test_launchers.py index 003cfaef6..e3fa35031 100644 --- a/tests/test_launchers.py +++ b/tests/test_launchers.py @@ -449,6 +449,57 @@ def test_sec_fundamentals_patch_preview_launcher_is_explicit_and_no_write(): assert forbidden not in result.stdout.lower() +def test_sec_fundamentals_patch_apply_launcher_requires_exact_hashes_and_confirmation(): + missing = subprocess.run( + ["make", "--dry-run", "sec-fundamentals-patch-apply"], + capture_output=True, + text=True, + check=False, + ) + assert missing.returncode != 0 + assert "PATCH_PREVIEW is required" in missing.stderr + + unconfirmed = subprocess.run( + [ + "make", + "--dry-run", + "sec-fundamentals-patch-apply", + "PATCH_PREVIEW=/tmp/reviewed-patch.json", + f"EXPECTED_PATCH_PREVIEW_SHA256={'1' * 64}", + f"EXPECTED_CANONICAL_SHA256={'2' * 64}", + ], + capture_output=True, + text=True, + check=False, + ) + assert unconfirmed.returncode != 0 + assert "CONFIRM_EXACT_FOUR_CELL_APPLY=1 is required" in unconfirmed.stderr + + result = subprocess.run( + [ + "make", + "--dry-run", + "sec-fundamentals-patch-apply", + "PATCH_PREVIEW=/tmp/reviewed-patch.json", + f"EXPECTED_PATCH_PREVIEW_SHA256={'1' * 64}", + f"EXPECTED_CANONICAL_SHA256={'2' * 64}", + "CONFIRM_EXACT_FOUR_CELL_APPLY=1", + ], + capture_output=True, + text=True, + check=False, + ) + assert result.returncode == 0 + assert "python3 -m src.sec_fundamentals_patch_apply" in result.stdout + assert '--patch-preview-path "/tmp/reviewed-patch.json"' in result.stdout + assert '--expected-patch-preview-sha256 "' + ("1" * 64) + '"' in result.stdout + assert '--expected-canonical-sha256 "' + ("2" * 64) + '"' in result.stdout + assert re.search(r'--repository-head "[0-9a-f]{40}"', result.stdout) + assert "--authorize-exact-four-cell-apply" in result.stdout + for forbidden in ("readiness-materialize", "imports-apply", "provider", "currency"): + assert forbidden not in result.stdout.lower() + + def test_reviewed_batch_packet_targets_forward_one_named_profile(): makefile = Path("Makefile").read_text(encoding="utf-8") for target in ("reviewed-batch", "fundamentals-batch-proof", "peer-batch-proof"): diff --git a/tests/test_sec_fundamentals_patch_apply.py b/tests/test_sec_fundamentals_patch_apply.py new file mode 100644 index 000000000..22bf8123c --- /dev/null +++ b/tests/test_sec_fundamentals_patch_apply.py @@ -0,0 +1,431 @@ +from __future__ import annotations + +import csv +import hashlib +import io +import json +import subprocess +from pathlib import Path + +import pytest + +from src.sec_fundamentals_patch_apply import ( + apply_sec_fundamentals_patch, + build_sec_fundamentals_patch_apply, + main, + render_sec_fundamentals_patch_apply, +) + + +_REPOSITORY_ROOT = Path(__file__).resolve().parents[1] +_HEAD = subprocess.run( + ["git", "rev-parse", "HEAD"], + cwd=_REPOSITORY_ROOT, + check=True, + capture_output=True, + text=True, +).stdout.strip() +_SEC_SHA = "3" * 64 + + +def _sha256(value: bytes) -> str: + return hashlib.sha256(value).hexdigest() + + +def _canonical() -> bytes: + return ( + "ticker,revenue,sec_filed_date,market_cap,source\n" + "AMD,5329000000.0,2018-02-27,250000000000,legacy\n" + "AAPL,265595000000.0,2018-11-05,3000000000000,legacy\n" + "NVDA,215938000000.0,2026-02-25,4000000000000,reviewed\n" + "MSFT,100,2025-01-01,999,untouched\n" + ).encode() + + +def _source_ref(ticker: str, *, unit: str) -> dict[str, object]: + cik = {"AAPL": "0000320193", "AMD": "0000002488"}[ticker] + filed = {"AAPL": "2025-10-31", "AMD": "2026-02-04"}[ticker] + accession = { + "AAPL": "0000320193-25-000079", + "AMD": "0000002488-26-000018", + }[ticker] + return { + "accession": accession, + "concept": "RevenueFromContractWithCustomerExcludingAssessedTax", + "filed": filed, + "fiscal_period": "FY", + "fiscal_year": 2025, + "form": "10-K", + "period_end": "2025-09-27" if ticker == "AAPL" else "2025-12-27", + "period_start": "2024-09-29" if ticker == "AAPL" else "2024-12-29", + "retrieval_timestamp": "2026-08-20T18:28:53Z", + "source_url": f"https://data.sec.gov/api/xbrl/companyfacts/CIK{cik}.json", + "taxonomy": "us-gaap", + "underlying_fact_unit": "USD" if unit == "date" else None, + "unit": unit, + } + + +def _cell( + ticker: str, + field: str, + column: str, + before: object, + after: object, +) -> dict[str, object]: + unit = "date" if field == "filing_dates" else "USD" + ref = _source_ref(ticker, unit=unit) + refs = [ref] + return { + "ticker": ticker, + "field": field, + "canonical_column": column, + "canonical_precondition": before, + "candidate_value": after, + "unit": unit, + "commercial_rights_approved": True, + "source_rights_status": "approved", + "field_scope_status": "approved", + "schema_status": "existing_canonical", + "retrieval_timestamp": "2026-08-20T18:28:53Z", + "source_url": ref["source_url"], + "period_start": ref["period_start"], + "period_end": ref["period_end"], + "filing_date": ref["filed"], + "accession": ref["accession"], + "concept": ref["concept"], + "taxonomy": ref["taxonomy"], + "form": ref["form"], + "source_refs": refs, + "provenance_sha256": _sha256( + json.dumps( + refs, + sort_keys=True, + separators=(",", ":"), + ensure_ascii=True, + allow_nan=False, + ).encode() + ), + } + + +def _packet( + canonical: bytes | None = None, + *, + canonical_path: str = "data/fundamentals.csv", + repository_head: str = _HEAD, +) -> bytes: + canonical = canonical or _canonical() + cells = [ + _cell("AAPL", "revenue", "revenue", 265_595_000_000, 416_161_000_000), + _cell("AAPL", "filing_dates", "sec_filed_date", "2018-11-05", "2025-10-31"), + _cell("AMD", "revenue", "revenue", 5_329_000_000, 34_639_000_000), + _cell("AMD", "filing_dates", "sec_filed_date", "2018-02-27", "2026-02-04"), + ] + coordinates = [[cell["ticker"], cell["canonical_column"]] for cell in cells] + identity = _sha256( + json.dumps( + { + "canonical_sha256": _sha256(canonical), + "sec_preview_sha256": _SEC_SHA, + "repository_head": repository_head, + "patch_coordinates": coordinates, + }, + sort_keys=True, + separators=(",", ":"), + ensure_ascii=True, + allow_nan=False, + ).encode() + ) + packet = { + "status": "inspection_only", + "canonical_apply_authorized": False, + "repository_writes": [], + "source_rights_mutated": False, + "readiness_mutated": False, + "materialization_performed": False, + "changed_cell_count": 4, + "patch_cells": cells, + "preconditions": { + "canonical_path": canonical_path, + "canonical_sha256": _sha256(canonical), + "sec_preview_sha256": _SEC_SHA, + "repository_head": repository_head, + }, + "projection_identity": identity, + "in_memory_projection_proof": { + "column_count_before": 5, + "column_count_after": 5, + "schema_added_columns": [], + "schema_removed_columns": [], + "row_count_before": 4, + "row_count_after": 4, + "row_order_unchanged": True, + "full_row_replacement": False, + "staged_input_used": False, + "untouched_cells_unchanged": True, + "untouched_cells_sha256_before": "a" * 64, + "untouched_cells_sha256_after": "a" * 64, + "untouched_rows_unchanged": True, + "untouched_rows_sha256_before": "b" * 64, + "untouched_rows_sha256_after": "b" * 64, + "untouched_columns_unchanged": True, + "untouched_columns_sha256_before": "c" * 64, + "untouched_columns_sha256_after": "c" * 64, + "projected_semantic_matrix_sha256": "d" * 64, + }, + "next_owner_decision": "Separately authorize only these cells.", + } + return (json.dumps(packet, indent=2, sort_keys=True) + "\n").encode() + + +def _build(packet: bytes | None = None, canonical: bytes | None = None, **kwargs): + canonical = canonical or _canonical() + canonical_path = kwargs.pop("canonical_path", "data/fundamentals.csv") + repository_head = kwargs.pop("repository_head", _HEAD) + repository_root = kwargs.pop("repository_root", _REPOSITORY_ROOT) + packet = packet or _packet( + canonical, + canonical_path=canonical_path, + repository_head=repository_head, + ) + return build_sec_fundamentals_patch_apply( + packet, + canonical, + canonical_path=canonical_path, + expected_patch_preview_sha256=_sha256(packet), + expected_canonical_sha256=_sha256(canonical), + repository_head=repository_head, + repository_root=repository_root, + authorization_confirmed=True, + **kwargs, + ) + + +def test_apply_projection_changes_exactly_four_cells_and_preserves_all_other_bytes(): + result = _build() + + assert result.receipt["applied"] is False + assert result.receipt["authorization_confirmed"] is True + assert result.receipt["changed_cell_count"] == 4 + assert result.receipt["repository_writes"] == [] + assert result.receipt["readiness_mutated"] is False + assert result.receipt["source_rights_mutated"] is False + assert result.receipt["schema_unchanged"] is True + assert result.receipt["row_order_unchanged"] is True + assert result.receipt["untouched_bytes_unchanged"] is True + assert [ + (cell["ticker"], cell["canonical_column"], cell["before"], cell["after"]) + for cell in result.receipt["changed_cells"] + ] == [ + ("AAPL", "revenue", "265595000000.0", "416161000000.0"), + ("AAPL", "sec_filed_date", "2018-11-05", "2025-10-31"), + ("AMD", "revenue", "5329000000.0", "34639000000.0"), + ("AMD", "sec_filed_date", "2018-02-27", "2026-02-04"), + ] + + before_lines = _canonical().splitlines(keepends=True) + after_lines = result.canonical_csv_bytes.splitlines(keepends=True) + assert len(before_lines) == len(after_lines) + assert before_lines[0] == after_lines[0] + assert before_lines[3:] == after_lines[3:] + assert after_lines[1] == b"AMD,34639000000.0,2026-02-04,250000000000,legacy\n" + assert after_lines[2] == b"AAPL,416161000000.0,2025-10-31,3000000000000,legacy\n" + + +def test_apply_projection_is_deterministic_and_preserves_csv_schema_and_order(): + first = _build() + second = _build() + + assert first.canonical_csv_bytes == second.canonical_csv_bytes + assert render_sec_fundamentals_patch_apply(first.receipt) == render_sec_fundamentals_patch_apply(second.receipt) + before = list(csv.reader(io.StringIO(_canonical().decode()))) + after = list(csv.reader(io.StringIO(first.canonical_csv_bytes.decode()))) + assert before[0] == after[0] + assert [row[0] for row in before[1:]] == [row[0] for row in after[1:]] + assert "currency" not in after[0] + + +@pytest.mark.parametrize( + ("mutation", "message"), + [ + (lambda packet: packet.update(changed_cell_count=5), "exactly four"), + (lambda packet: packet["patch_cells"].append(packet["patch_cells"][0]), "exactly four"), + (lambda packet: packet["patch_cells"][0].update(ticker="MSFT"), "exact reviewed coordinates"), + (lambda packet: packet["patch_cells"][0].update(canonical_column="market_cap"), "exact reviewed coordinates"), + (lambda packet: packet["patch_cells"][0].update(source_rights_status="review_required"), "rights"), + (lambda packet: packet["patch_cells"][0].update(field_scope_status="review_required"), "field scope"), + (lambda packet: packet["patch_cells"][0].update(schema_status="candidate_extra"), "schema"), + (lambda packet: packet["patch_cells"][0].update(provenance_sha256="0" * 64), "provenance"), + (lambda packet: packet["in_memory_projection_proof"].update(full_row_replacement=True), "projection proof"), + (lambda packet: packet["in_memory_projection_proof"].update(schema_added_columns=["currency"]), "projection proof"), + (lambda packet: packet.update(canonical_apply_authorized=True), "inspection-only"), + (lambda packet: packet.update(repository_writes=["data/fundamentals.csv"]), "inspection-only"), + ], +) +def test_apply_projection_rejects_any_scope_rights_provenance_or_proof_expansion(mutation, message): + packet = json.loads(_packet()) + mutation(packet) + mutated = (json.dumps(packet, indent=2, sort_keys=True) + "\n").encode() + + with pytest.raises(ValueError, match=message): + _build(packet=mutated) + + +def test_apply_projection_rejects_missing_authorization_or_any_hash_or_head_drift(): + packet = _packet() + canonical = _canonical() + + with pytest.raises(ValueError, match="explicit authorization"): + build_sec_fundamentals_patch_apply( + packet, + canonical, + canonical_path="data/fundamentals.csv", + expected_patch_preview_sha256=_sha256(packet), + expected_canonical_sha256=_sha256(canonical), + repository_head=_HEAD, + repository_root=_REPOSITORY_ROOT, + authorization_confirmed=False, + ) + with pytest.raises(ValueError, match="patch preview hash"): + build_sec_fundamentals_patch_apply( + packet, + canonical, + canonical_path="data/fundamentals.csv", + expected_patch_preview_sha256="0" * 64, + expected_canonical_sha256=_sha256(canonical), + repository_head=_HEAD, + repository_root=_REPOSITORY_ROOT, + authorization_confirmed=True, + ) + with pytest.raises(ValueError, match="canonical hash"): + _build(canonical=canonical.replace(b"999,untouched", b"998,untouched"), packet=packet) + with pytest.raises(ValueError, match="repository HEAD"): + _build(repository_head="f" * 40) + + +def test_apply_projection_rejects_caller_asserted_head_that_is_not_live_repository_head(): + stale_head = "f" * 40 + packet = _packet(repository_head=stale_head) + + with pytest.raises(ValueError, match="live repository HEAD"): + _build(packet=packet, repository_head=stale_head) + + +def test_apply_projection_and_writer_are_bound_to_one_exact_canonical_path(tmp_path): + alternate = tmp_path / "fundamentals.csv" + alternate.write_bytes(_canonical()) + packet = _packet(canonical_path="data/fundamentals.csv") + + with pytest.raises(ValueError, match="canonical path precondition"): + _build(packet=packet, canonical_path=str(alternate)) + + +def test_apply_writer_rejects_head_drift_between_projection_and_materialization(tmp_path): + repository = tmp_path / "repository" + repository.mkdir() + subprocess.run(["git", "init", "-q"], cwd=repository, check=True) + subprocess.run(["git", "config", "user.email", "test@example.com"], cwd=repository, check=True) + subprocess.run(["git", "config", "user.name", "Test"], cwd=repository, check=True) + canonical_path = repository / "data" / "fundamentals.csv" + canonical_path.parent.mkdir() + canonical_path.write_bytes(_canonical()) + subprocess.run(["git", "add", "data/fundamentals.csv"], cwd=repository, check=True) + subprocess.run(["git", "commit", "-qm", "baseline"], cwd=repository, check=True) + head = subprocess.run( + ["git", "rev-parse", "HEAD"], + cwd=repository, + check=True, + capture_output=True, + text=True, + ).stdout.strip() + packet = _packet( + canonical_path="data/fundamentals.csv", + repository_head=head, + ) + result = _build( + packet=packet, + canonical_path="data/fundamentals.csv", + repository_head=head, + repository_root=repository, + ) + (repository / "drift.txt").write_text("drift", encoding="utf-8") + subprocess.run(["git", "add", "drift.txt"], cwd=repository, check=True) + subprocess.run(["git", "commit", "-qm", "drift"], cwd=repository, check=True) + + with pytest.raises(ValueError, match="HEAD changed before materialization"): + apply_sec_fundamentals_patch(result) + + assert canonical_path.read_bytes() == _canonical() + + +def test_apply_projection_rejects_noncanonical_multiline_or_duplicate_ticker_csv(): + duplicate = _canonical().replace( + b"MSFT,100,2025-01-01,999,untouched\n", + b"AAPL,100,2025-01-01,999,untouched\n", + ) + with pytest.raises(ValueError, match="duplicate canonical ticker"): + _build(canonical=duplicate, packet=_packet(duplicate)) + + multiline = _canonical().replace(b"legacy\n", b'"leg\nacy"\n', 1) + with pytest.raises(ValueError, match="one physical line"): + _build(canonical=multiline, packet=_packet(multiline)) + + +def test_apply_function_writes_only_named_canonical_path_and_returns_auditable_receipt(tmp_path): + canonical_path = tmp_path / "fundamentals.csv" + canonical_path.write_bytes(_canonical()) + other = tmp_path / "keep.txt" + other.write_text("keep", encoding="utf-8") + result = _build( + packet=_packet(canonical_path=str(canonical_path)), + canonical_path=str(canonical_path), + ) + + receipt = apply_sec_fundamentals_patch(result) + + assert canonical_path.read_bytes() == result.canonical_csv_bytes + assert other.read_text(encoding="utf-8") == "keep" + assert sorted(path.name for path in tmp_path.iterdir()) == ["fundamentals.csv", "keep.txt"] + assert receipt["applied"] is True + assert receipt["repository_writes"] == [str(canonical_path)] + assert receipt["canonical_sha256_before"] == _sha256(_canonical()) + assert receipt["canonical_sha256_after"] == _sha256(result.canonical_csv_bytes) + assert result.canonical_path == canonical_path.resolve() + + +def test_cli_requires_exact_confirmation_and_second_apply_fails_stale_hash(tmp_path, capsys): + packet_path = tmp_path / "patch.json" + canonical_path = tmp_path / "fundamentals.csv" + packet = _packet(canonical_path=str(canonical_path)) + packet_path.write_bytes(packet) + canonical_path.write_bytes(_canonical()) + arguments = [ + "--patch-preview-path", + str(packet_path), + "--canonical-path", + str(canonical_path), + "--expected-patch-preview-sha256", + _sha256(packet), + "--expected-canonical-sha256", + _sha256(_canonical()), + "--repository-head", + _HEAD, + "--authorize-exact-four-cell-apply", + ] + + assert main(arguments) == 0 + receipt = json.loads(capsys.readouterr().out) + assert receipt["applied"] is True + assert receipt["changed_cell_count"] == 4 + with pytest.raises(SystemExit): + main(arguments) + + +def test_apply_receipt_contains_no_readiness_or_provider_materialization(): + text = render_sec_fundamentals_patch_apply(_build().receipt).lower() + + assert '"readiness_mutated": false' in text + assert '"source_rights_mutated": false' in text + for forbidden in ("yfinance", "yahoo", "stooq", "fmp_api_key", "alpha_vantage", "finnhub"): + assert forbidden not in text