Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
82 changes: 81 additions & 1 deletion data/quality/market.py
Original file line number Diff line number Diff line change
Expand Up @@ -139,6 +139,83 @@ def check_adj_factor(
)


def check_decreasing_adj_factor(
df: pd.DataFrame, *, dataset: str = "adj_factor", threshold: float = 0.01
) -> QualityFinding | None:
"""``adj_factor`` that FALLS materially from one date to the next, per symbol.

tushare's cumulative factor grows with each corporate action; it has no
mechanism to shrink. A material decrease therefore means the published series
is internally inconsistent, and because ``front_adjust`` multiplies raw prices
by it, the defect propagates straight into every return computed downstream.

This exists because a real one slipped through: ``920627.BJ`` oscillates
between two factor values, producing alternating -57.27% / +134.04% steps. It
was found by hand, after its +134% tail was briefly mistaken for a genuine
corporate-action distribution. A frame-level check would have named it
immediately. (It is a BSE name and sits in no index universe this project
evaluates, so nothing was restated — but that was luck, not detection.)

``threshold`` is RELATIVE (a fraction of the previous factor), never absolute:
factors range from ~1 to ~500 across the cache, so an absolute cut would both
miss real breaks in small-factor names and fire on rounding in large-factor
ones.

It is deliberately not zero, and 1% is not a round number picked for comfort —
it is where a full-cache sweep puts the only defensible cut. Sweeping every
negative step in all 5562 symbols (7898 events, 4479 symbols):

threshold events fired symbols fired that are NOT known-defective
0.01% 3653 2429 <- pure float noise
0.10% 179 2
0.50% 178 1 (603081.SH)
1.00% 177 0
2.00% 176 0

Two facts set the cut. Below 1% the check starts naming symbols whose status
cannot be decided: 603081.SH's worst step is -0.776% against a raw close move
of -1.810% where a genuine action needs +0.782%, which has the shape of a
defect — but at that magnitude an ordinary daily move swamps the implied one,
so the price test has no power and the finding would be uninterpretable. Above
1% a real event is missed: 1% fires on 177 events, 2% on only 176, and the
extra one belongs to a known-defective symbol.

1% also leaves the smallest CONFIRMED defect (000998.SZ, -2.636%) at 2.6x the
bar rather than 1.3x. Erring toward catching defects is right here: a false
positive costs one WARNING line in a report-only layer, a false negative lets
corrupted prices through silently.

WARNING rather than HARD, and the distinction from its sibling matters. Every
HARD check in this module is a zero-parameter algebraic impossibility
(duplicate keys, non-positive OHLC, high<low, adj_factor<=0). Every check with
a tunable threshold is WARNING. But note this one's WARNING does NOT mean
"might be legitimate" the way check_extreme_returns' does — a >50% close move
really can be a genuine split or a halt reopening, whereas a cumulative
adjustment factor has no legitimate way to shrink at all. The severity reflects
calibration risk at the threshold, not doubt about the underlying claim.

Never a filter — this layer only reports.
"""
d = reset_keys(df)
if not {"adj_factor", _SYMBOL, _DATE}.issubset(d.columns):
return None
d = d.sort_values([_SYMBOL, _DATE]).copy()
vals = pd.to_numeric(d["adj_factor"], errors="coerce")
d["adj_factor_change"] = vals.groupby(d[_SYMBOL]).pct_change().round(6)
mask = (d["adj_factor_change"] < -abs(threshold)).fillna(False)
if not mask.any():
return None
examples = row_examples(d, mask, [_SYMBOL, _DATE], extra="adj_factor_change")
return make_finding(
dataset, "decreasing_adj_factor", WARNING, count=int(mask.sum()),
examples=examples,
note=(
f"adj_factor fell by more than {abs(threshold):.2%} between consecutive "
f"dates; a cumulative adjustment factor should not decrease"
),
)


def check_extreme_returns(
df: pd.DataFrame, *, dataset: str = "market_daily", threshold: float = 0.5
) -> QualityFinding | None:
Expand Down Expand Up @@ -230,12 +307,15 @@ def run_market_checks(
def run_adj_factor_checks(
df: pd.DataFrame, *, dataset: str = "adj_factor"
) -> list[QualityFinding]:
"""Run the adj_factor frame checks (duplicate keys + positivity)."""
"""Run the adj_factor frame checks (duplicate keys, positivity, monotonicity)."""
findings = []
dup = check_duplicate_keys(df, dataset=dataset)
if dup is not None:
findings.append(dup)
adj = check_adj_factor(df, dataset=dataset)
if adj is not None:
findings.append(adj)
dec = check_decreasing_adj_factor(df, dataset=dataset)
if dec is not None:
findings.append(dec)
return findings
76 changes: 76 additions & 0 deletions tests/test_data_quality_market.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,7 @@
check_extreme_returns,
check_high_low_inversion,
check_missing_dates,
check_decreasing_adj_factor,
check_negative_volume_amount,
check_non_positive_ohlc,
run_adj_factor_checks,
Expand Down Expand Up @@ -143,3 +144,78 @@ def test_examples_bounded_to_five():
f = check_non_positive_ohlc(pd.DataFrame(rows))
assert f.count == 8
assert len(f.examples) == 5 # bounded


# --- decreasing adj_factor -------------------------------------------------- #
#
# A cumulative adjustment factor grows with each corporate action; it has no
# mechanism to shrink. front_adjust multiplies raw prices by it, so a decrease
# propagates into every downstream return. Fixtures below use the real defect.


def _adj(sym: str, values: list[float]) -> pd.DataFrame:
days = pd.date_range("2024-01-03", periods=len(values), freq="D")
return pd.DataFrame(
{"date": days, "symbol": [sym] * len(values), "adj_factor": values}
)


def test_decreasing_adj_factor_caught():
# The 920627.BJ signature: the factor oscillates between two values, giving
# alternating -57.27% / +134.04% steps. Two decreases in this window.
f = check_decreasing_adj_factor(_adj("920627.BJ", [2.3387, 1.0, 2.3387, 1.0]))
assert f is not None
assert f.severity == WARNING
assert f.count == 2
assert f.check == "decreasing_adj_factor"
findings = run_adj_factor_checks(_adj("920627.BJ", [2.3387, 1.0, 2.3387, 1.0]))
assert any(x.check == "decreasing_adj_factor" for x in findings)


def test_threshold_is_relative_not_absolute_at_small_factor_scale():
# A GENUINE -2.5% relative break on a small factor. An absolute-difference
# implementation sees only -0.005, well inside a 0.01 cut, and stays silent --
# a false negative. Every other fixture in this module sits at factor ~1-11,
# where a relative and an absolute cut happen to agree, so none of them can
# tell the two apart.
f = check_decreasing_adj_factor(_adj("000001.SZ", [0.200, 0.195]))
assert f is not None and f.count == 1


def test_threshold_is_relative_not_absolute_at_large_factor_scale():
# -0.006% relative: pure rounding on a large factor, must NOT flag. An
# absolute-difference implementation sees -0.03 and fires -- a false positive.
assert check_decreasing_adj_factor(_adj("600519.SH", [500.0, 499.97])) is None


def test_rounding_scale_negative_steps_are_not_flagged():
# Measured on the real CSI500 cache: ~18% of factor changes are negative and
# essentially all sit between -0.0008% and -0.08% -- tushare rounding on
# ex-dates, not corruption. Flagging them would bury the real defect in noise.
assert check_decreasing_adj_factor(_adj("000001.SZ", [11.267, 11.2669, 11.2589])) is None


def test_monotonic_increasing_adj_factor_is_clean():
assert check_decreasing_adj_factor(_adj("000001.SZ", [1.0, 1.05, 1.05, 1.30])) is None


def test_decreasing_adj_factor_does_not_bleed_across_symbols():
# Two symbols each individually non-decreasing, but whose concatenation steps
# DOWN at the boundary. A per-symbol check must see nothing.
df = pd.concat([_adj("000001.SZ", [5.0, 6.0]), _adj("000002.SZ", [1.0, 1.2])])
assert check_decreasing_adj_factor(df) is None


def test_decreasing_adj_factor_sorts_by_date_before_differencing():
# Rows arriving out of date order must not manufacture a phantom decrease:
# sorted, this series only ever rises.
df = _adj("000001.SZ", [1.0, 2.0, 3.0]).iloc[::-1].reset_index(drop=True)
assert check_decreasing_adj_factor(df) is None


def test_decreasing_adj_factor_reports_but_never_filters():
# Report-only: the input frame is not mutated and no rows are dropped.
df = _adj("920627.BJ", [2.3387, 1.0])
before = df.copy(deep=True)
check_decreasing_adj_factor(df)
pd.testing.assert_frame_equal(df, before)