mirosh111000

Мірошниченко_ГЙМ_ПР№7-8

Dec 10th, 2025 (edited)
82
0
Never
Not a member of Pastebin yet? Sign Up, it unlocks many cool features!
Python 13.40 KB | None | 0 0
  1. import numpy as np
  2. import pandas as pd
  3. import matplotlib.pyplot as plt
  4. import mplfinance as mpf
  5.  
  6. RANDOM_SEED = 42
  7.  
  8. N = 70
  9. A = 3
  10. B = 0.2
  11. C = 2
  12. D = 0.5
  13. F = 0.5
  14. G = 0.07
  15. SEASON_FREQ = 0.5236  
  16.  
  17. HIST_FILE = "DIXY_160101_161201.txt"
  18.  
  19. WINDOW_NOISE = 3  
  20. WINDOW_SEASON = 12
  21.  
  22. def generate_dynamic_series(
  23.     n: int = N,
  24.     a: float = A,
  25.     b: float = B,
  26.     c: float = C,
  27.     d: float = D,
  28.     f: float = F,
  29.     g: float = G,
  30.     season_freq: float = SEASON_FREQ,
  31.     random_seed: int = RANDOM_SEED,
  32. ) -> pd.DataFrame:
  33.     np.random.seed(random_seed)
  34.  
  35.     t = np.arange(1, n + 1)
  36.  
  37.     T = a + b * t
  38.  
  39.     season_raw = np.sin(np.sin(season_freq * t))
  40.  
  41.     S_add = c * season_raw
  42.     E_add = d * np.random.normal(loc=0.0, scale=1.0, size=n)
  43.     y_add = T + S_add + E_add
  44.  
  45.     S_mult = 1 + f * season_raw
  46.     E_mult = 1 + g * np.random.normal(loc=0.0, scale=1.0, size=n)
  47.     y_mult = T * S_mult * E_mult
  48.  
  49.     df = pd.DataFrame(
  50.         {
  51.             "t": t,
  52.             "T": T,
  53.             "S_add": S_add,
  54.             "E_add": E_add,
  55.             "y_add": y_add,
  56.             "S_mult": S_mult,
  57.             "E_mult": E_mult,
  58.             "y_mult": y_mult,
  59.         }
  60.     )
  61.  
  62.     return df
  63.  
  64.  
  65. def plot_dynamic_series(df: pd.DataFrame) -> None:
  66.     plt.figure(figsize=(10, 5))
  67.     plt.plot(df["t"], df["y_add"], label="y_add (адитивна модель)")
  68.     plt.title("Ряд динаміки (адитивна модель)")
  69.     plt.xlabel("t, міс.")
  70.     plt.ylabel("y_add")
  71.     plt.grid(True)
  72.     plt.legend()
  73.     plt.tight_layout()
  74.     plt.show()
  75.  
  76.     plt.figure(figsize=(10, 5))
  77.     plt.plot(df["t"], df["y_mult"], label="y_mult (мультиплікативна модель)", color="orange")
  78.     plt.title("Ряд динаміки (мультиплікативна модель)")
  79.     plt.xlabel("t, міс.")
  80.     plt.ylabel("y_mult")
  81.     plt.grid(True)
  82.     plt.legend()
  83.     plt.tight_layout()
  84.     plt.show()
  85.  
  86.  
  87. def load_historical_data(filename: str = HIST_FILE) -> pd.DataFrame:
  88.     df = pd.read_csv(
  89.         filename,
  90.         sep=",",
  91.         header=0,
  92.         names=["TICKER", "PER", "DATE", "TIME", "OPEN", "HIGH", "LOW", "CLOSE", "VOL"],
  93.     )
  94.  
  95.     df["DATE"] = pd.to_datetime(df["DATE"], format="%d/%m/%y", dayfirst=True)
  96.  
  97.     df = df.sort_values("DATE").set_index("DATE")
  98.  
  99.     return df
  100.  
  101.  
  102.  
  103. def plot_close_prices(df_hist: pd.DataFrame) -> None:
  104.     plt.figure(figsize=(10, 5))
  105.     plt.plot(df_hist.index, df_hist["CLOSE"], label="CLOSE")
  106.     plt.title("Ціна закриття акцій у часі")
  107.     plt.xlabel("Дата")
  108.     plt.ylabel("CLOSE")
  109.     plt.grid(True)
  110.     plt.legend()
  111.     plt.tight_layout()
  112.     plt.show()
  113.  
  114.  
  115.  
  116. def r2_score(y_true: np.ndarray, y_pred: np.ndarray) -> float:
  117.     y_true = np.asarray(y_true, dtype=float)
  118.     y_pred = np.asarray(y_pred, dtype=float)
  119.     ss_res = np.sum((y_true - y_pred) ** 2)
  120.     ss_tot = np.sum((y_true - np.mean(y_true)) ** 2)
  121.     return 1.0 - ss_res / ss_tot
  122.  
  123.  
  124.  
  125. def fit_trend_models(x: np.ndarray, y: np.ndarray, series_name: str = ""):
  126.     x = np.asarray(x, dtype=float)
  127.     y = np.asarray(y, dtype=float)
  128.  
  129.     models = {}
  130.  
  131.     coeffs_lin = np.polyfit(x, y, deg=1)
  132.     y_hat_lin = np.polyval(coeffs_lin, x)
  133.     models["linear"] = {
  134.         "y_hat": y_hat_lin,
  135.         "r2": r2_score(y, y_hat_lin),
  136.         "params": coeffs_lin,
  137.         "formula": f"y = {coeffs_lin[1]:.4f} + {coeffs_lin[0]:.4f} * t",
  138.     }
  139.  
  140.     if np.all(x > 0):
  141.         lx = np.log(x)
  142.         coeffs_log = np.polyfit(lx, y, deg=1)
  143.         y_hat_log = np.polyval(coeffs_log, lx)
  144.         models["logarithmic"] = {
  145.             "y_hat": y_hat_log,
  146.             "r2": r2_score(y, y_hat_log),
  147.             "params": coeffs_log,
  148.             "formula": f"y = {coeffs_log[1]:.4f} + {coeffs_log[0]:.4f} * ln(t)",
  149.         }
  150.  
  151.     coeffs_poly2 = np.polyfit(x, y, deg=2)
  152.     y_hat_poly2 = np.polyval(coeffs_poly2, x)
  153.     models["poly2"] = {
  154.         "y_hat": y_hat_poly2,
  155.         "r2": r2_score(y, y_hat_poly2),
  156.         "params": coeffs_poly2,
  157.         "formula": (
  158.             f"y = {coeffs_poly2[2]:.4f} + {coeffs_poly2[1]:.4f} * t + "
  159.             f"{coeffs_poly2[0]:.4f} * t^2"
  160.         ),
  161.     }
  162.  
  163.     if np.all(y > 0):
  164.         ly = np.log(y)
  165.         coeffs_exp = np.polyfit(x, ly, deg=1)
  166.         a = np.exp(coeffs_exp[1])
  167.         b = coeffs_exp[0]
  168.         y_hat_exp = a * np.exp(b * x)
  169.         models["exponential"] = {
  170.             "y_hat": y_hat_exp,
  171.             "r2": r2_score(y, y_hat_exp),
  172.             "params": (a, b),
  173.             "formula": f"y = {a:.4f} * exp({b:.4f} * t)",
  174.         }
  175.  
  176.     best_key = max(models.keys(), key=lambda k: models[k]["r2"])
  177.     best_model = models[best_key]
  178.  
  179.     print(f"\n=== Підбір тренда для {series_name} ===")
  180.     for name, m in models.items():
  181.         print(f"{name:>12}: R^2 = {m['r2']:.4f}, формула: {m['formula']}")
  182.  
  183.     print(f"Найкраща модель для {series_name}: {best_key} з R^2 = {best_model['r2']:.4f}")
  184.  
  185.     return best_key, best_model, models
  186.  
  187.  
  188. def plot_trend(x: np.ndarray, y: np.ndarray, trend_model, series_name: str = "") -> None:
  189.     x = np.asarray(x, dtype=float)
  190.     y = np.asarray(y, dtype=float)
  191.     y_hat = np.asarray(trend_model["y_hat"], dtype=float)
  192.  
  193.     plt.figure(figsize=(10, 5))
  194.     plt.plot(x, y, label=f"{series_name} (вихідні дані)")
  195.     plt.plot(x, y_hat, label=f"Тренд: {trend_model['formula']}", linewidth=2)
  196.     plt.title(f"{series_name}: вихідні дані та лінія тренда")
  197.     plt.xlabel("t")
  198.     plt.ylabel(series_name)
  199.     plt.grid(True)
  200.     plt.legend()
  201.     plt.tight_layout()
  202.     plt.show()
  203.  
  204.  
  205.  
  206. def linear_trend_two_ways(x: np.ndarray, y: np.ndarray):
  207.     x = np.asarray(x, dtype=float)
  208.     y = np.asarray(y, dtype=float)
  209.  
  210.     coeffs_polyfit = np.polyfit(x, y, deg=1)
  211.     b1_polyfit, b0_polyfit = coeffs_polyfit[0], coeffs_polyfit[1]
  212.  
  213.     X_design = np.column_stack([np.ones_like(x), x])  
  214.     beta, *_ = np.linalg.lstsq(X_design, y, rcond=None)
  215.     b0_lstsq, b1_lstsq = beta[0], beta[1]
  216.  
  217.     print("\n=== Лінійний тренд двома способами (аналог ЛИНЕЙН і 'Аналізу даних') ===")
  218.     print(f"polyfit:       y = {b0_polyfit:.4f} + {b1_polyfit:.4f} * t")
  219.     print(f"lstsq (МНК):   y = {b0_lstsq:.4f} + {b1_lstsq:.4f} * t")
  220.  
  221.     y_hat = b0_lstsq + b1_lstsq * x
  222.  
  223.     plt.figure(figsize=(10, 5))
  224.     plt.plot(x, y, label="CLOSE (вихідні дані)")
  225.     plt.plot(x, y_hat, label=f"Лінійний тренд (МНК): y = {b0_lstsq:.4f} + {b1_lstsq:.4f} * t", linewidth=2)
  226.     plt.title("Ціна закриття та лінійний тренд")
  227.     plt.xlabel("t (номер спостереження)")
  228.     plt.ylabel("CLOSE")
  229.     plt.grid(True)
  230.     plt.legend()
  231.     plt.tight_layout()
  232.     plt.show()
  233.  
  234.  
  235.  
  236. def moving_average_manual(values: np.ndarray, window: int) -> np.ndarray:
  237.     values = np.asarray(values, dtype=float)
  238.     n = len(values)
  239.     result = np.full(n, np.nan)
  240.  
  241.     if window < 1 or window > n:
  242.         return result
  243.  
  244.     half = window // 2
  245.     for i in range(half, n - half):
  246.         result[i] = np.mean(values[i - half : i + half + 1])
  247.  
  248.     return result
  249.  
  250.  
  251.  
  252. def moving_average_convolve(values: np.ndarray, window: int) -> np.ndarray:
  253.     values = np.asarray(values, dtype=float)
  254.     if window < 1:
  255.         return np.full_like(values, np.nan, dtype=float)
  256.     kernel = np.ones(window) / window
  257.     conv = np.convolve(values, kernel, mode="same")
  258.     return conv
  259.  
  260.  
  261.  
  262. def smooth_additive_series(
  263.     df_dyn: pd.DataFrame,
  264.     window_noise: int = WINDOW_NOISE,
  265.     window_season: int = WINDOW_SEASON,
  266. ) -> pd.DataFrame:
  267.     df = df_dyn.copy()
  268.     t = df["t"].to_numpy()
  269.     y = df["y_add"].to_numpy()
  270.  
  271.     ma_manual_short = moving_average_manual(y, window_noise)
  272.     ma_rolling_short = pd.Series(y).rolling(window_noise, center=True).mean().to_numpy()
  273.     ma_conv_short = moving_average_convolve(y, window_noise)
  274.  
  275.     ma_manual_long = moving_average_manual(y, window_season)
  276.     ma_rolling_long = pd.Series(y).rolling(window_season, center=True).mean().to_numpy()
  277.     ma_conv_long = moving_average_convolve(y, window_season)
  278.  
  279.     window_median = 5
  280.     med_series = pd.Series(y).rolling(window_median, center=True).median().to_numpy()
  281.  
  282.     df["MA_manual_short"] = ma_manual_short
  283.     df["MA_rolling_short"] = ma_rolling_short
  284.     df["MA_conv_short"] = ma_conv_short
  285.  
  286.     df["MA_manual_long"] = ma_manual_long
  287.     df["MA_rolling_long"] = ma_rolling_long
  288.     df["MA_conv_long"] = ma_conv_long
  289.  
  290.     df["Median_5"] = med_series
  291.  
  292.     plt.figure(figsize=(10, 5))
  293.     plt.plot(t, y, label="y_add (вихідний ряд)", alpha=0.5)
  294.     plt.plot(t, ma_manual_short, label=f"MA manual (вікно={window_noise})")
  295.     plt.plot(t, ma_rolling_short, label=f"MA rolling (вікно={window_noise})")
  296.     plt.plot(t, ma_conv_short, label=f"MA convolve (вікно={window_noise})")
  297.     plt.title("Згладжування y_add (видалення шуму, коротке вікно)")
  298.     plt.xlabel("t")
  299.     plt.ylabel("y_add")
  300.     plt.grid(True)
  301.     plt.legend()
  302.     plt.tight_layout()
  303.     plt.show()
  304.  
  305.     plt.figure(figsize=(10, 5))
  306.     plt.plot(t, y, label="y_add (вихідний ряд)", alpha=0.5)
  307.     plt.plot(t, ma_manual_long, label=f"MA manual (вікно={window_season})")
  308.     plt.plot(t, ma_rolling_long, label=f"MA rolling (вікно={window_season})")
  309.     plt.plot(t, ma_conv_long, label=f"MA convolve (вікно={window_season})")
  310.     plt.title("Згладжування y_add (видалення шуму та сезонності, довге вікно)")
  311.     plt.xlabel("t")
  312.     plt.ylabel("y_add")
  313.     plt.grid(True)
  314.     plt.legend()
  315.     plt.tight_layout()
  316.     plt.show()
  317.  
  318.     plt.figure(figsize=(10, 5))
  319.     plt.plot(t, y, label="y_add (вихідний ряд)", alpha=0.5)
  320.     plt.plot(t, med_series, label=f"Ковзна медіана (вікно={window_median})", linewidth=2)
  321.     plt.title("Ковзна медіана для y_add")
  322.     plt.xlabel("t")
  323.     plt.ylabel("y_add")
  324.     plt.grid(True)
  325.     plt.legend()
  326.     plt.tight_layout()
  327.     plt.show()
  328.  
  329.     return df
  330.  
  331.  
  332.  
  333. def build_ohlc_resampled(df_hist: pd.DataFrame):
  334.     ohlc_dict = {
  335.         "OPEN": "first",
  336.         "HIGH": "max",
  337.         "LOW": "min",
  338.         "CLOSE": "last",
  339.     }
  340.  
  341.     ohlc_month = df_hist.resample("M").apply(ohlc_dict).dropna()
  342.     ohlc_year = df_hist.resample("Y").apply(ohlc_dict).dropna()
  343.  
  344.     return ohlc_month, ohlc_year
  345.  
  346.  
  347.  
  348. def plot_ohlc_charts(df_hist: pd.DataFrame) -> None:
  349.  
  350.     ohlc_month, ohlc_year = build_ohlc_resampled(df_hist)
  351.  
  352.     for df_ohlc, name in [(ohlc_month, "Місячні"), (ohlc_year, "Річні")]:
  353.         data = df_ohlc.copy()
  354.         data.index.name = "Date"
  355.         data.rename(
  356.             columns={
  357.                 "OPEN": "Open",
  358.                 "HIGH": "High",
  359.                 "LOW": "Low",
  360.                 "CLOSE": "Close",
  361.             },
  362.             inplace=True,
  363.         )
  364.  
  365.         mpf.plot(
  366.             data,
  367.             type="candle",
  368.             title=f"Свічковий графік ({name} дані)",
  369.             style="yahoo",
  370.             volume=False,
  371.         )
  372.  
  373.         mpf.plot(
  374.             data,
  375.             type="ohlc",
  376.             title=f"Графік барів ({name} дані)",
  377.             style="yahoo",
  378.             volume=False,
  379.         )
  380.  
  381.  
  382.  
  383. def main():
  384.     print("=== Практична робота №7–8. Ряди динаміки. Варіант 2 ===")
  385.  
  386.     df_dyn = generate_dynamic_series()
  387.     print("\nПерші 5 рядків згенерованих даних:")
  388.     print(df_dyn.head())
  389.     plot_dynamic_series(df_dyn)
  390.  
  391.     try:
  392.         df_hist = load_historical_data()
  393.     except FileNotFoundError:
  394.         print(
  395.             f"\nФайл {HIST_FILE} не знайдено. "
  396.             f"Скопіюйте файл історичних даних у поточну папку і запустіть скрипт знову."
  397.         )
  398.         df_hist = None
  399.     except Exception as e:
  400.         print(f"\nПомилка при завантаженні історичних даних: {e}")
  401.         df_hist = None
  402.  
  403.     if df_hist is not None:
  404.         print("\nПерші 5 рядків історичних даних:")
  405.         print(df_hist[["OPEN", "HIGH", "LOW", "CLOSE", "VOL"]].head())
  406.         plot_close_prices(df_hist)
  407.  
  408.     t_dyn = df_dyn["t"].to_numpy()
  409.  
  410.     _, best_model_add, _ = fit_trend_models(t_dyn, df_dyn["y_add"], "y_add (адитивна)")
  411.     plot_trend(t_dyn, df_dyn["y_add"], best_model_add, "y_add (адитивна)")
  412.  
  413.     _, best_model_mult, _ = fit_trend_models(
  414.         t_dyn, df_dyn["y_mult"], "y_mult (мультиплікативна)"
  415.     )
  416.     plot_trend(t_dyn, df_dyn["y_mult"], best_model_mult, "y_mult (мультиплікативна)")
  417.  
  418.     if df_hist is not None:
  419.         y_close = df_hist["CLOSE"].to_numpy()
  420.         t_close = np.arange(1, len(y_close) + 1)
  421.         _, best_model_close, _ = fit_trend_models(
  422.             t_close, y_close, "CLOSE (історичні дані)"
  423.         )
  424.         plot_trend(t_close, y_close, best_model_close, "CLOSE (історичні дані)")
  425.  
  426.         linear_trend_two_ways(t_close, y_close)
  427.  
  428.     smooth_additive_series(df_dyn, window_noise=WINDOW_NOISE, window_season=WINDOW_SEASON)
  429.  
  430.     if df_hist is not None:
  431.         plot_ohlc_charts(df_hist)
  432.  
  433. if __name__ == "__main__":
  434.     main()
  435.  
Advertisement
Add Comment
Please, Sign In to add comment