Update exercice_fr.Rmd

parent 2ebb49c9
--- import os
title: "Analyse de l'incidence de la varicelle" import re
author: "Votre nom" import csv
date: "`r Sys.Date()`" from datetime import date
output: import pandas as pd
html_document:
toc: true DATA_PATH = r"C:\Users\nargi\Downloads\inc-25-PAY.csv"
---
if not os.path.exists(DATA_PATH):
```{r setup, include=FALSE} raise FileNotFoundError(f"File not found: {DATA_PATH}")
knitr::opts_chunk$set(echo = TRUE)
Préparation des données # ------------------------------------------------------------
# 1) Load Sentinelles CSV robustly (skip broken metadata / find real header)
Les données de la varicelle proviennent du Réseau Sentinelles. # ------------------------------------------------------------
Pour assurer la reproductibilité, nous conservons l’URL d’origine mais utilisons une copie locale des données. with open(DATA_PATH, "r", encoding="utf-8", errors="replace") as f:
# URL d'origine (traçabilité) lines = f.read().splitlines()
data_url <- "https://www.sentiweb.fr/datasets/all/inc-25-PAY.json" # Find header line: contains letters + separators and NOT just the metadata blob
header_idx = None
# Fichier local for i, ln in enumerate(lines):
if not ln.strip():
local_file <- "incidence-VARI-3.csv" continue
if ln.lstrip().startswith("#"):
# Télécharger uniquement si le fichier local n'existe pas continue
low = ln.lower()
if (!file.exists(local_file)) { if ((";" in ln) or ("," in ln) or ("\t" in ln) or ("|" in ln)) and re.search(r"[A-Za-z_]", ln):
download.file(data_url, destfile = local_file, mode = "wb") header_idx = i
} break
# Lecture du fichier CSV local (la première ligne est un commentaire) if header_idx is None:
raise ValueError("Could not find a header row (CSV table) in the file.")
data <- read.csv(local_file, skip = 1)
head(data) # detect delimiter from that header line
tail(data) header_line = lines[header_idx]
Conversion des numéros de semaine try:
dialect = csv.Sniffer().sniff(header_line, delimiters=[",", ";", "\t", "|"])
Les semaines sont au format ISO-8601. Nous les convertissons en dates sep = dialect.delimiter
(correspondant au lundi de chaque semaine). except Exception:
library(parsedate) sep = ";" # common in FR exports
convert_week <- function(w) { content = "\n".join(lines[header_idx:])
ws <- paste(w) df = pd.read_csv(pd.io.common.StringIO(content), sep=sep, engine="python", on_bad_lines="skip")
iso <- paste0(substr(ws, 1, 4), "-W", substr(ws, 5, 6))
as.character(parse_iso_8601(iso)) df = df.loc[:, ~df.columns.astype(str).str.match(r"^Unnamed")]
} if df.empty:
raise ValueError("Parsed dataframe is empty. The file may not contain tabular data.")
data$date <- as.Date(convert_week(data$week))
data <- data[order(data$date),] # ------------------------------------------------------------
Vérification : les dates doivent être espacées de 7 jours. # 2) Identify week column and incidence column
# ------------------------------------------------------------
Inspection des données def score_week_col(s: pd.Series) -> float:
x = s.dropna().astype(str).str.strip().head(200)
On utilise le taux d’incidence pour 100 000 habitants (inc100). if x.empty:
plot(data$date, data$inc100, type = "l", return 0.0
xlab = "Date", ylab = "Incidence hebdomadaire (pour 100 000)") patterns = [
Incidence annuelle (année épidémique) r"^\d{6}$", # YYYYWW
r"^\d{4}-?W\d{1,2}$", # YYYY-WW or YYYYWww
Conformément à l’énoncé, chaque période annuelle est définie r"^\d{4}s\d{1,2}$", # YYYYsWW
du 1er septembre de l’année N-1 au 1er septembre de l’année N. ]
pic_annuel <- function(annee) { return max(x.str.match(p, case=False).mean() for p in patterns)
debut <- paste0(annee - 1, "-09-01")
fin <- paste0(annee, "-09-01") week_col = max(df.columns, key=lambda c: score_week_col(df[c]))
semaines <- data$date > debut & data$date <= fin if score_week_col(df[week_col]) < 0.3:
sum(data$inc100[semaines], na.rm = TRUE) raise ValueError(f"Could not detect week column. Columns: {list(df.columns)}")
}
annees <- (as.integer(format(min(data$date), "%Y")) + 1) : # incidence column: prefer inc100 like in your R code, else pick best numeric column containing 'inc'
as.integer(format(max(data$date), "%Y")) lower_to_orig = {str(c).strip().lower(): c for c in df.columns}
inc_col = None
inc_annuelle <- data.frame( for k in ["inc100", "inc_100", "incidence100", "incidence_100"]:
annee = annees, if k in lower_to_orig:
incidence = sapply(annees, pic_annuel) inc_col = lower_to_orig[k]
break
def to_num(s: pd.Series) -> pd.Series:
return pd.to_numeric(s.astype(str).str.replace(",", ".", regex=False), errors="coerce")
if inc_col is None:
cands = []
for c in df.columns:
if c == week_col:
continue
if "inc" in str(c).lower():
coerced = to_num(df[c])
if coerced.notna().sum() > 0:
df[c] = coerced
cands.append(c)
if not cands:
# fallback: any numeric column
for c in df.columns:
if c == week_col:
continue
coerced = to_num(df[c])
if coerced.notna().sum() > 0:
df[c] = coerced
cands.append(c)
if not cands:
raise ValueError("Could not find any numeric incidence column.")
inc_col = max(cands, key=lambda c: df[c].var(skipna=True))
df[inc_col] = to_num(df[inc_col])
# ------------------------------------------------------------
# 3) Convert ISO week -> Monday date (like your R parse_iso_8601 approach)
# ------------------------------------------------------------
def week_to_monday(w) -> pd.Timestamp:
w = str(w).strip()
m = re.match(r"^(?P<y>\d{4})(?P<w>\d{2})$", w) # YYYYWW
if m:
y = int(m.group("y")); ww = int(m.group("w"))
return pd.Timestamp(date.fromisocalendar(y, ww, 1))
m = re.match(r"^(?P<y>\d{4})-?W(?P<w>\d{1,2})$", w, flags=re.IGNORECASE) # YYYYWww
if m:
y = int(m.group("y")); ww = int(m.group("w"))
return pd.Timestamp(date.fromisocalendar(y, ww, 1))
m = re.match(r"^(?P<y>\d{4})s(?P<w>\d{1,2})$", w, flags=re.IGNORECASE) # YYYYsww
if m:
y = int(m.group("y")); ww = int(m.group("w"))
return pd.Timestamp(date.fromisocalendar(y, ww, 1))
ts = pd.to_datetime(w, errors="coerce")
if pd.isna(ts):
raise ValueError(f"Unrecognized week format: {w}")
return ts
df["date"] = df[week_col].apply(week_to_monday)
df = df.sort_values("date").reset_index(drop=True)
# ------------------------------------------------------------
# 4) Define epidemic year EXACTLY as: (Sep 1 N-1, Sep 1 N]
# We label it by the END year N (like your R variable 'annee')
# ------------------------------------------------------------
def season_end_year(d: pd.Timestamp) -> int:
# If date is strictly after Sep 1 of its calendar year => belongs to season ending next year
sep1 = pd.Timestamp(f"{d.year}-09-01")
return d.year + 1 if d > sep1 else d.year
df["annee"] = df["date"].apply(season_end_year)
# Keep reasonable years (at least one full season)
valid_years = sorted(df["annee"].unique())
# ------------------------------------------------------------
# 5) Compute BOTH metrics per epidemic year:
# - cumulative = sum(inc100)
# - peak = max(inc100)
# ------------------------------------------------------------
annual = (
df.groupby("annee", as_index=False)
.agg(
incidence_sum=(inc_col, "sum"),
incidence_peak=(inc_col, "max"),
n_weeks=(inc_col, "count"),
first_date=("date", "min"),
last_date=("date", "max"),
)
.sort_values("annee")
) )
head(inc_annuelle) # strongest/weakest by SUM
Résultats demandés strong_sum = annual.loc[annual["incidence_sum"].idxmax(), ["annee", "incidence_sum"]]
Année avec l’épidémie la plus forte weak_sum = annual.loc[annual["incidence_sum"].idxmin(), ["annee", "incidence_sum"]]
inc_annuelle[which.max(inc_annuelle$incidence), ]
inc_annuelle[which.min(inc_annuelle$incidence), ]
# strongest/weakest by PEAK
strong_peak = annual.loc[annual["incidence_peak"].idxmax(), ["annee", "incidence_peak"]]
weak_peak = annual.loc[annual["incidence_peak"].idxmin(), ["annee", "incidence_peak"]]
print(f"✅ Loaded: {DATA_PATH}")
print(f"✅ Separator: {repr(sep)}")
print(f"✅ Week column: {week_col}")
print(f"✅ Incidence column: {inc_col}")
print(f"✅ Date range: {df['date'].min().date()} -> {df['date'].max().date()}")
print("\n--- Epidemic year definition: (Sep 1 N-1, Sep 1 N] labeled by N ---\n")
print("RESULTS using CUMULATIVE incidence (SUM over the epidemic year):")
print(f" Strongest: year {int(strong_sum['annee'])} | sum = {strong_sum['incidence_sum']:.3f}")
print(f" Weakest: year {int(weak_sum['annee'])} | sum = {weak_sum['incidence_sum']:.3f}")
print("\nRESULTS using PEAK incidence (MAX weekly over the epidemic year):")
print(f" Strongest: year {int(strong_peak['annee'])} | peak = {strong_peak['incidence_peak']:.3f}")
print(f" Weakest: year {int(weak_peak['annee'])} | peak = {weak_peak['incidence_peak']:.3f}")
print("\n--- Sanity check (first 12 epidemic years) ---")
print(annual.head(12).to_string(index=False))
Markdown is supported
0% or
You are about to add 0 people to the discussion. Proceed with caution.
Finish editing this message first!
Please register or to comment