diff --git a/.gitignore b/.gitignore
deleted file mode 100644
index de9949e..0000000
--- a/.gitignore
+++ /dev/null
@@ -1,11 +0,0 @@
-@"
-.env
-__pycache__/
-*.pyc
-*.pyo
-venv/
-.venv/
-*.db
-.pytest_cache/
-htmlcov/
-"@
\ No newline at end of file
diff --git a/api/__init__.py b/api/__init__.py
deleted file mode 100644
index e69de29..0000000
diff --git a/api/auth.py b/api/auth.py
deleted file mode 100644
index e5c3450..0000000
--- a/api/auth.py
+++ /dev/null
@@ -1,58 +0,0 @@
-from datetime import datetime, timedelta, timezone
-from typing import Optional
-from jose import JWTError, jwt
-from passlib.context import CryptContext
-from pydantic import BaseModel, EmailStr
-
-SECRET_KEY = "your-secret-key-change-in-production"
-ALGORITHM = "HS256"
-ACCESS_TOKEN_EXPIRE_MINUTES = 30
-
-pwd_context = CryptContext(schemes=["argon2"], deprecated="auto")
-
-class Token(BaseModel):
- access_token: str
- token_type: str
-
-class TokenData(BaseModel):
- email: Optional[str] = None
-
-class UserLogin(BaseModel):
- email: EmailStr
- password: str
-
-class UserCreate(BaseModel):
- email: EmailStr
- password: str
- full_name: Optional[str] = None
-
-class UserResponse(BaseModel):
- id: int
- email: str
- full_name: Optional[str] = None
-
-def verify_password(plain_password: str, hashed_password: str) -> bool:
- return pwd_context.verify(plain_password, hashed_password)
-
-def get_password_hash(password: str) -> str:
- return pwd_context.hash(password)
-
-def create_access_token(data: dict, expires_delta: Optional[timedelta] = None):
- to_encode = data.copy()
- if expires_delta:
- expire = datetime.now(timezone.utc) + expires_delta
- else:
- expire = datetime.now(timezone.utc) + timedelta(minutes=15)
- to_encode.update({"exp": expire})
- encoded_jwt = jwt.encode(to_encode, SECRET_KEY, algorithm=ALGORITHM)
- return encoded_jwt
-
-def decode_token(token: str) -> Optional[str]:
- try:
- payload = jwt.decode(token, SECRET_KEY, algorithms=[ALGORITHM])
- email: str = payload.get("sub")
- if email is None:
- return None
- return email
- except JWTError:
- return None
\ No newline at end of file
diff --git a/api/main.py b/api/main.py
deleted file mode 100644
index 728f29e..0000000
--- a/api/main.py
+++ /dev/null
@@ -1,141 +0,0 @@
-from fastapi import FastAPI, Depends, HTTPException, status
-from fastapi.middleware.cors import CORSMiddleware
-from datetime import timedelta
-from database.database import SessionLocal
-from database.models import Document, User
-from api.auth import (
- UserLogin, UserCreate, Token, UserResponse, get_password_hash, verify_password,
- create_access_token, decode_token, ACCESS_TOKEN_EXPIRE_MINUTES
-)
-
-app = FastAPI()
-
-app.add_middleware(
- CORSMiddleware,
- allow_origins=["*"],
- allow_credentials=True,
- allow_methods=["*"],
- allow_headers=["*"],
-)
-
-@app.get("/")
-def root():
- return {"message": "RegWatch API running"}
-
-@app.post("/test-login")
-def test_login(credentials: UserLogin):
- return {"received": credentials.dict()}
-
-@app.post("/signup", response_model=UserResponse)
-def signup(user: UserCreate):
- db = SessionLocal()
-
- existing_user = db.query(User).filter(User.email == user.email).first()
- if existing_user:
- db.close()
- raise HTTPException(
- status_code=status.HTTP_400_BAD_REQUEST,
- detail="Email already registered"
- )
-
- hashed_password = get_password_hash(user.password)
- db_user = User(
- email=user.email,
- password=hashed_password,
- full_name=user.full_name
- )
- db.add(db_user)
- db.commit()
- db.refresh(db_user)
- db.close()
-
- return UserResponse(
- id=db_user.id,
- email=db_user.email,
- full_name=db_user.full_name
- )
-
-@app.post("/login", response_model=Token)
-def login(credentials: UserLogin):
- try:
- db = SessionLocal()
-
- user = db.query(User).filter(User.email == credentials.email).first()
- db.close()
-
- if not user:
- raise HTTPException(
- status_code=status.HTTP_401_UNAUTHORIZED,
- detail="User not found"
- )
-
- if not verify_password(credentials.password, user.password):
- raise HTTPException(
- status_code=status.HTTP_401_UNAUTHORIZED,
- detail="Invalid password"
- )
-
- access_token_expires = timedelta(minutes=ACCESS_TOKEN_EXPIRE_MINUTES)
- access_token = create_access_token(
- data={"sub": user.email}, expires_delta=access_token_expires
- )
-
- return Token(access_token=access_token, token_type="bearer")
- except HTTPException:
- raise
- except Exception as e:
- print(f"ERROR in login: {str(e)}")
- import traceback
- traceback.print_exc()
- raise HTTPException(
- status_code=500,
- detail=f"Server error: {str(e)}"
- )
-
-@app.get("/me", response_model=UserResponse)
-def get_current_user(token: str):
- email = decode_token(token)
- if not email:
- raise HTTPException(
- status_code=status.HTTP_401_UNAUTHORIZED,
- detail="Invalid token"
- )
-
- db = SessionLocal()
- user = db.query(User).filter(User.email == email).first()
- db.close()
-
- if not user:
- raise HTTPException(
- status_code=status.HTTP_404_NOT_FOUND,
- detail="User not found"
- )
-
- return UserResponse(
- id=user.id,
- email=user.email,
- full_name=user.full_name
- )
-
-@app.get("/documents")
-def get_documents():
- db = SessionLocal()
-
- documents = db.query(Document).all()
-
- result = []
-
- for doc in documents:
- result.append({
- "id": doc.id,
- "title": doc.title,
- "ingredient": doc.ingredient,
- "source": doc.source,
- "type": doc.document_type,
- "date": doc.meeting_date,
- "pdf_url": doc.pdf_url
- })
-
- db.close()
-
- return result
\ No newline at end of file
diff --git a/check_db.py b/check_db.py
deleted file mode 100644
index aa1a60a..0000000
--- a/check_db.py
+++ /dev/null
@@ -1,11 +0,0 @@
-from database.database import SessionLocal
-from database.models import Document
-
-db = SessionLocal()
-
-documents = db.query(Document).all()
-
-print(f"Total documents: {len(documents)}")
-
-for doc in documents[:5]:
- print(doc.title)
\ No newline at end of file
diff --git a/database/__init__.py b/database/__init__.py
deleted file mode 100644
index e69de29..0000000
diff --git a/database/database.py b/database/database.py
deleted file mode 100644
index a59806f..0000000
--- a/database/database.py
+++ /dev/null
@@ -1,19 +0,0 @@
-from sqlalchemy import create_engine
-from sqlalchemy.orm import sessionmaker, declarative_base
-import os
-
-DB_PATH = os.path.join(os.path.dirname(os.path.dirname(__file__)), "regwatch.db")
-DATABASE_URL = f"sqlite:///{DB_PATH}"
-
-engine = create_engine(
- DATABASE_URL,
- connect_args={"check_same_thread": False}
-)
-
-SessionLocal = sessionmaker(
- autocommit=False,
- autoflush=False,
- bind=engine
-)
-
-Base = declarative_base()
\ No newline at end of file
diff --git a/database/models.py b/database/models.py
deleted file mode 100644
index be93f0b..0000000
--- a/database/models.py
+++ /dev/null
@@ -1,21 +0,0 @@
-from sqlalchemy import Column, Integer, String
-from database.database import Base
-
-class User(Base):
- __tablename__ = "users"
-
- id = Column(Integer, primary_key=True, index=True)
- email = Column(String, unique=True, index=True)
- password = Column(String)
- full_name = Column(String, nullable=True)
-
-class Document(Base):
- __tablename__ = "documents"
-
- id = Column(Integer, primary_key=True, index=True)
- title = Column(String)
- ingredient = Column(String)
- source = Column(String)
- document_type = Column(String)
- meeting_date = Column(String, nullable=True)
- pdf_url = Column(String, unique=True)
\ No newline at end of file
diff --git a/init_db.py b/init_db.py
deleted file mode 100644
index 9143580..0000000
--- a/init_db.py
+++ /dev/null
@@ -1,27 +0,0 @@
-from database.database import engine, SessionLocal
-from database.models import Base, User
-from api.auth import get_password_hash
-
-Base.metadata.create_all(bind=engine)
-
-# Créer un utilisateur de test
-db = SessionLocal()
-
-# Vérifier si l'utilisateur existe déjà
-existing_user = db.query(User).filter(User.email == "admin@test.com").first()
-
-if not existing_user:
- test_user = User(
- email="admin@test.com",
- password=get_password_hash("password123"),
- full_name="Admin Test User"
- )
- db.add(test_user)
- db.commit()
- print("Test user created: admin@test.com / password123")
-else:
- print("Test user already exists")
-
-db.close()
-
-print("Database initialized successfully.")
\ No newline at end of file
diff --git a/requirements.txt b/requirements.txt
deleted file mode 100644
index b8eb833..0000000
--- a/requirements.txt
+++ /dev/null
@@ -1,11 +0,0 @@
-fastapi
-uvicorn
-sqlalchemy
-requests
-beautifulsoup4
-pydantic[email]
-ollama
-python-jose[cryptography]
-passlib[argon2]
-python-multipart
-pdfplumber
\ No newline at end of file
diff --git a/scrapers/__init__.py b/scrapers/__init__.py
deleted file mode 100644
index e69de29..0000000
diff --git a/scrapers/cir_scraper.py b/scrapers/cir_scraper.py
deleted file mode 100644
index b6fd36e..0000000
--- a/scrapers/cir_scraper.py
+++ /dev/null
@@ -1,276 +0,0 @@
-import sys
-import os
-import requests
-import time
-import re
-import io
-import ollama
-import json
-import pdfplumber
-from bs4 import BeautifulSoup
-from urllib.parse import urljoin
-
-sys.path.append(
- os.path.abspath(
- os.path.join(os.path.dirname(__file__), "..")
- )
-)
-
-from database.database import SessionLocal, engine
-from database.models import Document, Base
-
-
-headers = {
- "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
-}
-
-session = requests.Session()
-session.headers.update(headers)
-
-BASE_URL = "https://www.cir-safety.org"
-MEETING_NUMBERS = range(115, 175)
-PDF_LIMIT = 100
-
-url_templates = [
- "https://www.cir-safety.org/meeting/{num}th-expert-panel-meeting",
- "https://www.cir-safety.org/meeting/{num}st-expert-panel-meeting",
- "https://www.cir-safety.org/meeting/{num}nd-expert-panel-meeting",
- "https://www.cir-safety.org/meeting/{num}rd-expert-panel-meeting",
- "https://www.cir-safety.org/meeting/{num}th-cir-expert-panel-meeting",
- "https://www.cir-safety.org/meeting/{num}st-cir-expert-panel-meeting",
-]
-
-PDF_RE = re.compile(r"\.pdf", re.I)
-SKIP_KEYWORDS = ["Agenda", "Minutes", "Status Report"]
-
-
-def extract_pages_text(pdf_bytes: bytes, max_pages: int = 3) -> str:
- """Extrait le texte des 3 premières pages du PDF."""
- try:
- with pdfplumber.open(io.BytesIO(pdf_bytes)) as pdf:
- if not pdf.pages:
- return ""
- texts = []
- for i, page in enumerate(pdf.pages[:max_pages]):
- text = page.extract_text() or ""
- if text.strip():
- texts.append(f"--- PAGE {i+1} ---\n{text}")
- return "\n\n".join(texts)[:5000]
- except Exception as e:
- print(f" ⚠ Erreur extraction PDF: {e}")
- return ""
-
-
-def extract_info_with_ollama(text: str, pdf_url: str) -> dict:
- """
- Envoie le texte des premières pages à Ollama
- et récupère les infos structurées en JSON.
- """
- if not text.strip():
- filename = pdf_url.split("/")[-1].replace("_", " ").replace(".pdf", "")
- return {
- "title": filename,
- "ingredient": None,
- "document_type": "document",
- "date": None,
- }
-
- prompt = f"""You are a regulatory document parser specialized in cosmetic safety documents.
-
-Analyze the following text extracted from the first 3 pages of a regulatory document and extract the information below.
-
-Return ONLY a valid JSON object with these exact fields:
-{{
- "title": "the real and complete official title of the document as it appears in the text",
- "ingredient": "the cosmetic ingredient(s) name only, or null if this document is not about a specific ingredient (e.g. it's a general study, methodology paper, meeting report, status report, etc.)",
- "document_type": "one of: final report, draft report, tentative report, safety assessment, opinion, strategy, study, meeting report, status report, other",
- "date": "year only as a string e.g. '2023', or null if not found"
-}}
-
-Important rules:
-- title: look carefully in pages 1 and 2 for the REAL title, it is usually the largest or most prominent text. Do NOT use a generic description.
-- ingredient: extract ONLY the ingredient name(s). Remove phrases like 'Safety Assessment of', 'Final Report on', 'Opinion on', 'Amended Safety Assessment of'. If the document is a general study, methodology, or administrative document with no specific ingredient, return null.
-- document_type: detect from keywords. If you see 'Final' → 'final report', 'Draft' → 'draft report', 'Tentative' → 'tentative report', 'Opinion' → 'opinion', 'Strategy' → 'strategy', 'Study' → 'study'.
-- date: find the most recent year mentioned in the document header or footer (between 2000-2030).
-- Return ONLY the JSON object, no markdown, no explanation.
-
-Document text (first 3 pages):
-{text}"""
-
- try:
- response = ollama.chat(
- model="llama3.2",
- messages=[{"role": "user", "content": prompt}],
- options={"temperature": 0}
- )
-
- raw = response["message"]["content"].strip()
- raw = re.sub(r"```json|```", "", raw).strip()
-
- # Extrait le JSON même s'il y a du texte autour
- match = re.search(r"\{.*\}", raw, re.DOTALL)
- if match:
- raw = match.group(0)
-
- data = json.loads(raw)
-
- ingredient = data.get("ingredient")
- if ingredient and str(ingredient).strip().lower() in ("null", "none", "n/a", ""):
- ingredient = None
-
- return {
- "title": str(data.get("title") or "").strip()[:200] or "Unknown",
- "ingredient": str(ingredient).strip()[:200] if ingredient else None,
- "document_type": str(data.get("document_type") or "document").strip().lower(),
- "date": str(data.get("date")).strip() if data.get("date") else None,
- }
-
- except Exception as e:
- print(f" ⚠ Ollama error: {e}")
- filename = pdf_url.split("/")[-1].replace("_", " ").replace(".pdf", "")
- return {
- "title": filename,
- "ingredient": None,
- "document_type": "document",
- "date": None,
- }
-
-
-def get_working_url(num):
- for template in url_templates:
- url = template.format(num=num)
- try:
- r = session.get(url, timeout=20)
- if r.status_code == 200:
- return url, r
- except Exception:
- pass
- return None, None
-
-
-def extract_pdf_urls_from_page(r):
- soup = BeautifulSoup(r.text, "html.parser")
- all_pdf_links = soup.find_all("a", href=PDF_RE)
- print(f" → {len(all_pdf_links)} PDF links found")
-
- if not all_pdf_links:
- print(" ⚠ No PDF links detected.")
- return []
-
- urls = []
- seen = set()
-
- for link in all_pdf_links:
- href = link.get("href", "")
- pdf_url = urljoin(BASE_URL, href)
-
- if pdf_url in seen:
- continue
- seen.add(pdf_url)
-
- context_el = None
- for tag in ("tr", "li", "div", "p"):
- context_el = link.find_parent(tag)
- if context_el:
- break
- text = context_el.get_text(" ", strip=True) if context_el else ""
-
- if any(kw in text for kw in SKIP_KEYWORDS):
- continue
-
- urls.append(pdf_url)
-
- return urls
-
-
-def save_documents(docs):
- db = SessionLocal()
- saved = 0
- try:
- for d in docs:
- exists = db.query(Document).filter(Document.pdf_url == d["pdf_url"]).first()
- if not exists:
- db.add(Document(**d))
- saved += 1
- db.commit()
- print(f" ✓ {saved} new document(s) saved")
- except Exception as e:
- print(f" DB error: {e}")
- db.rollback()
- finally:
- db.close()
-
-
-def scrape_cir(limit=PDF_LIMIT):
- Base.metadata.create_all(bind=engine)
- total = 0
- print(f"CIR scraper — limit: {limit} PDFs\n")
-
- for meeting_num in MEETING_NUMBERS:
- if total >= limit:
- print(f"\n✓ Limit of {limit} reached.")
- break
-
- print(f"\n{'='*60}")
- print(f"Meeting: {meeting_num} ({total}/{limit})")
-
- url, r = get_working_url(meeting_num)
- if not url:
- print(" ✗ No page found")
- continue
-
- print(f" ✓ {url}")
-
- try:
- pdf_urls = extract_pdf_urls_from_page(r)
- docs_to_save = []
-
- for pdf_url in pdf_urls:
- if total >= limit:
- break
-
- print(f"\n Downloading: {pdf_url.split('/')[-1]}")
-
- try:
- pdf_resp = session.get(pdf_url, timeout=30)
- if pdf_resp.status_code != 200:
- print(f" ✗ HTTP {pdf_resp.status_code}")
- continue
-
- text = extract_pages_text(pdf_resp.content, max_pages=3)
- print(f" → {len(text)} chars extracted")
-
- info = extract_info_with_ollama(text, pdf_url)
- print(f" → Title: {info['title'][:70]}")
- print(f" → Ingredient: {info['ingredient'] or '(none — general document)'}")
- print(f" → Type: {info['document_type']} | Date: {info['date']}")
-
- docs_to_save.append({
- "title": info["title"],
- "ingredient": info["ingredient"] or "N/A",
- "source": "CIR",
- "document_type": info["document_type"],
- "meeting_date": info["date"],
- "pdf_url": pdf_url,
- })
- total += 1
- time.sleep(1)
-
- except Exception as e:
- print(f" ✗ Error processing PDF: {e}")
- continue
-
- if docs_to_save:
- save_documents(docs_to_save)
-
- time.sleep(2)
-
- except Exception as e:
- print(f" ✗ Failed: {e}")
-
- print(f"\n{'='*60}")
- print(f"CIR scraping complete: {total}/{limit} PDFs")
-
-
-if __name__ == "__main__":
- scrape_cir()
\ No newline at end of file
diff --git a/scrapers/sccs_scraper.py b/scrapers/sccs_scraper.py
deleted file mode 100644
index 35feb7a..0000000
--- a/scrapers/sccs_scraper.py
+++ /dev/null
@@ -1,271 +0,0 @@
-import sys
-import os
-import requests
-import time
-import re
-import io
-import ollama
-import json
-import pdfplumber
-from bs4 import BeautifulSoup
-from urllib.parse import urljoin
-
-sys.path.append(
- os.path.abspath(
- os.path.join(os.path.dirname(__file__), "..")
- )
-)
-
-from database.database import SessionLocal, engine
-from database.models import Document, Base
-
-BASE_URL = "https://health.ec.europa.eu"
-START_URL = (
- "https://health.ec.europa.eu/"
- "scientific-committees/"
- "scientific-committee-consumer-safety-sccs/"
- "sccs-opinions_en"
-)
-MAX_PDFS = 100
-
-headers = {"User-Agent": "Mozilla/5.0"}
-session = requests.Session()
-session.headers.update(headers)
-
-
-def extract_pages_text(pdf_bytes: bytes, max_pages: int = 3) -> str:
- """Extrait le texte des 3 premières pages du PDF."""
- try:
- with pdfplumber.open(io.BytesIO(pdf_bytes)) as pdf:
- if not pdf.pages:
- return ""
- texts = []
- for i, page in enumerate(pdf.pages[:max_pages]):
- text = page.extract_text() or ""
- if text.strip():
- texts.append(f"--- PAGE {i+1} ---\n{text}")
- return "\n\n".join(texts)[:5000]
- except Exception as e:
- print(f" ⚠ Erreur extraction PDF: {e}")
- return ""
-
-
-def extract_info_with_ollama(text: str, pdf_url: str, default_type: str = "opinion") -> dict:
- """
- Envoie le texte des premières pages à Ollama
- et récupère les infos structurées en JSON.
- """
- if not text.strip():
- filename = pdf_url.split("/")[-1].replace("_", " ").replace(".pdf", "")
- return {
- "title": filename,
- "ingredient": None,
- "document_type": default_type,
- "date": None,
- }
-
- prompt = f"""You are a regulatory document parser specialized in cosmetic safety documents.
-
-Analyze the following text extracted from the first 3 pages of a regulatory document and extract the information below.
-
-Return ONLY a valid JSON object with these exact fields:
-{{
- "title": "the real and complete official title of the document as it appears in the text",
- "ingredient": "the cosmetic ingredient(s) name only, or null if this document is not about a specific ingredient (e.g. it's a general study, methodology paper, meeting report, status report, etc.)",
- "document_type": "one of: final report, draft report, tentative report, safety assessment, opinion, strategy, study, meeting report, status report, other",
- "date": "year only as a string e.g. '2023', or null if not found"
-}}
-
-Important rules:
-- title: look carefully in pages 1 and 2 for the REAL title, it is usually the largest or most prominent text. Do NOT use a generic description.
-- ingredient: extract ONLY the ingredient name(s). Remove phrases like 'Safety Assessment of', 'Final Report on', 'Opinion on', 'Amended Safety Assessment of'. If the document is a general study, methodology, or administrative document with no specific ingredient, return null.
-- document_type: detect from keywords. If you see 'Final' → 'final report', 'Draft' → 'draft report', 'Tentative' → 'tentative report', 'Opinion' → 'opinion', 'Strategy' → 'strategy', 'Study' → 'study'.
-- date: find the most recent year mentioned in the document header or footer (between 2000-2030).
-- Return ONLY the JSON object, no markdown, no explanation.
-
-Document text (first 3 pages):
-{text}"""
-
- try:
- response = ollama.chat(
- model="llama3.2",
- messages=[{"role": "user", "content": prompt}],
- options={"temperature": 0}
- )
-
- raw = response["message"]["content"].strip()
- raw = re.sub(r"```json|```", "", raw).strip()
-
- match = re.search(r"\{.*\}", raw, re.DOTALL)
- if match:
- raw = match.group(0)
-
- data = json.loads(raw)
-
- ingredient = data.get("ingredient")
- if ingredient and str(ingredient).strip().lower() in ("null", "none", "n/a", ""):
- ingredient = None
-
- return {
- "title": str(data.get("title") or "").strip()[:200] or "Unknown",
- "ingredient": str(ingredient).strip()[:200] if ingredient else None,
- "document_type": str(data.get("document_type") or default_type).strip().lower(),
- "date": str(data.get("date")).strip() if data.get("date") else None,
- }
-
- except Exception as e:
- print(f" ⚠ Ollama error: {e}")
- filename = pdf_url.split("/")[-1].replace("_", " ").replace(".pdf", "")
- return {
- "title": filename,
- "ingredient": None,
- "document_type": default_type,
- "date": None,
- }
-
-
-def get_soup(url):
- try:
- r = session.get(url, timeout=25)
- if r.status_code == 429:
- print("429 → waiting 60s")
- time.sleep(60)
- r = session.get(url, timeout=25)
- r.raise_for_status()
- return BeautifulSoup(r.text, "html.parser")
- except Exception as e:
- print(e)
- return None
-
-
-def save_document(data: dict) -> bool:
- db = SessionLocal()
- try:
- exists = db.query(Document).filter(Document.pdf_url == data["pdf_url"]).first()
- if exists:
- return False
- db.add(Document(**data))
- db.commit()
- return True
- except Exception as e:
- print(f" DB error: {e}")
- db.rollback()
- return False
- finally:
- db.close()
-
-
-def process_pdf(pdf_url: str) -> dict | None:
- print(f"\n Downloading: {pdf_url.split('/')[-1]}")
- try:
- r = session.get(pdf_url, timeout=30)
- if r.status_code != 200:
- print(f" ✗ HTTP {r.status_code}")
- return None
-
- text = extract_pages_text(r.content, max_pages=3)
- print(f" → {len(text)} chars extracted")
-
- info = extract_info_with_ollama(text, pdf_url, default_type="opinion")
- print(f" → Title: {info['title'][:70]}")
- print(f" → Ingredient: {info['ingredient'] or '(none — general document)'}")
- print(f" → Type: {info['document_type']} | Date: {info['date']}")
-
- return {
- "title": info["title"],
- "ingredient": info["ingredient"] or "N/A",
- "source": "SCCS",
- "document_type": info["document_type"],
- "meeting_date": info["date"],
- "pdf_url": pdf_url,
- }
-
- except Exception as e:
- print(f" ✗ Error: {e}")
- return None
-
-
-def process_detail_page(url: str, saved_count: int, limit: int) -> int:
- if saved_count >= limit:
- return 0
-
- time.sleep(2)
- print(f"\n Page: {url.split('/')[-1]}")
-
- soup = get_soup(url)
- if not soup:
- return 0
-
- added = 0
- excluded = ["draft", "preliminary", "summary", "factsheet", "infographic"]
-
- for a in soup.find_all("a", href=re.compile(r"\.pdf$", re.I)):
- if saved_count + added >= limit:
- break
-
- link_text = a.get_text(strip=True).lower()
- if any(x in link_text for x in excluded):
- continue
-
- pdf_url = urljoin(BASE_URL, a["href"])
- doc = process_pdf(pdf_url)
-
- if doc:
- if save_document(doc):
- added += 1
- print(f" ✓ [{saved_count + added}/{limit}] saved")
-
- time.sleep(1)
-
- return added
-
-
-def scrape_sccs(limit=MAX_PDFS):
- Base.metadata.create_all(bind=engine)
-
- saved_count = 0
- urls_to_visit = [START_URL]
- visited = set()
-
- print(f"SCCS scraper — limit: {limit} PDFs\n")
-
- while urls_to_visit:
- if saved_count >= limit:
- print(f"\n✓ Limit of {limit} reached.")
- break
-
- current = urls_to_visit.pop(0)
- if current in visited:
- continue
- visited.add(current)
-
- print(f"\n{'='*60}")
- print(f"Exploring: {current.split('/')[-1]} ({saved_count}/{limit})")
-
- soup = get_soup(current)
- if not soup:
- continue
-
- for a in soup.find_all("a", href=True):
- if saved_count >= limit:
- break
-
- href = a["href"].lower()
-
- if "opinions" in href or "sccs_o_" in href:
- full_url = urljoin(BASE_URL, a["href"])
-
- if "_en" in full_url and not full_url.endswith(".pdf"):
- if any(x in full_url for x in ["2016-2021", "2013-2016", "2009-2012"]):
- if full_url not in urls_to_visit:
- urls_to_visit.append(full_url)
- else:
- added = process_detail_page(full_url, saved_count, limit)
- saved_count += added
-
- print(f"\n{'='*60}")
- print(f"SCCS scraping complete: {saved_count}/{limit} PDFs")
-
-
-if __name__ == "__main__":
- scrape_sccs()
\ No newline at end of file
diff --git a/tests/__init__.py b/tests/__init__.py
deleted file mode 100644
index e69de29..0000000
diff --git a/tests/test_regwatch_complet.py b/tests/test_regwatch_complet.py
deleted file mode 100644
index 729c2d2..0000000
--- a/tests/test_regwatch_complet.py
+++ /dev/null
@@ -1,531 +0,0 @@
-import pytest
-import sys
-import os
-import time
-
-sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), "..")))
-
-# ── Import app FastAPI ────────────────────────────────────────
-try:
- from fastapi.testclient import TestClient
- from api.main import app
- client = TestClient(app)
- APP_OK = True
-except Exception as e:
- APP_OK = False
-
-def skip_no_app():
- if not APP_OK:
- pytest.skip("App / BDD non disponible")
-
-
-# ══════════════════════════════════════════════════════════════
-# PARTIE 1 — TESTS UNITAIRES (scraper CIR)
-# ══════════════════════════════════════════════════════════════
-
-class TestUnitairesExtractPagesText:
- """
- Tests unitaires de extract_pages_text()
- Réf. cir_scraper.py l.46 — extraction texte PDF
- """
-
- def test_bytes_vide_retourne_chaine_vide(self):
- """Bytes vides → '' sans crash"""
- from scrapers.cir_scraper import extract_pages_text
- assert extract_pages_text(b"") == ""
-
- def test_pdf_invalide_retourne_chaine_vide(self):
- """Contenu non-PDF → '' (le except l.57 fonctionne)"""
- from scrapers.cir_scraper import extract_pages_text
- assert extract_pages_text(b"ceci n est pas un pdf") == ""
-
- def test_retour_toujours_une_str(self):
- """La fonction retourne toujours str, jamais None"""
- from scrapers.cir_scraper import extract_pages_text
- result = extract_pages_text(b"")
- assert isinstance(result, str)
-
- def test_limite_5000_caracteres(self):
- """Le texte est tronqué à 5000 chars max (l.55 : [:5000])"""
- long_text = "A" * 10000
- assert len(long_text[:5000]) == 5000
-
-
-class TestUnitairesFallbackOllama:
- """
- Tests unitaires du fallback Ollama (cir_scraper.py l.65)
- Quand le texte extrait est vide, on utilise le nom du fichier PDF
- """
-
- def test_fallback_utilise_nom_fichier(self):
- """pdf_url → nom du fichier sans extension (l.66-68)"""
- pdf_url = "https://cir-safety.org/sites/files/Retinol_2024.pdf"
- filename = pdf_url.split("/")[-1].replace("_", " ").replace(".pdf", "")
- assert filename == "Retinol 2024"
-
- def test_fallback_retourne_4_cles(self):
- """Le fallback retourne bien title, ingredient, document_type, date"""
- fallback = {
- "title": "Mon Document",
- "ingredient": None,
- "document_type": "document",
- "date": None,
- }
- assert all(k in fallback for k in ["title", "ingredient", "document_type", "date"])
- assert fallback["ingredient"] is None
- assert fallback["document_type"] == "document"
-
- def test_nettoyage_ingredient_null_string(self):
- """Ollama retourne 'null'/'none'/'n/a' → converti en None (l.112-113)"""
- for val in ["null", "none", "n/a", "", "NULL", "None"]:
- result = None if str(val).strip().lower() in ("null", "none", "n/a", "") else val
- assert result is None, f"'{val}' aurait dû être None"
-
- def test_titre_tronque_200_chars(self):
- """Le titre est limité à 200 chars (l.117 : [:200])"""
- long = "A" * 300
- assert len(str(long).strip()[:200]) == 200
-
-
-class TestUnitairesTemplatesURL:
- """
- Tests unitaires des templates d'URL CIR (cir_scraper.py l.31-40)
- """
-
- def test_tous_les_templates_contiennent_num(self):
- """Chaque template contient {num}"""
- from scrapers.cir_scraper import url_templates
- for t in url_templates:
- assert "{num}" in t
-
- def test_tous_les_templates_sont_https(self):
- """Tous les templates utilisent HTTPS"""
- from scrapers.cir_scraper import url_templates
- for t in url_templates:
- assert t.startswith("https://")
-
- def test_format_url_remplace_num(self):
- """Le format {num} est bien remplacé"""
- from scrapers.cir_scraper import url_templates
- url = url_templates[0].format(num=150)
- assert "150" in url
- assert "{num}" not in url
-
- def test_meeting_numbers_couvre_115_a_174(self):
- """MEETING_NUMBERS = range(115, 175) → 60 réunions"""
- from scrapers.cir_scraper import MEETING_NUMBERS
- nums = list(MEETING_NUMBERS)
- assert nums[0] == 115
- assert nums[-1] == 174
- assert len(nums) == 60
-
- def test_pdf_limit_est_100(self):
- """PDF_LIMIT = 100 (l.32)"""
- from scrapers.cir_scraper import PDF_LIMIT
- assert PDF_LIMIT == 100
-
-
-class TestUnitairesFiltragePDF:
- """
- Tests unitaires du filtrage des liens PDF (cir_scraper.py l.148)
- """
-
- def test_skip_keywords_contient_les_3_valeurs(self):
- """SKIP_KEYWORDS = ['Agenda', 'Minutes', 'Status Report']"""
- from scrapers.cir_scraper import SKIP_KEYWORDS
- assert "Agenda" in SKIP_KEYWORDS
- assert "Minutes" in SKIP_KEYWORDS
- assert "Status Report" in SKIP_KEYWORDS
-
- def test_regex_pdf_detecte_extension_pdf(self):
- """PDF_RE détecte .pdf / .PDF / .Pdf"""
- from scrapers.cir_scraper import PDF_RE
- assert PDF_RE.search("document.pdf")
- assert PDF_RE.search("document.PDF")
- assert PDF_RE.search("rapport.Pdf")
-
- def test_regex_pdf_ne_detecte_pas_autres(self):
- """PDF_RE ne détecte pas .docx, .png"""
- from scrapers.cir_scraper import PDF_RE
- assert not PDF_RE.search("document.docx")
- assert not PDF_RE.search("image.png")
-
- def test_agenda_est_filtre(self):
- """Un contexte contenant 'Agenda' → filtré"""
- from scrapers.cir_scraper import SKIP_KEYWORDS
- context = "116th Expert Panel Meeting Agenda"
- assert any(kw in context for kw in SKIP_KEYWORDS)
-
- def test_final_report_non_filtre(self):
- """Un 'Final Report' passe le filtre"""
- from scrapers.cir_scraper import SKIP_KEYWORDS
- context = "Final Report on the Safety Assessment of Retinol"
- assert not any(kw in context for kw in SKIP_KEYWORDS)
-
-
-class TestUnitairesDeduplication:
- """
- Tests unitaires de la logique de déduplication (cir_scraper.py l.172)
- """
-
- def test_nouvelle_url_acceptee(self):
- """Une URL absente du set → à insérer"""
- existing = {"https://cir.org/doc1.pdf", "https://cir.org/doc2.pdf"}
- assert "https://cir.org/doc3.pdf" not in existing
-
- def test_url_existante_rejetee(self):
- """Une URL déjà présente → doublons ignoré"""
- existing = {"https://cir.org/doc1.pdf"}
- assert "https://cir.org/doc1.pdf" in existing
-
- def test_structure_doc_6_champs(self):
- """Un doc à sauvegarder contient les 6 champs requis"""
- doc = {
- "title": "Final Report on Retinol",
- "ingredient": "Retinol",
- "source": "CIR",
- "document_type": "final report",
- "meeting_date": "2024",
- "pdf_url": "https://cir.org/retinol_2024.pdf",
- }
- for champ in ["title", "ingredient", "source", "document_type", "meeting_date", "pdf_url"]:
- assert champ in doc
-
- def test_source_est_cir(self):
- """La source est toujours 'CIR' pour ce scraper"""
- assert "CIR" == "CIR"
-
-
-class TestUnitairesHeadersHTTP:
- """
- Tests unitaires des headers HTTP anti-blocage (cir_scraper.py l.22)
- """
-
- def test_user_agent_simule_chrome(self):
- """User-Agent contient Mozilla et Chrome"""
- from scrapers.cir_scraper import headers
- assert "User-Agent" in headers
- assert "Mozilla" in headers["User-Agent"]
- assert "Chrome" in headers["User-Agent"]
-
- def test_session_utilise_les_headers(self):
- """La session requests hérite des headers"""
- from scrapers.cir_scraper import session
- assert "User-Agent" in session.headers
- assert "Mozilla" in session.headers["User-Agent"]
-
-
-# ══════════════════════════════════════════════════════════════
-# PARTIE 2 — TESTS D'INTÉGRATION (endpoints FastAPI)
-# ══════════════════════════════════════════════════════════════
-
-class TestIntegrationRacine:
- """Tests de GET / (main.py l.21)"""
-
- def test_racine_200(self):
- skip_no_app()
- assert client.get("/").status_code == 200
-
- def test_racine_retourne_message_running(self):
- skip_no_app()
- data = client.get("/").json()
- assert "message" in data
- assert "running" in data["message"].lower() or "RegWatch" in data["message"]
-
- def test_racine_retourne_json(self):
- skip_no_app()
- r = client.get("/")
- assert r.headers["content-type"].startswith("application/json")
-
-
-class TestIntegrationSignup:
- """Tests de POST /signup (main.py l.28)"""
-
- def test_email_invalide_422(self):
- skip_no_app()
- r = client.post("/signup", json={"email": "pasunemail", "password": "test", "full_name": "T"})
- assert r.status_code == 422
-
- def test_champs_manquants_422(self):
- skip_no_app()
- r = client.post("/signup", json={"email": "test@test.com"})
- assert r.status_code == 422
-
- def test_email_deja_existant_400(self):
- """Email déjà pris → 400 'Email already registered' (l.33-37)"""
- skip_no_app()
- email = f"doublon_{int(time.time())}@regwatch.fr"
- client.post("/signup", json={"email": email, "password": "test123", "full_name": "T"})
- r = client.post("/signup", json={"email": email, "password": "autre", "full_name": "T"})
- assert r.status_code == 400
-
- def test_signup_reussi_retourne_id_email(self):
- """Signup valide → 200 avec id et email"""
- skip_no_app()
- email = f"ok_{int(time.time())}@regwatch.fr"
- r = client.post("/signup", json={"email": email, "password": "test123", "full_name": "OK"})
- assert r.status_code == 200
- data = r.json()
- assert "id" in data
- assert data["email"] == email
-
-
-class TestIntegrationLogin:
- """Tests de POST /login (main.py l.55)"""
-
- def test_email_inexistant_401(self):
- """Email inconnu → 401 'User not found' (l.65-68)"""
- skip_no_app()
- r = client.post("/login", json={"email": "xyz_inconnu@regwatch.fr", "password": "test"})
- assert r.status_code == 401
- assert "not found" in r.json()["detail"].lower()
-
- def test_mauvais_mot_de_passe_401(self):
- """Mauvais mdp → 401 'Invalid password' (l.70-74)"""
- skip_no_app()
- email = f"mdp_{int(time.time())}@regwatch.fr"
- client.post("/signup", json={"email": email, "password": "bonmdp", "full_name": "T"})
- r = client.post("/login", json={"email": email, "password": "mauvaismdp"})
- assert r.status_code == 401
-
- def test_login_reussi_retourne_token_bearer(self):
- """Login valide → token JWT + token_type='bearer' (l.76-82)"""
- skip_no_app()
- email = f"jwt_{int(time.time())}@regwatch.fr"
- client.post("/signup", json={"email": email, "password": "test123", "full_name": "T"})
- r = client.post("/login", json={"email": email, "password": "test123"})
- assert r.status_code == 200
- data = r.json()
- assert "access_token" in data
- assert data["token_type"] == "bearer"
- assert len(data["access_token"]) > 10
-
- def test_corps_vide_422(self):
- skip_no_app()
- assert client.post("/login", json={}).status_code == 422
-
-
-class TestIntegrationMe:
- """Tests de GET /me?token=... (main.py l.88)"""
-
- def test_sans_token_422(self):
- """Paramètre token manquant → 422"""
- skip_no_app()
- assert client.get("/me").status_code == 422
-
- def test_token_invalide_401(self):
- """Token JWT falsifié → 401 'Invalid token' (l.90-93)"""
- skip_no_app()
- r = client.get("/me?token=token_completement_faux")
- assert r.status_code == 401
- assert "invalid" in r.json()["detail"].lower() or "token" in r.json()["detail"].lower()
-
- def test_token_valide_retourne_user(self):
- """Token valide → 200 avec email de l'utilisateur"""
- skip_no_app()
- email = f"me_{int(time.time())}@regwatch.fr"
- client.post("/signup", json={"email": email, "password": "test123", "full_name": "T"})
- token = client.post("/login", json={"email": email, "password": "test123"}).json()["access_token"]
- r = client.get(f"/me?token={token}")
- assert r.status_code == 200
- assert r.json()["email"] == email
-
-
-class TestIntegrationDocuments:
- """Tests de GET /documents (main.py l.104)"""
-
- def test_sans_auth_retourne_200(self):
- """Pas de JWT requis sur cet endpoint (main.py l.104)"""
- skip_no_app()
- assert client.get("/documents").status_code == 200
-
- def test_retourne_liste_json(self):
- skip_no_app()
- assert isinstance(client.get("/documents").json(), list)
-
- def test_structure_7_champs_par_document(self):
- """Chaque doc a id, title, ingredient, source, type, date, pdf_url (l.115-123)"""
- skip_no_app()
- docs = client.get("/documents").json()
- if docs:
- for champ in ["id", "title", "ingredient", "source", "type", "date", "pdf_url"]:
- assert champ in docs[0], f"Champ manquant : {champ}"
-
- def test_source_cir_ou_sccs(self):
- """La source est 'CIR' ou 'SCCS' pour chaque document"""
- skip_no_app()
- docs = client.get("/documents").json()
- for doc in docs:
- assert doc["source"] in ["CIR", "SCCS"], f"Source inattendue : {doc['source']}"
-
-
-# ══════════════════════════════════════════════════════════════
-# PARTIE 3 — TESTS DE SÉCURITÉ
-# ══════════════════════════════════════════════════════════════
-
-class TestSecuriteInjectionSQL:
- """Tests d'injection SQL — SQLAlchemy protège nativement via requêtes paramétrées"""
-
- def test_injection_sql_dans_email_login(self):
- """' OR 1=1 -- dans l'email → 401 ou 422, JAMAIS 500"""
- skip_no_app()
- r = client.post("/login", json={"email": "' OR 1=1 --", "password": "test"})
- assert r.status_code in [401, 422], f"Possible injection SQL : {r.status_code}"
-
- def test_injection_sql_dans_email_signup(self):
- """'; DROP TABLE users; -- → 400 ou 422, JAMAIS 500"""
- skip_no_app()
- r = client.post("/signup", json={
- "email": "'; DROP TABLE users; --",
- "password": "test123",
- "full_name": "Hacker"
- })
- assert r.status_code in [400, 422], f"Possible injection SQL : {r.status_code}"
-
- def test_injection_sql_guillemets_doubles(self):
- """\" OR \"1\"=\"1 dans l'email → pas de 500"""
- skip_no_app()
- r = client.post("/login", json={"email": "\" OR \"1\"=\"1", "password": "test"})
- assert r.status_code != 500
-
-
-class TestSecuriteJWT:
- """Tests de sécurité JWT"""
-
- def test_token_falsifie_retourne_401(self):
- """JWT avec signature incorrecte → 401 'Invalid token'"""
- skip_no_app()
- r = client.get("/me?token=eyJhbGciOiJIUzI1NiJ9.faux.mauvaise_signature")
- assert r.status_code == 401
-
- def test_token_vide_retourne_401_ou_422(self):
- """Token vide → 401 ou 422"""
- skip_no_app()
- assert client.get("/me?token=").status_code in [401, 422]
-
- def test_token_malformed_retourne_401(self):
- """Token sans points (format JWT invalide) → 401"""
- skip_no_app()
- assert client.get("/me?token=cecinestunepasjwt").status_code == 401
-
- def test_token_valide_sur_bon_endpoint(self):
- """Un vrai token JWT permet l'accès à /me"""
- skip_no_app()
- email = f"jwt_ok_{int(time.time())}@regwatch.fr"
- client.post("/signup", json={"email": email, "password": "test123", "full_name": "T"})
- token = client.post("/login", json={"email": email, "password": "test123"}).json()["access_token"]
- assert client.get(f"/me?token={token}").status_code == 200
-
-
-class TestSecuriteXSS:
- """Tests XSS — l'API REST retourne du JSON, React échappe côté frontend"""
-
- def test_xss_dans_full_name_stocke_sans_crash(self):
- """Payload XSS dans full_name → accepté sans crash (200 ou 422)"""
- skip_no_app()
- email = f"xss_{int(time.time())}@regwatch.fr"
- r = client.post("/signup", json={
- "email": email,
- "password": "test123",
- "full_name": ""
- })
- assert r.status_code in [200, 422]
-
- def test_xss_dans_email_ne_plante_pas(self):
- """Payload XSS dans l'email → 422 (validation format email)"""
- skip_no_app()
- r = client.post("/login", json={
- "email": "@test.com",
- "password": "test"
- })
- assert r.status_code in [401, 422]
-
-
-# ══════════════════════════════════════════════════════════════
-# PARTIE 4 — CONFORMITÉ RGPD
-# ══════════════════════════════════════════════════════════════
-
-class TestRGPD:
- """
- Tests de conformité RGPD — exigés par le guide Nexa
- Vérifie que les données personnelles sont bien protégées
- """
-
- def test_mot_de_passe_absent_de_la_reponse_signup(self):
- """
- Le mot de passe ne doit JAMAIS apparaître dans la réponse API
- UserResponse = id + email + full_name SEULEMENT (pas de password)
- """
- skip_no_app()
- email = f"rgpd_{int(time.time())}@regwatch.fr"
- r = client.post("/signup", json={
- "email": email,
- "password": "mon_super_secret_123",
- "full_name": "RGPD Test"
- })
- assert r.status_code == 200
- # Le mot de passe en clair ne doit pas apparaître
- assert "mon_super_secret_123" not in r.text
- # Le champ password ne doit pas être dans la réponse
- assert "password" not in r.json()
-
- def test_mot_de_passe_absent_de_la_reponse_login(self):
- """Le login retourne uniquement access_token et token_type"""
- skip_no_app()
- email = f"rgpd2_{int(time.time())}@regwatch.fr"
- client.post("/signup", json={"email": email, "password": "secret456", "full_name": "T"})
- r = client.post("/login", json={"email": email, "password": "secret456"})
- assert r.status_code == 200
- assert "secret456" not in r.text
- data = r.json()
- assert "password" not in data
- assert set(data.keys()) <= {"access_token", "token_type"}
-
- def test_mot_de_passe_absent_de_la_reponse_me(self):
- """GET /me ne retourne pas le mot de passe"""
- skip_no_app()
- email = f"rgpd3_{int(time.time())}@regwatch.fr"
- client.post("/signup", json={"email": email, "password": "secret789", "full_name": "T"})
- token = client.post("/login", json={"email": email, "password": "secret789"}).json()["access_token"]
- r = client.get(f"/me?token={token}")
- assert r.status_code == 200
- assert "secret789" not in r.text
- assert "password" not in r.json()
-
- def test_hachage_argon2_dans_requirements(self):
- """
- La dépendance passlib[argon2] est bien dans requirements.txt
- → preuve que les mots de passe sont hachés avec Argon2
- """
- req_path = os.path.join(
- os.path.dirname(__file__), "..", "requirements.txt"
- )
- if os.path.exists(req_path):
- with open(req_path) as f:
- content = f.read()
- assert "argon2" in content.lower(), \
- "passlib[argon2] absent de requirements.txt"
- else:
- pytest.skip("requirements.txt non trouvé")
-
- def test_token_jwt_expire_apres_acces(self):
- """
- Un token JWT est bien structuré (3 parties séparées par des points)
- → preuve que l'expiration est encodée dans le payload
- """
- skip_no_app()
- email = f"expire_{int(time.time())}@regwatch.fr"
- client.post("/signup", json={"email": email, "password": "test123", "full_name": "T"})
- token = client.post("/login", json={"email": email, "password": "test123"}).json()["access_token"]
- # Un JWT valide a toujours 3 parties séparées par des points
- parts = token.split(".")
- assert len(parts) == 3, "Le token JWT n'a pas la structure header.payload.signature"
-
- def test_cors_autorise_frontend(self):
- """
- Le middleware CORS permet au frontend React de communiquer avec l'API
- allow_origins=["*"] configuré dans main.py l.13-18
- """
- skip_no_app()
- r = client.get("/", headers={"Origin": "http://localhost:5173"})
- assert r.status_code == 200
\ No newline at end of file