Initial commit

This commit is contained in:
jeremy bayse
2026-06-22 16:40:24 +02:00
commit ad127156ff
78 changed files with 13478 additions and 0 deletions
+185
View File
@@ -0,0 +1,185 @@
import sys
import os
import json
# Force stdout to UTF-8 encoding
if hasattr(sys.stdout, 'reconfigure'):
sys.stdout.reconfigure(encoding='utf-8')
import base64
import glob
# Add user site-packages to sys.path (needed on Windows when executed from PHP server environment)
user_site_packages = os.path.expanduser("~\\AppData\\Roaming\\Python\\Python313\\site-packages")
if os.path.exists(user_site_packages):
sys.path.append(user_site_packages)
# Absolute path fallbacks for Windows User profile
fallback_path = "C:\\Users\\jerem\\AppData\\Roaming\\Python\\Python313\\site-packages"
if os.path.exists(fallback_path) and fallback_path not in sys.path:
sys.path.append(fallback_path)
for path in glob.glob(os.path.expanduser("~\\AppData\\Roaming\\Python\\Python*\\site-packages")):
if path not in sys.path:
sys.path.append(path)
# Also scan C:\Users just in case the home directory is mapped differently
for path in glob.glob("C:\\Users\\*\\AppData\\Roaming\\Python\\Python*\\site-packages"):
if path not in sys.path:
sys.path.append(path)
import fitz # PyMuPDF
import requests
def extract_image_description(image_bytes, api_key):
"""Call Mistral Vision API (pixtral-12b-2409) to describe an image."""
if not api_key:
return ""
url = "https://api.mistral.ai/v1/chat/completions"
headers = {
"Authorization": f"Bearer {api_key}",
"Content-Type": "application/json"
}
# Base64 encode the image
base64_image = base64.b64encode(image_bytes).decode('utf-8')
payload = {
"model": "pixtral-12b-2409",
"messages": [
{
"role": "user",
"content": [
{
"type": "text",
"text": "Décris cette image, ce pictogramme ou ce schéma en une phrase descriptive claire pour un moteur de recherche."
},
{
"type": "image_url",
"image_url": f"data:image/jpeg;base64,{base64_image}"
}
]
}
],
"temperature": 0.2,
"max_tokens": 150
}
try:
response = requests.post(url, json=payload, headers=headers, timeout=15)
if response.status_code == 200:
result = response.json()
return result['choices'][0]['message']['content'].strip()
else:
return f"[Erreur Vision Status {response.status_code}]"
except Exception as e:
return f"[Erreur Vision: {str(e)}]"
def chunk_text(text, chunk_size=800, overlap=120):
"""Split text into overlapping chunks."""
words = text.split()
chunks = []
current_chunk = []
current_len = 0
for word in words:
current_chunk.append(word)
current_len += len(word) + 1
if current_len >= chunk_size:
chunks.append(" ".join(current_chunk))
# Slide window back
overlap_words = int(overlap / 6) # Approx 6 chars/word
if overlap_words > 0 and len(current_chunk) > overlap_words:
current_chunk = current_chunk[-overlap_words:]
current_len = len(" ".join(current_chunk))
else:
current_chunk = []
current_len = 0
if current_chunk:
chunks.append(" ".join(current_chunk))
return [c for c in chunks if len(c.strip()) > 30]
def main():
if len(sys.argv) < 4:
print(json.dumps({"error": "Missing arguments. Usage: python parse_pdf.py <pdf_path> <mistral_api_key> <enable_vision>"}))
sys.exit(1)
pdf_path = sys.argv[1]
api_key = sys.argv[2]
enable_vision = sys.argv[3] == '1'
if not os.path.exists(pdf_path):
print(json.dumps({"error": f"File not found: {pdf_path}"}))
sys.exit(1)
filename = os.path.basename(pdf_path)
output_chunks = []
try:
doc = fitz.open(pdf_path)
for page_num in range(len(doc)):
page = doc[page_num]
# Extract page text preserving block layouts
page_text = page.get_text("blocks")
# Sort blocks top-to-bottom, left-to-right
page_text.sort(key=lambda b: (b[1], b[0]))
text_lines = []
for b in page_text:
if len(b) > 4 and isinstance(b[4], str) and b[4].strip():
text_lines.append(b[4].strip())
full_page_text = "\n".join(text_lines)
# Extract images if vision is enabled
image_descriptions = []
if enable_vision:
images = page.get_images(full=True)
for img_idx, img in enumerate(images):
xref = img[0]
try:
base_image = doc.extract_image(xref)
image_bytes = base_image["image"]
# Skip tiny images (e.g., icons, bullets) to save API calls
if len(image_bytes) < 4000:
continue
desc = extract_image_description(image_bytes, api_key)
if desc:
image_descriptions.append(f"[Illustration Page {page_num+1} - Description: {desc}]")
except Exception as e:
# Log error internally but continue
pass
# Combine text and image descriptions
combined_text = full_page_text
if image_descriptions:
combined_text += "\n\nDescriptions d'illustrations sur cette page:\n" + "\n".join(image_descriptions)
# Chunk the page text
page_chunks = chunk_text(combined_text)
for chunk in page_chunks:
output_chunks.append({
"content": chunk,
"source": filename,
"title": f"{filename} (Page {page_num+1})"
})
doc.close()
print(json.dumps(output_chunks, ensure_ascii=False))
except Exception as e:
print(json.dumps({"error": f"Failed to parse PDF: {str(e)}"}))
sys.exit(1)
if __name__ == "__main__":
main()