loko/streetup/interventions/management/commands/export_strikes_interventions.py
2026-07-22 14:48:40 +02:00

338 lines
12 KiB
Python

"""
Management command to export interventions mentioning strikes/demonstrations/protests.
Searches for keywords related to strikes, protests, and demonstrations in intervention notes,
with support for abbreviations and accent-insensitive matching.
"""
from django.core.management.base import BaseCommand, CommandError
from django.db.models import Prefetch
from datetime import datetime
import openpyxl
from openpyxl.styles import Font, PatternFill, Alignment, Border, Side
import re
import unicodedata
from decimal import Decimal
def normalize_text(text):
"""
Normalize text by removing accents and converting to lowercase.
This allows matching 'grève', 'greve', 'Grève', etc. as the same word.
"""
if not text:
return ""
# Remove accents
text = unicodedata.normalize('NFD', text)
text = ''.join(char for char in text if unicodedata.category(char) != 'Mn')
# Convert to lowercase
return text.lower()
def get_search_keywords():
"""Return normalized keywords for strike/protest/demonstration mentions."""
keywords = [
'manifestation', 'manif', 'manifest',
'grève', 'greve', 'grev', 'grevistes',
'strike', 'staking',
#'demonstratie', 'demo',
'betoging',
'protest', 'protestation',
#'blocage', 'blocus',
'cortège', 'cortege',
]
normalized_keywords = []
for kw in keywords:
normalized = normalize_text(kw)
if normalized and normalized not in normalized_keywords:
normalized_keywords.append(normalized)
return normalized_keywords
def contains_strike_keywords(text):
"""
Check if text contains any strike/protest/demonstration keywords.
Returns the list of matched keywords if found, empty list otherwise.
"""
if not text:
return []
normalized_text = normalize_text(text)
keywords = get_search_keywords()
matched_keywords = []
for keyword in keywords:
if re.search(rf'(?<!\w){re.escape(keyword)}(?!\w)', normalized_text):
matched_keywords.append(keyword)
return matched_keywords
def _sanitize_cell_value(value):
"""Remove illegal characters from cell values for Excel."""
if isinstance(value, str):
# Remove control characters except tab, newline, carriage return
return re.sub(r'[\x00-\x08\x0b\x0c\x0e-\x1f]', '', value)
return value
class Command(BaseCommand):
help = 'Export interventions mentioning strikes, protests, or demonstrations to Excel'
def add_arguments(self, parser):
parser.add_argument(
'--output',
type=str,
default='strikes_interventions.xlsx',
help='Output file path for the Excel export',
)
parser.add_argument(
'--start-date',
type=str,
help='Start date in YYYY-MM-DD format',
)
parser.add_argument(
'--end-date',
type=str,
help='End date in YYYY-MM-DD format',
)
def handle(self, *args, **options):
from interventions.models import Intervention, InterventionNote, InterventionContractPost
output_file = options['output']
start_date = None
end_date = None
# Parse dates if provided
if options['start_date']:
try:
start_date = datetime.strptime(options['start_date'], '%Y-%m-%d').date()
except ValueError:
raise CommandError(f"Invalid start date format: {options['start_date']}")
if options['end_date']:
try:
end_date = datetime.strptime(options['end_date'], '%Y-%m-%d').date()
except ValueError:
raise CommandError(f"Invalid end date format: {options['end_date']}")
# Get all notes with potential strike mentions
self.stdout.write("Searching for interventions with strike/protest mentions...")
interventions_with_strikes = set()
strike_notes_map = {} # intervention_id -> list of matching notes
strike_titles_map = {} # intervention_id -> matched title keywords
# Get all notes with potential keywords
all_notes = InterventionNote.objects.select_related('intervention').all()
for note in all_notes:
matched_keywords = contains_strike_keywords(note.content)
if matched_keywords:
interventions_with_strikes.add(note.intervention_id)
if note.intervention_id not in strike_notes_map:
strike_notes_map[note.intervention_id] = []
strike_notes_map[note.intervention_id].append({
'note': note,
'keywords': matched_keywords
})
# Match keywords in intervention titles as well
for intervention_id, title in Intervention.objects.values_list('id', 'title').iterator():
matched_title_keywords = contains_strike_keywords(title)
if matched_title_keywords:
interventions_with_strikes.add(intervention_id)
strike_titles_map[intervention_id] = matched_title_keywords
# Get interventions with their related data
queryset = Intervention.objects.filter(
id__in=interventions_with_strikes
).select_related(
'symptom',
'contract',
'contract__company',
'source_category',
'created_by',
'thematic',
'asset_category'
).prefetch_related(
Prefetch(
'notes',
queryset=InterventionNote.objects.order_by('note_time')
),
Prefetch(
'interv_contract_posts',
queryset=InterventionContractPost.objects.select_related('contract_post')
)
)
# Apply date filters
if start_date:
queryset = queryset.filter(begin_time__date__gte=start_date)
if end_date:
queryset = queryset.filter(begin_time__date__lte=end_date)
queryset = queryset.order_by('-begin_time')
interventions = list(queryset)
self.stdout.write(
self.style.SUCCESS(f"Found {len(interventions)} intervention(s) with strike mentions")
)
if not interventions:
self.stdout.write(self.style.WARNING("No interventions found"))
return
# Create Excel workbook
wb = openpyxl.Workbook()
ws = wb.active
ws.title = "Interventions"
# Set up header row
headers = [
"Code",
"Title",
"Date",
"Begin Time",
"End Time",
"Status",
"Location (Address)",
"Latitude",
"Longitude",
"Total Billed Amount",
"Notes with Keywords",
"All Notes",
]
# Style header
header_fill = PatternFill(start_color="4472C4", end_color="4472C4", fill_type="solid")
header_font = Font(bold=True, color="FFFFFF")
thin_border = Border(
left=Side(style='thin'),
right=Side(style='thin'),
top=Side(style='thin'),
bottom=Side(style='thin')
)
ws.append(headers)
for cell in ws[1]:
cell.fill = header_fill
cell.font = header_font
cell.alignment = Alignment(wrap_text=True, vertical='top')
cell.border = thin_border
# Add data rows
for intervention in interventions:
# Calculate billed amount
total_amount = Decimal('0')
for contract_post in intervention.interv_contract_posts.all():
try:
total_amount += contract_post.total_price or Decimal('0')
except Exception:
pass
# Get location info
address = intervention.address or ""
lat = intervention.lat if intervention.lat else ""
lon = intervention.lon if intervention.lon else ""
# Get begin date
begin_time = intervention.begin_time or intervention.planned_begin_time or intervention.creation_time
begin_date = begin_time.strftime("%Y-%m-%d") if begin_time else ""
begin_time_str = begin_time.strftime("%H:%M") if begin_time else ""
# Get end time
end_time_str = ""
if intervention.end_time:
end_time_str = intervention.end_time.strftime("%H:%M")
# Get strike notes with keywords
strike_notes_text = ""
if intervention.id in strike_titles_map:
title_keywords = ", ".join(sorted(set(strike_titles_map[intervention.id])))
strike_notes_text = f"[TITLE: {title_keywords}] {_sanitize_cell_value(intervention.title or '')}"
if intervention.id in strike_notes_map:
strike_notes_list = []
for note_info in strike_notes_map[intervention.id]:
note = note_info['note']
keywords = note_info['keywords']
keywords_str = ", ".join(sorted(set(keywords)))
note_text = _sanitize_cell_value(note.content or "")
strike_notes_list.append(f"[{keywords_str}] {note_text}")
notes_block = "\n---\n".join(strike_notes_list)
if strike_notes_text:
strike_notes_text = f"{strike_notes_text}\n---\n{notes_block}"
else:
strike_notes_text = notes_block
# Get all notes combined
all_notes_text = ""
if intervention.notes.exists():
notes_list = []
for note in intervention.notes.all():
note_text = _sanitize_cell_value(note.content or "")
note_type = note.get_note_type_display() if hasattr(note, 'get_note_type_display') else note.note_type
notes_list.append(f"[{note_type}] {note_text}")
all_notes_text = "\n---\n".join(notes_list)
row = [
_sanitize_cell_value(intervention.code or ""),
_sanitize_cell_value(intervention.title or ""),
begin_date,
begin_time_str,
end_time_str,
intervention.status or "",
_sanitize_cell_value(address),
lat,
lon,
str(total_amount) if total_amount > 0 else "",
_sanitize_cell_value(strike_notes_text),
_sanitize_cell_value(all_notes_text),
]
ws.append(row)
# Style cells
for cell in ws[ws.max_row]:
cell.border = thin_border
cell.alignment = Alignment(wrap_text=True, vertical='top')
# Adjust column widths
column_widths = {
'A': 12, # Code
'B': 30, # Title
'C': 12, # Date
'D': 10, # Begin Time
'E': 10, # End Time
'F': 12, # Status
'G': 25, # Address
'H': 12, # Latitude
'I': 12, # Longitude
'J': 15, # Total Billed Amount
'K': 35, # Notes with Keywords
'L': 35, # All Notes
}
for col, width in column_widths.items():
ws.column_dimensions[col].width = width
# Freeze header row
ws.freeze_panes = "A2"
# Save file
try:
wb.save(output_file)
self.stdout.write(
self.style.SUCCESS(f"✓ Successfully exported to {output_file}")
)
self.stdout.write(f" - {len(interventions)} intervention(s) exported")
self.stdout.write(f" - Keywords searched: {', '.join(get_search_keywords())}")
except Exception as e:
raise CommandError(f"Failed to save file: {str(e)}")