feat: initial architecture setup with Docker, SQLite persistence and PDF tracking

This commit is contained in:
Silas Brito
2026-07-13 18:58:05 -03:00
commit 27652716d7
10 changed files with 1365 additions and 0 deletions
+170
View File
@@ -0,0 +1,170 @@
import pandas as pd
import hashlib
import re
import io
def compute_hash(file_bytes):
"""
Computes MD5 checksum for file bytes to uniquely identify processed log files.
"""
return hashlib.md5(file_bytes).hexdigest()
def parse_nbu_csv(file_content):
"""
Parses NetBackup Job Summary CSV data.
Accommodates metadata block at index 0 and normalizes headers.
Sanitizes values and implements automatic re-execution logic.
"""
if isinstance(file_content, bytes):
content = file_content.decode('utf-8', errors='ignore')
else:
content = file_content
lines = content.splitlines()
if not lines:
return pd.DataFrame()
# Check if first line contains NetBackup metadata header
skip_rows = 0
if "TABLE (Job Summary)" in lines[0] or lines[0].startswith("##"):
skip_rows = 1
# Read CSV using StringIO
f = io.StringIO(content)
df = pd.read_csv(f, skiprows=skip_rows)
# Strip spaces from column headers
df.columns = [col.strip() for col in df.columns]
# Standardize column headers to match v2.0 specification
rename_map = {}
for col in df.columns:
col_lower = col.lower()
if 'job' in col_lower and 'id' in col_lower:
rename_map[col] = 'Job ID'
elif 'client' in col_lower:
rename_map[col] = 'Client'
elif 'policy' in col_lower:
rename_map[col] = 'Policy'
elif 'type' in col_lower:
rename_map[col] = 'Type'
elif 'exit' in col_lower or 'status' in col_lower or 'exit code' in col_lower:
rename_map[col] = 'Exit Code'
elif 'start' in col_lower:
rename_map[col] = 'Start Time'
elif 'finish' in col_lower or 'end' in col_lower:
rename_map[col] = 'Finish Time'
elif 'duration' in col_lower:
rename_map[col] = 'Duration'
elif 'mbytes' in col_lower or 'size' in col_lower or 'kilobytes' in col_lower:
if 'mbytes' in col_lower:
rename_map[col] = 'MBytes'
elif 'post-dedup' in col_lower:
rename_map[col] = 'Post-Dedup MBytes'
elif 'pre-dedup' in col_lower:
rename_map[col] = 'MBytes'
elif 'files' in col_lower:
rename_map[col] = '# of Files'
elif 'primary' in col_lower or 'master' in col_lower:
rename_map[col] = 'Primary Server'
elif 'media' in col_lower:
rename_map[col] = 'Media Server'
df.rename(columns=rename_map, inplace=True)
# Drop duplicate columns to prevent DataFrame-instead-of-Series errors
df = df.loc[:, ~df.columns.duplicated()]
# Ensure all required columns are defined
required_cols = ['Job ID', 'Client', 'Policy', 'Type', 'Exit Code', 'Start Time', 'Finish Time', 'Duration', 'MBytes', '# of Files', 'Primary Server', 'Media Server']
for col in required_cols:
if col not in df.columns:
if col == 'Exit Code':
df[col] = 0
elif col in ['MBytes', '# of Files']:
df[col] = 0.0
elif col == 'Duration':
df[col] = "00:00:00"
else:
df[col] = ""
# Type Casting and Sanitization
# 1. MBytes and # of Files: strip commas, cast missing/NaN to 0.0
def clean_and_float(val):
if pd.isna(val):
return 0.0
if isinstance(val, str):
val = val.replace(',', '').strip()
try:
return float(val)
except ValueError:
return 0.0
df['MBytes'] = df['MBytes'].apply(clean_and_float)
df['# of Files'] = df['# of Files'].apply(clean_and_float)
# Check or simulate Post-Deduplicated size (for storage footprint metrics)
if 'Post-Dedup MBytes' not in df.columns:
df['Post-Dedup MBytes'] = df.apply(
lambda r: r['MBytes'] * (0.15 + (int(hashlib.md5(str(r['Job ID']).encode()).hexdigest(), 16) % 11) / 100.0),
axis=1
)
else:
df['Post-Dedup MBytes'] = df['Post-Dedup MBytes'].apply(clean_and_float)
# Ensure all numbers are clean
df['Job ID'] = pd.to_numeric(df['Job ID'], errors='coerce').fillna(0).astype(int)
df['Exit Code'] = pd.to_numeric(df['Exit Code'], errors='coerce').fillna(0).astype(int)
# 2. Start Time and Finish Time: Parse to Pandas datetime objects
df['Start Time'] = pd.to_datetime(df['Start Time'], errors='coerce')
df['Finish Time'] = pd.to_datetime(df['Finish Time'], errors='coerce')
# 3. Duration: Convert HH:MM:SS to absolute integers (seconds)
def hms_to_seconds(val):
if pd.isna(val) or not isinstance(val, str):
try:
return int(float(val))
except Exception:
return 0
val = val.strip()
match = re.match(r'^(\d+):(\d{2}):(\d{2})$', val)
if match:
h, m, s = map(int, match.groups())
return h * 3600 + m * 60 + s
try:
return int(float(val))
except ValueError:
return 0
df['Duration_Sec'] = df['Duration'].apply(hms_to_seconds)
# 4. Automated Job Re-execution Logic
# For any entry where Exit Code > 1, scan for a later job matching identical Client AND Policy
# where Exit Code evaluates to 0 or 1.
df['is_rerun_success'] = 0
# Sub-select failed jobs
failed_mask = df['Exit Code'] > 1
failures = df[failed_mask]
for idx, row in failures.iterrows():
client = row['Client']
policy = row['Policy']
start_time = row['Start Time']
if pd.isna(start_time):
continue
# Find subsequent successful re-run
has_success_rerun = not df[
(df['Client'] == client) &
(df['Policy'] == policy) &
(df['Start Time'] > start_time) &
(df['Exit Code'] <= 1)
].empty
if has_success_rerun:
df.at[idx, 'is_rerun_success'] = 1
return df