media-profiling / generate_summary.py
Miras Baisbay
results of 6 modes in 30 samples
b7dc3d4
Raw
History Blame Contribute Delete
3.28 kB
import os
import glob
import re
from tabulate import tabulate
def generate_summary_from_logs(results_dir="results"):
# Find all .log files in the results directory
log_files = glob.glob(os.path.join(results_dir, "*.log"))
if not log_files:
print(f"No .log files found in the '{results_dir}' directory.")
return
summary_data = []
# Regex to capture the summary header: "--- model_name x mode_name SUMMARY ---"
header_pattern = re.compile(r"---\s+(.*?)\s+x\s+(.*?)\s+SUMMARY\s+---")
for filepath in log_files:
with open(filepath, 'r', encoding='utf-8') as f:
content = f.read()
# Find all summary blocks (usually just one per file, but this handles multiple safely)
for match in header_pattern.finditer(content):
model_name = match.group(1).strip()
mode_name = match.group(2).strip()
# Slice the log content from the header onwards to search for metrics
block = content[match.end():]
# Helper function to extract specific metric values
def extract_val(key, is_pct=False):
# Matches numbers, N/A, None, or NaN
pattern = rf"{key}:\s*([0-9\.]+|N/A|None|nan|NaN)"
m = re.search(pattern, block, re.IGNORECASE)
if m:
val = m.group(1).upper()
if val in ("N/A", "NONE", "NAN"):
return "N/A"
# Add percent sign if applicable, otherwise just return the number
return f"{m.group(1)}%" if is_pct else m.group(1)
return "N/A"
# Extract the exact keys from your log format
n_samples = extract_val("n_samples")
fs_prec = extract_val("factscore_precision_pct", is_pct=True)
fs_recall = extract_val("factscore_recall_pct", is_pct=True)
err_rate = extract_val("error_rate_pct", is_pct=True)
meteor = extract_val("meteor_pct", is_pct=True)
rouge_l = extract_val("rougeL_pct", is_pct=True)
bias_mae = extract_val("bias_mae")
fact_mae = extract_val("factuality_mae")
fc_det = extract_val("fc_detection_pct", is_pct=True)
summary_data.append([
model_name, mode_name, n_samples,
fs_prec, fs_recall, err_rate,
meteor, rouge_l, bias_mae, fact_mae, fc_det
])
# Sort alphabetically by Model, then by Mode
summary_data.sort(key=lambda x: (x[0], x[1]))
# Exact headers requested
headers = [
"Model", "Mode", "N",
"FS Prec.", "FS Recall", "Err Rate",
"METEOR", "ROUGE-L", "Bias MAE", "Fact. MAE", "FC Det."
]
# Write output to TSV
summary_path = os.path.join(results_dir, "summary_table.tsv")
tsv_content = tabulate(summary_data, headers=headers, tablefmt="tsv")
with open(summary_path, "w", encoding='utf-8') as f:
f.write(tsv_content)
# Print nicely formatted table to the console
print(f"\nGenerated summary table from logs at: {summary_path}\n")
print(tabulate(summary_data, headers=headers, tablefmt="plain"))
if __name__ == "__main__":
generate_summary_from_logs()