- texparse.py: extração de estrutura abntex2 (título, autores, resumo, palavras-chave, seções, citações, figuras), sem dependências externas - bib.py: parser .bib (BibTeX) com normalização de autores e diacríticos - analyze.py: checklist NBR 10520/6022/14724, métricas textuais, Flesch adaptado pt-BR e heurísticas de prosa - report.py: relatório Markdown + JSON - artigos/rumo-a-eficiencia: artigo analisado (main.tex + referencias.bib)
173 lines
5.8 KiB
Python
173 lines
5.8 KiB
Python
"""Gera o relatório em Markdown e o artefato JSON."""
|
||
|
||
from __future__ import annotations
|
||
|
||
import json
|
||
import re
|
||
from datetime import datetime, timezone
|
||
from pathlib import Path
|
||
|
||
from .analyze import Analysis
|
||
from .bib import Reference
|
||
from .texparse import count_words
|
||
|
||
|
||
def _md_escape(text: str) -> str:
|
||
return text.replace("|", "\\|")
|
||
|
||
|
||
def _ref_line(ref: Reference) -> str:
|
||
parts = [ref.year and ref.year, ref.title, ref.authors, ref.journal]
|
||
parts = [re.sub(r"\s*\n\s*", " ", p) for p in parts if p]
|
||
if not parts:
|
||
return ref.key
|
||
return "; ".join(parts) if len(parts) > 1 else parts[0]
|
||
|
||
|
||
def build_markdown(a: Analysis) -> str:
|
||
art, m = a.article, a.metrics
|
||
lines: list[str] = []
|
||
add = lines.append
|
||
|
||
add(f"# Análise do artigo")
|
||
add("")
|
||
add(f"> **{art.title}**")
|
||
add(f"> Autores: {', '.join(art.authors)} — {art.local}, {art.data} — {art.instituicao}")
|
||
add("")
|
||
|
||
resumo = m["resumo_checks"]
|
||
add(f"**Resultado global:** ✅ {resumo['ok']} ok · ⚠️ {resumo['warn']} aviso(s) · ❌ {resumo['fail']} falha(s) · 📝 {len(a.issues)} questão(ões) de prosa")
|
||
add("")
|
||
|
||
# 1. Estrutura ---------------------------------------------------------
|
||
add("## 1. Estrutura e elementos")
|
||
add("")
|
||
add("| # | Verificação | Status | Detalhe |")
|
||
add("|---|---|---|---|")
|
||
for c in a.checks:
|
||
add(f"| {c.code} | {c.label} | {c.icon} | {_md_escape(c.detail)} |")
|
||
add("")
|
||
|
||
# 2. Métricas textuais ------------------------------------------------
|
||
add("## 2. Métricas textuais")
|
||
add("")
|
||
add("| Métrica | Valor |")
|
||
add("|---|---|")
|
||
add(f"| Palavras (corpo) | {m['palavras_corpo']:,} |")
|
||
add(f"| Palavras (total, c/ resumo) | {m['palavras_totais']:,} |")
|
||
add(f"| Frases | {m['sentencas']} |")
|
||
add(f"| Média de palavras por frase | {m['media_palavras_por_sentenca']} |")
|
||
add(f"| Vocabulário único | {m['vocabulario_unico']:,} |")
|
||
add(f"| Legibilidade (Flesch adaptado) | {m['flesch']} — {m['flesch_faixa']} |")
|
||
add("")
|
||
|
||
add("### Distribuição por seção")
|
||
add("")
|
||
add("| Seção | Palavras | % do total |")
|
||
add("|---|---:|---:|")
|
||
for s in m["secoes"]:
|
||
add(f"| {s['titulo']} | {s['palavras']:,} | {s['percentual']}% |")
|
||
add("")
|
||
|
||
# 3. Citações ----------------------------------------------------------
|
||
stats = m["citacoes"]
|
||
add("## 3. Citações e referências (NBR 10520 / 6023)")
|
||
add("")
|
||
add(f"- Citações no texto: **{stats['total_citations']}** ({stats['unique_cited']} referências distintas)")
|
||
add(f"- Entradas no `.bib`: **{stats['defined_refs']}**")
|
||
if stats["orphan_citations"]:
|
||
add(f"- ❌ Citadas mas **ausentes do .bib**: {', '.join(stats['orphan_citations'])}")
|
||
if stats["unused_refs"]:
|
||
add(f"- ⚠️ No .bib mas **não citadas no texto**: {', '.join(stats['unused_refs'])}")
|
||
add("")
|
||
if stats["most_cited"]:
|
||
add("### Referências mais citadas")
|
||
add("")
|
||
add("| Referência | Cit.ªs | Entrada |")
|
||
add("|---|---:|---|")
|
||
for key, n, label in stats["most_cited"]:
|
||
add(f"| {_md_escape(label)} | {n} | `{key}` |")
|
||
add("")
|
||
if stats["year_distribution"]:
|
||
add("### Referências por ano")
|
||
add("")
|
||
years = sorted(stats["year_distribution"].items())
|
||
add("| Ano | N.º |")
|
||
add("|---:|---:|")
|
||
for year, n in years:
|
||
add(f"| {year} | {n} |")
|
||
add("")
|
||
|
||
# 4. Prosa -------------------------------------------------------------
|
||
add("## 4. Questões de prosa")
|
||
add("")
|
||
if a.issues:
|
||
for i in a.issues:
|
||
add(f"- {i}")
|
||
else:
|
||
add("Nenhuma questão detectada pelas heurísticas.")
|
||
add("")
|
||
if m["top_frases"]:
|
||
add("### Expressões mais repetidas (5 palavras)")
|
||
add("")
|
||
for phrase, n in m["top_frases"]:
|
||
add(f"- \"{phrase}\" — {n}×")
|
||
add("")
|
||
|
||
# 5. Referências -------------------------------------------------------
|
||
add("## 5. Referências do .bib")
|
||
add("")
|
||
for key in sorted(a.refs):
|
||
ref = a.refs[key]
|
||
cited = key in set(a.article.citations)
|
||
mark = "✅" if cited else "⚠️"
|
||
add(f"{mark} `{key}` — {_md_escape(_ref_line(ref))}")
|
||
add("")
|
||
|
||
add("---")
|
||
add(f"_Gerado em {datetime.now(timezone.utc):%Y-%m-%d %H:%M} UTC "
|
||
f"pelo projeto `analise-artigo` (v1.0)._")
|
||
return "\n".join(lines) + "\n"
|
||
|
||
|
||
def build_json(a: Analysis) -> str:
|
||
d = {
|
||
"article": {
|
||
"title": a.article.title,
|
||
"authors": a.article.authors,
|
||
"local": a.article.local,
|
||
"data": a.article.data,
|
||
"instituicao": a.article.instituicao,
|
||
"keywords": a.article.keywords,
|
||
"abstract_words": len(count_words(a.article.abstract)),
|
||
"sections": [
|
||
{"titulo": s.title, "nivel": s.level, "palavras": s.words}
|
||
for s in a.article.sections
|
||
],
|
||
"figures": a.article.figures,
|
||
"tables": a.article.tables,
|
||
},
|
||
"checks": [
|
||
{"code": c.code, "label": c.label, "status": c.status,
|
||
"detail": c.detail}
|
||
for c in a.checks
|
||
],
|
||
"metrics": a.metrics,
|
||
"issues": a.issues,
|
||
"references": {
|
||
k: {"type": r.type, "fields": r.fields,
|
||
"cited": k in set(a.article.citations)}
|
||
for k, r in a.refs.items()
|
||
},
|
||
}
|
||
return json.dumps(d, ensure_ascii=False, indent=2)
|
||
|
||
|
||
def write_reports(a: Analysis, outdir: Path) -> tuple[Path, Path]:
|
||
outdir.mkdir(parents=True, exist_ok=True)
|
||
md_path = outdir / "relatorio.md"
|
||
json_path = outdir / "relatorio.json"
|
||
md_path.write_text(build_markdown(a), encoding="utf-8")
|
||
json_path.write_text(build_json(a), encoding="utf-8")
|
||
return md_path, json_path
|