Projeto analise-artigo: análise ABNT de artigos LaTeX

- texparse.py: extração de estrutura abntex2 (título, autores, resumo,
  palavras-chave, seções, citações, figuras), sem dependências externas
- bib.py: parser .bib (BibTeX) com normalização de autores e diacríticos
- analyze.py: checklist NBR 10520/6022/14724, métricas textuais, Flesch
  adaptado pt-BR e heurísticas de prosa
- report.py: relatório Markdown + JSON
- artigos/rumo-a-eficiencia: artigo analisado (main.tex + referencias.bib)
This commit is contained in:
2026-09-25 13:24:56 -03:00
commit 550440f4a7
14 changed files with 1591 additions and 0 deletions

264
analise_artigo/texparse.py Normal file
View File

@@ -0,0 +1,264 @@
"""Extrai a estrutura textual de um documento LaTeX (abntex2/ABNT).
Converte o .tex em texto limpo preservando:
- título, autores, local/data (elementos pré-textuais);
- resumo e palavras-chave;
- seções/subseções e seus parágrafos;
- chaves de citação (\\cite, \\citeonline, \\citeauthor ...).
Não depende de pandoc: usa apenas a biblioteca padrão.
"""
from __future__ import annotations
import re
from dataclasses import dataclass, field
from pathlib import Path
@dataclass
class Section:
"""Seção do artigo com o texto limpo já convertido para prosa."""
title: str
level: int = 1
@property
def words(self) -> int:
return len(count_words(self.text))
@dataclass
class Article:
title: str = ""
authors: list[str] = field(default_factory=list)
local: str = ""
data: str = ""
instituicao: str = ""
abstract: str = ""
keywords: list[str] = field(default_factory=list)
sections: list[Section] = field(default_factory=list)
citations: list[str] = field(default_factory=list) # ordem de aparição
figures: list[str] = field(default_factory=list) # captions das figuras
tables: list[str] = field(default_factory=list) # captions de quadros
text_words: int = 0
def section(self, title: str) -> Section | None:
for s in self.sections:
if s.title.strip().lower() == title.strip().lower():
return s
return None
# ---------------------------------------------------------------------------
# Conversão LaTeX -> texto
# ---------------------------------------------------------------------------
_BRACE_CMDS = ["textbf", "emph", "textit", "texttt", "underline",
"bold", "itshape", "scshape", "MakeUppercase", "MakeLowercase"]
_CITE_RE = re.compile(r"\\cite\w*\{([^}]*)\}")
_BRACE_CMD_RE = {
cmd: re.compile(r"\\" + cmd + r"\{([^{}]*)\}") for cmd in _BRACE_CMDS
}
# Chaves com barra escapada no LaTeX (ex.: "\_", "\#").
# "\," "\;" "\%" "\$" "\&" já são tratados pela regex de duas linhas acima.
_ESCAPE_MAP = {
"\\_": "_",
"\\#": "#",
"\\\\": "\n",
}
def count_words(text: str) -> list[str]:
return re.findall(
r"[A-Za-z0-9À-ÖØ-öø-ÿ]+(?:['’-][A-Za-z0-9À-ÖØ-öø-ÿ]+)*", text
)
def _fold_brace_commands(s: str) -> str:
changed = True
while changed:
changed = False
for cmd, pat in _BRACE_CMD_RE.items():
s2, n = pat.subn(r"\1", s)
if n:
s, changed = s2, True
return s
def strip_latex(text: str, citations: list[str] | None = None) -> str:
"""Remove comandos e ambientes, preservando o conteúdo legível."""
# citações: captura as chaves e remove do texto
def _cite(m: re.Match) -> str:
if citations is not None:
for key in m.group(1).split(","):
key = key.strip()
if key:
citations.append(key)
return " "
text = _CITE_RE.sub(_cite, text)
# rótulos/referências internas e figuras
text = re.sub(r"\\label\{[^}]*\}", " ", text)
text = re.sub(r"\\(ref|pageref|eqref)\{[^}]*\}", " (ref.) ", text)
text = re.sub(r"\\includegraphics.*", " ", text)
text = re.sub(r"\\noindent", " ", text)
# comandos com argumento em chaves genéricos (legend, caption, textbf...)
text = _fold_brace_commands(text)
text = re.sub(r"\\(legend|caption|text|footnote)\{[^{}]*(?:\{[^{}]*\}[^{}]*)*\}", " ", text)
# ambientes: manter marcadores de item, eliminar o resto
text = re.sub(r"\\begin\{itemize\}", "\n- ", text)
text = re.sub(r"\\begin\{enumerate\}", "\n1. ", text)
text = re.sub(r"\\item\b", "\n- ", text)
text = re.sub(r"\\(end|begin)\{[^}]*\}", " ", text)
text = re.sub(r"\\toprule|\\midrule|\\bottomrule", " ", text)
text = text.replace("\\\\", "\n") # quebras de linha (\\) em tabular/autor
# comandos restantes sem argumento
text = re.sub(r"\\[a-zA-Z]+\*?", " ", text)
text = re.sub(r"\\[,;:\\!%$&]", " ", text)
for esc, ch in _ESCAPE_MAP.items():
text = text.replace(esc, ch)
# aspas e travessões
text = text.replace("``", '"').replace("''", '"')
text = text.replace("---", " — ").replace("--", " – ")
# tabular: & vira coluna
text = text.replace("&", " | ")
# limpar pontuação e espaços
text = re.sub(r"\[[a-z]*\]", " ", text) # posicionadores [htb]
text = re.sub(r"[{}]+", " ", text)
text = re.sub(r"[ \t]+", " ", text)
text = re.sub(r" +([,.;:!?])", r"\1", text)
text = re.sub(r"\n\s*", "\n", text)
text = re.sub(r"\n{3,}", "\n\n", text)
return text.strip()
# ---------------------------------------------------------------------------
# Parsing da estrutura
# ---------------------------------------------------------------------------
_SECTION_RE = re.compile(r"^\s*\\(section|subsection|subsubsection)\*?\s*\{([^}]+)\}")
def _capture_command(text: str, name: str) -> str:
"""Extrai o último conteúdo de \\name{...} ou \\<name> ... \\end{<name>}."""
m = re.search(r"\\" + name + r"\s*\{", text)
if not m:
return ""
start = m.end()
depth = 1
i = start
while i < len(text) and depth:
if text[i] == "{":
depth += 1
elif text[i] == "}":
depth -= 1
i += 1
return text[start : i - 1].strip()
def _extract_authors(text: str) -> list[str]:
raw = _capture_command(text, "autor")
if not raw:
raw = _capture_command(text, "author")
return [a.strip() for a in re.split(r"\\\\|\\[ ]", raw) if a.strip()]
def parse_article(tex_path: Path | str) -> Article:
tex_path = Path(tex_path)
text = tex_path.read_text(encoding="utf-8")
# remove comentários de linha (não escapados)
text = re.sub(r"(?<!\\)%[^\n]*", "", text)
citations: list[str] = []
raw_sections: list[tuple[int, str, list[str]]] = []
title = strip_latex(_capture_command(text, "titulo")
or _capture_command(text, "title"))
authors = _extract_authors(text)
local = strip_latex(_capture_command(text, "local"))
data = strip_latex(_capture_command(text, "data"))
instituicao = strip_latex(_capture_command(text, "instituicao")
or _capture_command(text, "institution"))
# resumo (ambiente resumoumacoluna do abntex2)
m = re.search(r"\\begin\{resumoumacoluna\}(.*?)\\end\{resumoumacoluna\}",
text, re.S)
abstract_raw = m.group(1) if m else ""
abstract = strip_latex(abstract_raw, citations)
keywords: list[str] = []
km = re.search(
r"Palavras-chave[: ]*\s*(.+)", abstract
)
if km:
kw_body = strip_latex(abstract)
kw_text = kw_body.split("Palavras-chave", 1)[-1]
kw_text = kw_text.split(":", 1)[-1].strip()
keywords = [k.strip() for k in re.split(r"[.;]", kw_text) if k.strip()]
# seções: percorre linha a linha acumulando o corpo de cada seção
current: tuple[int, str, list[str]] | None = None
preamble_text = ""
figures: list[str] = []
tables: list[str] = []
for line in text.splitlines():
sm = _SECTION_RE.match(line)
if sm:
level = {"section": 1, "subsection": 2, "subsubsection": 3}[sm.group(1)]
title_s = strip_latex(sm.group(2), citations)
if current:
raw_sections.append(current)
current = (level, title_s, [])
continue
if current is None:
preamble_text += line + "\n"
continue
current[2].append(line)
if current:
raw_sections.append(current)
article = Article(
title=title,
authors=authors,
local=local,
data=data,
instituicao=instituicao,
abstract=abstract,
keywords=keywords,
)
for level, title_s, lines in raw_sections:
sec = Section(title=title_s, level=level)
body = "\n".join(lines)
# captura captions de figuras/quadros antes de remover
for cap in re.findall(r"\\caption\{([^}]+)\}", body):
if "\\includegraphics" in body or "figura" in body.lower():
figures.append(strip_latex(cap))
else:
tables.append(strip_latex(cap))
sec.text = strip_latex(body, citations)
article.sections.append(sec)
article.citations = citations
article.figures = figures
article.tables = tables
article.text_words = (
len(count_words(title))
+ len(count_words(abstract))
+ sum(len(count_words(s.text)) for s in article.sections)
)
return article