Files
Jário José 550440f4a7 Projeto analise-artigo: análise ABNT de artigos LaTeX
- texparse.py: extração de estrutura abntex2 (título, autores, resumo,
  palavras-chave, seções, citações, figuras), sem dependências externas
- bib.py: parser .bib (BibTeX) com normalização de autores e diacríticos
- analyze.py: checklist NBR 10520/6022/14724, métricas textuais, Flesch
  adaptado pt-BR e heurísticas de prosa
- report.py: relatório Markdown + JSON
- artigos/rumo-a-eficiencia: artigo analisado (main.tex + referencias.bib)
2026-09-25 13:24:56 -03:00

265 lines
8.5 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Extrai a estrutura textual de um documento LaTeX (abntex2/ABNT).
Converte o .tex em texto limpo preservando:
- título, autores, local/data (elementos pré-textuais);
- resumo e palavras-chave;
- seções/subseções e seus parágrafos;
- chaves de citação (\\cite, \\citeonline, \\citeauthor ...).
Não depende de pandoc: usa apenas a biblioteca padrão.
"""
from __future__ import annotations
import re
from dataclasses import dataclass, field
from pathlib import Path
@dataclass
class Section:
"""Seção do artigo com o texto limpo já convertido para prosa."""
title: str
level: int = 1
@property
def words(self) -> int:
return len(count_words(self.text))
@dataclass
class Article:
title: str = ""
authors: list[str] = field(default_factory=list)
local: str = ""
data: str = ""
instituicao: str = ""
abstract: str = ""
keywords: list[str] = field(default_factory=list)
sections: list[Section] = field(default_factory=list)
citations: list[str] = field(default_factory=list) # ordem de aparição
figures: list[str] = field(default_factory=list) # captions das figuras
tables: list[str] = field(default_factory=list) # captions de quadros
text_words: int = 0
def section(self, title: str) -> Section | None:
for s in self.sections:
if s.title.strip().lower() == title.strip().lower():
return s
return None
# ---------------------------------------------------------------------------
# Conversão LaTeX -> texto
# ---------------------------------------------------------------------------
_BRACE_CMDS = ["textbf", "emph", "textit", "texttt", "underline",
"bold", "itshape", "scshape", "MakeUppercase", "MakeLowercase"]
_CITE_RE = re.compile(r"\\cite\w*\{([^}]*)\}")
_BRACE_CMD_RE = {
cmd: re.compile(r"\\" + cmd + r"\{([^{}]*)\}") for cmd in _BRACE_CMDS
}
# Chaves com barra escapada no LaTeX (ex.: "\_", "\#").
# "\," "\;" "\%" "\$" "\&" já são tratados pela regex de duas linhas acima.
_ESCAPE_MAP = {
"\\_": "_",
"\\#": "#",
"\\\\": "\n",
}
def count_words(text: str) -> list[str]:
return re.findall(
r"[A-Za-z0-9À-ÖØ-öø-ÿ]+(?:['’-][A-Za-z0-9À-ÖØ-öø-ÿ]+)*", text
)
def _fold_brace_commands(s: str) -> str:
changed = True
while changed:
changed = False
for cmd, pat in _BRACE_CMD_RE.items():
s2, n = pat.subn(r"\1", s)
if n:
s, changed = s2, True
return s
def strip_latex(text: str, citations: list[str] | None = None) -> str:
"""Remove comandos e ambientes, preservando o conteúdo legível."""
# citações: captura as chaves e remove do texto
def _cite(m: re.Match) -> str:
if citations is not None:
for key in m.group(1).split(","):
key = key.strip()
if key:
citations.append(key)
return " "
text = _CITE_RE.sub(_cite, text)
# rótulos/referências internas e figuras
text = re.sub(r"\\label\{[^}]*\}", " ", text)
text = re.sub(r"\\(ref|pageref|eqref)\{[^}]*\}", " (ref.) ", text)
text = re.sub(r"\\includegraphics.*", " ", text)
text = re.sub(r"\\noindent", " ", text)
# comandos com argumento em chaves genéricos (legend, caption, textbf...)
text = _fold_brace_commands(text)
text = re.sub(r"\\(legend|caption|text|footnote)\{[^{}]*(?:\{[^{}]*\}[^{}]*)*\}", " ", text)
# ambientes: manter marcadores de item, eliminar o resto
text = re.sub(r"\\begin\{itemize\}", "\n- ", text)
text = re.sub(r"\\begin\{enumerate\}", "\n1. ", text)
text = re.sub(r"\\item\b", "\n- ", text)
text = re.sub(r"\\(end|begin)\{[^}]*\}", " ", text)
text = re.sub(r"\\toprule|\\midrule|\\bottomrule", " ", text)
text = text.replace("\\\\", "\n") # quebras de linha (\\) em tabular/autor
# comandos restantes sem argumento
text = re.sub(r"\\[a-zA-Z]+\*?", " ", text)
text = re.sub(r"\\[,;:\\!%$&]", " ", text)
for esc, ch in _ESCAPE_MAP.items():
text = text.replace(esc, ch)
# aspas e travessões
text = text.replace("``", '"').replace("''", '"')
text = text.replace("---", " — ").replace("--", " – ")
# tabular: & vira coluna
text = text.replace("&", " | ")
# limpar pontuação e espaços
text = re.sub(r"\[[a-z]*\]", " ", text) # posicionadores [htb]
text = re.sub(r"[{}]+", " ", text)
text = re.sub(r"[ \t]+", " ", text)
text = re.sub(r" +([,.;:!?])", r"\1", text)
text = re.sub(r"\n\s*", "\n", text)
text = re.sub(r"\n{3,}", "\n\n", text)
return text.strip()
# ---------------------------------------------------------------------------
# Parsing da estrutura
# ---------------------------------------------------------------------------
_SECTION_RE = re.compile(r"^\s*\\(section|subsection|subsubsection)\*?\s*\{([^}]+)\}")
def _capture_command(text: str, name: str) -> str:
"""Extrai o último conteúdo de \\name{...} ou \\<name> ... \\end{<name>}."""
m = re.search(r"\\" + name + r"\s*\{", text)
if not m:
return ""
start = m.end()
depth = 1
i = start
while i < len(text) and depth:
if text[i] == "{":
depth += 1
elif text[i] == "}":
depth -= 1
i += 1
return text[start : i - 1].strip()
def _extract_authors(text: str) -> list[str]:
raw = _capture_command(text, "autor")
if not raw:
raw = _capture_command(text, "author")
return [a.strip() for a in re.split(r"\\\\|\\[ ]", raw) if a.strip()]
def parse_article(tex_path: Path | str) -> Article:
tex_path = Path(tex_path)
text = tex_path.read_text(encoding="utf-8")
# remove comentários de linha (não escapados)
text = re.sub(r"(?<!\\)%[^\n]*", "", text)
citations: list[str] = []
raw_sections: list[tuple[int, str, list[str]]] = []
title = strip_latex(_capture_command(text, "titulo")
or _capture_command(text, "title"))
authors = _extract_authors(text)
local = strip_latex(_capture_command(text, "local"))
data = strip_latex(_capture_command(text, "data"))
instituicao = strip_latex(_capture_command(text, "instituicao")
or _capture_command(text, "institution"))
# resumo (ambiente resumoumacoluna do abntex2)
m = re.search(r"\\begin\{resumoumacoluna\}(.*?)\\end\{resumoumacoluna\}",
text, re.S)
abstract_raw = m.group(1) if m else ""
abstract = strip_latex(abstract_raw, citations)
keywords: list[str] = []
km = re.search(
r"Palavras-chave[: ]*\s*(.+)", abstract
)
if km:
kw_body = strip_latex(abstract)
kw_text = kw_body.split("Palavras-chave", 1)[-1]
kw_text = kw_text.split(":", 1)[-1].strip()
keywords = [k.strip() for k in re.split(r"[.;]", kw_text) if k.strip()]
# seções: percorre linha a linha acumulando o corpo de cada seção
current: tuple[int, str, list[str]] | None = None
preamble_text = ""
figures: list[str] = []
tables: list[str] = []
for line in text.splitlines():
sm = _SECTION_RE.match(line)
if sm:
level = {"section": 1, "subsection": 2, "subsubsection": 3}[sm.group(1)]
title_s = strip_latex(sm.group(2), citations)
if current:
raw_sections.append(current)
current = (level, title_s, [])
continue
if current is None:
preamble_text += line + "\n"
continue
current[2].append(line)
if current:
raw_sections.append(current)
article = Article(
title=title,
authors=authors,
local=local,
data=data,
instituicao=instituicao,
abstract=abstract,
keywords=keywords,
)
for level, title_s, lines in raw_sections:
sec = Section(title=title_s, level=level)
body = "\n".join(lines)
# captura captions de figuras/quadros antes de remover
for cap in re.findall(r"\\caption\{([^}]+)\}", body):
if "\\includegraphics" in body or "figura" in body.lower():
figures.append(strip_latex(cap))
else:
tables.append(strip_latex(cap))
sec.text = strip_latex(body, citations)
article.sections.append(sec)
article.citations = citations
article.figures = figures
article.tables = tables
article.text_words = (
len(count_words(title))
+ len(count_words(abstract))
+ sum(len(count_words(s.text)) for s in article.sections)
)
return article