Projeto analise-artigo: análise ABNT de artigos LaTeX
- texparse.py: extração de estrutura abntex2 (título, autores, resumo, palavras-chave, seções, citações, figuras), sem dependências externas - bib.py: parser .bib (BibTeX) com normalização de autores e diacríticos - analyze.py: checklist NBR 10520/6022/14724, métricas textuais, Flesch adaptado pt-BR e heurísticas de prosa - report.py: relatório Markdown + JSON - artigos/rumo-a-eficiencia: artigo analisado (main.tex + referencias.bib)
This commit is contained in:
264
analise_artigo/texparse.py
Normal file
264
analise_artigo/texparse.py
Normal file
@@ -0,0 +1,264 @@
|
||||
"""Extrai a estrutura textual de um documento LaTeX (abntex2/ABNT).
|
||||
|
||||
Converte o .tex em texto limpo preservando:
|
||||
- título, autores, local/data (elementos pré-textuais);
|
||||
- resumo e palavras-chave;
|
||||
- seções/subseções e seus parágrafos;
|
||||
- chaves de citação (\\cite, \\citeonline, \\citeauthor ...).
|
||||
|
||||
Não depende de pandoc: usa apenas a biblioteca padrão.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
@dataclass
|
||||
class Section:
|
||||
"""Seção do artigo com o texto limpo já convertido para prosa."""
|
||||
|
||||
title: str
|
||||
level: int = 1
|
||||
|
||||
@property
|
||||
def words(self) -> int:
|
||||
return len(count_words(self.text))
|
||||
|
||||
|
||||
@dataclass
|
||||
class Article:
|
||||
title: str = ""
|
||||
authors: list[str] = field(default_factory=list)
|
||||
local: str = ""
|
||||
data: str = ""
|
||||
instituicao: str = ""
|
||||
abstract: str = ""
|
||||
keywords: list[str] = field(default_factory=list)
|
||||
sections: list[Section] = field(default_factory=list)
|
||||
citations: list[str] = field(default_factory=list) # ordem de aparição
|
||||
figures: list[str] = field(default_factory=list) # captions das figuras
|
||||
tables: list[str] = field(default_factory=list) # captions de quadros
|
||||
text_words: int = 0
|
||||
|
||||
def section(self, title: str) -> Section | None:
|
||||
for s in self.sections:
|
||||
if s.title.strip().lower() == title.strip().lower():
|
||||
return s
|
||||
return None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Conversão LaTeX -> texto
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
_BRACE_CMDS = ["textbf", "emph", "textit", "texttt", "underline",
|
||||
"bold", "itshape", "scshape", "MakeUppercase", "MakeLowercase"]
|
||||
|
||||
_CITE_RE = re.compile(r"\\cite\w*\{([^}]*)\}")
|
||||
_BRACE_CMD_RE = {
|
||||
cmd: re.compile(r"\\" + cmd + r"\{([^{}]*)\}") for cmd in _BRACE_CMDS
|
||||
}
|
||||
# Chaves com barra escapada no LaTeX (ex.: "\_", "\#").
|
||||
# "\," "\;" "\%" "\$" "\&" já são tratados pela regex de duas linhas acima.
|
||||
_ESCAPE_MAP = {
|
||||
"\\_": "_",
|
||||
"\\#": "#",
|
||||
"\\\\": "\n",
|
||||
}
|
||||
|
||||
|
||||
def count_words(text: str) -> list[str]:
|
||||
return re.findall(
|
||||
r"[A-Za-z0-9À-ÖØ-öø-ÿ]+(?:['’-][A-Za-z0-9À-ÖØ-öø-ÿ]+)*", text
|
||||
)
|
||||
|
||||
|
||||
def _fold_brace_commands(s: str) -> str:
|
||||
changed = True
|
||||
while changed:
|
||||
changed = False
|
||||
for cmd, pat in _BRACE_CMD_RE.items():
|
||||
s2, n = pat.subn(r"\1", s)
|
||||
if n:
|
||||
s, changed = s2, True
|
||||
return s
|
||||
|
||||
|
||||
def strip_latex(text: str, citations: list[str] | None = None) -> str:
|
||||
"""Remove comandos e ambientes, preservando o conteúdo legível."""
|
||||
|
||||
# citações: captura as chaves e remove do texto
|
||||
def _cite(m: re.Match) -> str:
|
||||
if citations is not None:
|
||||
for key in m.group(1).split(","):
|
||||
key = key.strip()
|
||||
if key:
|
||||
citations.append(key)
|
||||
return " "
|
||||
|
||||
text = _CITE_RE.sub(_cite, text)
|
||||
|
||||
# rótulos/referências internas e figuras
|
||||
text = re.sub(r"\\label\{[^}]*\}", " ", text)
|
||||
text = re.sub(r"\\(ref|pageref|eqref)\{[^}]*\}", " (ref.) ", text)
|
||||
text = re.sub(r"\\includegraphics.*", " ", text)
|
||||
text = re.sub(r"\\noindent", " ", text)
|
||||
|
||||
# comandos com argumento em chaves genéricos (legend, caption, textbf...)
|
||||
text = _fold_brace_commands(text)
|
||||
text = re.sub(r"\\(legend|caption|text|footnote)\{[^{}]*(?:\{[^{}]*\}[^{}]*)*\}", " ", text)
|
||||
|
||||
# ambientes: manter marcadores de item, eliminar o resto
|
||||
text = re.sub(r"\\begin\{itemize\}", "\n- ", text)
|
||||
text = re.sub(r"\\begin\{enumerate\}", "\n1. ", text)
|
||||
text = re.sub(r"\\item\b", "\n- ", text)
|
||||
text = re.sub(r"\\(end|begin)\{[^}]*\}", " ", text)
|
||||
text = re.sub(r"\\toprule|\\midrule|\\bottomrule", " ", text)
|
||||
text = text.replace("\\\\", "\n") # quebras de linha (\\) em tabular/autor
|
||||
|
||||
# comandos restantes sem argumento
|
||||
text = re.sub(r"\\[a-zA-Z]+\*?", " ", text)
|
||||
text = re.sub(r"\\[,;:\\!%$&]", " ", text)
|
||||
for esc, ch in _ESCAPE_MAP.items():
|
||||
text = text.replace(esc, ch)
|
||||
|
||||
# aspas e travessões
|
||||
text = text.replace("``", '"').replace("''", '"')
|
||||
text = text.replace("---", " — ").replace("--", " – ")
|
||||
|
||||
# tabular: & vira coluna
|
||||
text = text.replace("&", " | ")
|
||||
|
||||
# limpar pontuação e espaços
|
||||
text = re.sub(r"\[[a-z]*\]", " ", text) # posicionadores [htb]
|
||||
text = re.sub(r"[{}]+", " ", text)
|
||||
text = re.sub(r"[ \t]+", " ", text)
|
||||
text = re.sub(r" +([,.;:!?])", r"\1", text)
|
||||
text = re.sub(r"\n\s*", "\n", text)
|
||||
text = re.sub(r"\n{3,}", "\n\n", text)
|
||||
return text.strip()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Parsing da estrutura
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
_SECTION_RE = re.compile(r"^\s*\\(section|subsection|subsubsection)\*?\s*\{([^}]+)\}")
|
||||
|
||||
|
||||
def _capture_command(text: str, name: str) -> str:
|
||||
"""Extrai o último conteúdo de \\name{...} ou \\<name> ... \\end{<name>}."""
|
||||
m = re.search(r"\\" + name + r"\s*\{", text)
|
||||
if not m:
|
||||
return ""
|
||||
start = m.end()
|
||||
depth = 1
|
||||
i = start
|
||||
while i < len(text) and depth:
|
||||
if text[i] == "{":
|
||||
depth += 1
|
||||
elif text[i] == "}":
|
||||
depth -= 1
|
||||
i += 1
|
||||
return text[start : i - 1].strip()
|
||||
|
||||
|
||||
def _extract_authors(text: str) -> list[str]:
|
||||
raw = _capture_command(text, "autor")
|
||||
if not raw:
|
||||
raw = _capture_command(text, "author")
|
||||
return [a.strip() for a in re.split(r"\\\\|\\[ ]", raw) if a.strip()]
|
||||
|
||||
|
||||
def parse_article(tex_path: Path | str) -> Article:
|
||||
tex_path = Path(tex_path)
|
||||
text = tex_path.read_text(encoding="utf-8")
|
||||
|
||||
# remove comentários de linha (não escapados)
|
||||
text = re.sub(r"(?<!\\)%[^\n]*", "", text)
|
||||
|
||||
citations: list[str] = []
|
||||
raw_sections: list[tuple[int, str, list[str]]] = []
|
||||
|
||||
title = strip_latex(_capture_command(text, "titulo")
|
||||
or _capture_command(text, "title"))
|
||||
authors = _extract_authors(text)
|
||||
local = strip_latex(_capture_command(text, "local"))
|
||||
data = strip_latex(_capture_command(text, "data"))
|
||||
instituicao = strip_latex(_capture_command(text, "instituicao")
|
||||
or _capture_command(text, "institution"))
|
||||
|
||||
# resumo (ambiente resumoumacoluna do abntex2)
|
||||
m = re.search(r"\\begin\{resumoumacoluna\}(.*?)\\end\{resumoumacoluna\}",
|
||||
text, re.S)
|
||||
abstract_raw = m.group(1) if m else ""
|
||||
abstract = strip_latex(abstract_raw, citations)
|
||||
|
||||
keywords: list[str] = []
|
||||
km = re.search(
|
||||
r"Palavras-chave[: ]*\s*(.+)", abstract
|
||||
)
|
||||
if km:
|
||||
kw_body = strip_latex(abstract)
|
||||
kw_text = kw_body.split("Palavras-chave", 1)[-1]
|
||||
kw_text = kw_text.split(":", 1)[-1].strip()
|
||||
keywords = [k.strip() for k in re.split(r"[.;]", kw_text) if k.strip()]
|
||||
|
||||
# seções: percorre linha a linha acumulando o corpo de cada seção
|
||||
current: tuple[int, str, list[str]] | None = None
|
||||
preamble_text = ""
|
||||
figures: list[str] = []
|
||||
tables: list[str] = []
|
||||
|
||||
for line in text.splitlines():
|
||||
sm = _SECTION_RE.match(line)
|
||||
if sm:
|
||||
level = {"section": 1, "subsection": 2, "subsubsection": 3}[sm.group(1)]
|
||||
title_s = strip_latex(sm.group(2), citations)
|
||||
if current:
|
||||
raw_sections.append(current)
|
||||
current = (level, title_s, [])
|
||||
continue
|
||||
|
||||
if current is None:
|
||||
preamble_text += line + "\n"
|
||||
continue
|
||||
current[2].append(line)
|
||||
|
||||
if current:
|
||||
raw_sections.append(current)
|
||||
|
||||
article = Article(
|
||||
title=title,
|
||||
authors=authors,
|
||||
local=local,
|
||||
data=data,
|
||||
instituicao=instituicao,
|
||||
abstract=abstract,
|
||||
keywords=keywords,
|
||||
)
|
||||
|
||||
for level, title_s, lines in raw_sections:
|
||||
sec = Section(title=title_s, level=level)
|
||||
body = "\n".join(lines)
|
||||
# captura captions de figuras/quadros antes de remover
|
||||
for cap in re.findall(r"\\caption\{([^}]+)\}", body):
|
||||
if "\\includegraphics" in body or "figura" in body.lower():
|
||||
figures.append(strip_latex(cap))
|
||||
else:
|
||||
tables.append(strip_latex(cap))
|
||||
sec.text = strip_latex(body, citations)
|
||||
article.sections.append(sec)
|
||||
|
||||
article.citations = citations
|
||||
article.figures = figures
|
||||
article.tables = tables
|
||||
article.text_words = (
|
||||
len(count_words(title))
|
||||
+ len(count_words(abstract))
|
||||
+ sum(len(count_words(s.text)) for s in article.sections)
|
||||
)
|
||||
return article
|
||||
Reference in New Issue
Block a user