Refactor GitHub authentication and improve code consistency in parse.py

This commit is contained in:
Wilson Freitas
2026-01-04 11:26:38 -03:00
parent 6c364f9ec3
commit 12748262e1
+17 -19
View File
@@ -1,23 +1,22 @@
import os import os
import re import re
import pandas as pd import pandas as pd
from threading import Thread from threading import Thread
from github import Github from github import Auth, Github
# using an access token # using an access token
g = Github(os.environ['GITHUB_ACCESS_TOKEN']) auth = Auth.Token(os.environ["GITHUB_ACCESS_TOKEN"])
g = Github(auth=auth)
def extract_repo(url): def extract_repo(url):
reu = re.compile(r'^https://github.com/([\w-]+/[-\w\.]+)$') reu = re.compile(r"^https://github.com/([\w-]+/[-\w\.]+)$")
m = reu.match(url) m = reu.match(url)
if m: if m:
return m.group(1) return m.group(1)
else: else:
return '' return ""
def get_last_commit(repo): def get_last_commit(repo):
@@ -25,16 +24,15 @@ def get_last_commit(repo):
if repo: if repo:
r = g.get_repo(repo) r = g.get_repo(repo)
cs = r.get_commits() cs = r.get_commits()
return cs[0].commit.author.date.strftime('%Y-%m-%d') return cs[0].commit.author.date.strftime("%Y-%m-%d")
else: else:
return '' return ""
except: except:
print('ERROR ' + repo) print("ERROR " + repo)
return 'error' return "error"
class Project(Thread): class Project(Thread):
def __init__(self, match, section): def __init__(self, match, section):
super().__init__() super().__init__()
self._match = match self._match = match
@@ -43,8 +41,8 @@ class Project(Thread):
def run(self): def run(self):
m = self._match m = self._match
is_github = 'github.com' in m.group(2) is_github = "github.com" in m.group(2)
is_cran = 'cran.r-project.org' in m.group(2) is_cran = "cran.r-project.org" in m.group(2)
repo = extract_repo(m.group(2)) repo = extract_repo(m.group(2))
print(repo) print(repo)
last_commit = get_last_commit(repo) last_commit = get_last_commit(repo)
@@ -56,21 +54,21 @@ class Project(Thread):
description=m.group(3), description=m.group(3),
github=is_github, github=is_github,
cran=is_cran, cran=is_cran,
repo=repo repo=repo,
) )
projects = [] projects = []
with open('README.md', 'r', encoding='utf8') as f: with open("README.md", "r", encoding="utf8") as f:
ret = re.compile(r'^(#+) (.*)$') ret = re.compile(r"^(#+) (.*)$")
rex = re.compile(r'^\s*- \[(.*)\]\((.*)\) - (.*)$') rex = re.compile(r"^\s*- \[(.*)\]\((.*)\) - (.*)$")
m_titles = [] m_titles = []
last_head_level = 0 last_head_level = 0
for line in f: for line in f:
m = rex.match(line) m = rex.match(line)
if m: if m:
p = Project(m, ' > '.join(m_titles[1:])) p = Project(m, " > ".join(m_titles[1:]))
p.start() p.start()
projects.append(p) projects.append(p)
else: else:
@@ -92,5 +90,5 @@ while True:
projects = [p.regs for p in projects] projects = [p.regs for p in projects]
df = pd.DataFrame(projects) df = pd.DataFrame(projects)
df.to_csv('site/projects.csv', index=False) df.to_csv("site/projects.csv", index=False)
# df.to_markdown('projects.md', index=False) # df.to_markdown('projects.md', index=False)