Coverage for website.py: 100%
303 statements
« prev ^ index » next coverage.py v7.15.4, created at 2026-10-01 05:56 +0000
« prev ^ index » next coverage.py v7.15.4, created at 2026-10-01 05:56 +0000
1# This file is part of Vallenato.fr.
2#
3# Vallenato.fr is free software: you can redistribute it and/or modify
4# it under the terms of the GNU Affero General Public License as published by
5# the Free Software Foundation, either version 3 of the License, or
6# (at your option) any later version.
7#
8# Vallenato.fr is distributed in the hope that it will be useful,
9# but WITHOUT ANY WARRANTY; without even the implied warranty of
10# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
11# GNU Affero General Public License for more details.
12#
13# You should have received a copy of the GNU Affero General Public License
14# along with Vallenato.fr. If not, see <http://www.gnu.org/licenses/>.
16import datetime
17import json
18import logging
19import os
20import re
21import shutil
22import sys
24from sitemap import generator
25from slugify import slugify
27from youtube import (
28 HttpError,
29 yt_get_authenticated_service,
30 yt_get_my_uploads_list,
31 yt_list_my_uploaded_videos,
32)
34logger = logging.getLogger(__name__)
36# File that can contain the data downloaded from YouTube
37UPLOADED_VIDEOS_DUMP_FILE = "data/uploaded_videos_dump.json"
38# File containing the list of videos that have hardcoded locations
39LOCATION_SPECIAL_CASES_FILE = "data/location_special_cases.json"
40# File containing the already-identified latitude/longitude
41GEOLOCATIONS_FILE = "data/geolocations.json"
42# Output file used for the website
43WEBSITE_DATA_FILE = "../website/src/data.js"
44# Sitemap file
45SITEMAP_FILE = "../website/prod/sitemap.xml"
46# Version of the external libraries
47LEAFLET_VERSION = "1.9.4"
48BOOTSTRAP_VERSION = "4.6.2"
49JQUERY_VERSION = "3.7.1"
50BOOTSTRAP_TOGGLE_VERSION = "3.6.1"
53def get_dumped_uploaded_videos(dump_file):
54 uploaded_videos = []
55 # Used a previously dumped file if it exists, to bypass the network transactions
56 if os.path.exists(dump_file):
57 with open(dump_file) as in_file:
58 uploaded_videos = json.load(in_file)
59 return uploaded_videos
62def save_uploaded_videos(uploaded_videos, dump_file):
63 with open(dump_file, "w") as out_file:
64 json.dump(uploaded_videos, out_file, sort_keys=True, indent=2)
67def determine_videos_slug(uploaded_videos):
68 logger.debug("Determining each video's slug...")
69 for vid in uploaded_videos:
70 vid["slug"] = slugify(vid["title"]).replace("-desde-", "-")
71 return uploaded_videos
74def get_uploaded_videos(args, dump_file):
75 uploaded_videos = get_dumped_uploaded_videos(dump_file)
76 if not uploaded_videos:
77 youtube = yt_get_authenticated_service(args)
78 # Get the list of videos uploaded to YouTube
79 try:
80 uploads_playlist_id = yt_get_my_uploads_list(youtube)
81 if uploads_playlist_id:
82 uploaded_videos = yt_list_my_uploaded_videos(
83 uploads_playlist_id, youtube
84 )
85 logger.debug(f"Uploaded videos: {uploaded_videos}")
86 else:
87 logger.info("There is no uploaded videos playlist for this user.")
88 except HttpError as e:
89 logger.debug(f"An HTTP error {e.resp.status} occurred:\n{e.content}")
90 logger.critical("Exiting...")
91 sys.exit(19)
92 # Create a slug for each video (to be used for the website URLs)
93 uploaded_videos = determine_videos_slug(uploaded_videos)
94 if args.dump_uploaded_videos:
95 save_uploaded_videos(uploaded_videos, dump_file)
96 return uploaded_videos
99def identify_locations_names(uploaded_videos, location_special_cases_file, dump_file):
100 logger.debug("Identify each video's location name")
101 with open(location_special_cases_file) as in_file:
102 special_cases = json.load(in_file)
103 locations = {}
104 incomplete_locations = False
105 for vid in uploaded_videos:
106 vid["location"] = identify_single_location_name(vid, special_cases)
107 if not vid["location"]:
108 incomplete_locations = True
109 elif vid["location"] not in locations:
110 locations[vid["location"]] = {"latitude": None, "longitude": None}
111 if incomplete_locations:
112 # The script is going to exit, to prevent unnecessary downloading from
113 # YouTube again, save the downloaded information regardless of the
114 # --dump_uploaded_videos parameter
115 logger.warning(
116 f"Dumping the list of uploaded videos from YouTube to the '{dump_file}' file, so as not to have to download it again after you have edited the '{location_special_cases_file}' file."
117 )
118 save_uploaded_videos(uploaded_videos, dump_file)
119 logger.critical(
120 f"Please add the new/missing location to the file '{location_special_cases_file}'. Exiting..."
121 )
122 sys.exit(20)
123 logger.info(f"Found {len(locations)} different location name.")
124 return (uploaded_videos, locations)
127def identify_single_location_name(vid, special_cases):
128 location = None
129 if vid["id"] in special_cases:
130 location = special_cases[vid["id"]]
131 logger.debug("Video {}, location '{}'".format(vid["id"], location))
132 else:
133 for search_string in (", desde ", ", cerca de "):
134 loc_index = vid["title"].find(search_string)
135 if loc_index > 0:
136 location = vid["title"][loc_index + len(search_string) :]
137 logger.debug("Video {}, location '{}'".format(vid["id"], location))
138 break
140 # Each video should now have a location identified. If not, this will end the script.
141 if not location:
142 logger.critical(
143 "No Location found for {}, '{}'".format(vid["id"], vid["title"])
144 )
145 return location
148def determine_geolocation(locations, geolocations_file):
149 logger.debug(f"Searching geolocation for {len(locations)} locations...")
150 # Load the list of saved geolocations
151 with open(geolocations_file) as in_file:
152 geolocations = json.load(in_file)
153 incomplete_geolocations = 0
154 for l in locations:
155 if (
156 l in geolocations
157 and geolocations[l]["latitude"]
158 and geolocations[l]["longitude"]
159 ):
160 logger.debug(f"Geolocation found for {l}")
161 locations[l]["latitude"] = geolocations[l]["latitude"]
162 locations[l]["longitude"] = geolocations[l]["longitude"]
163 else:
164 logger.critical(f"No geolocation found for {l}.")
165 # TODO: Search and suggest a geolocation
166 geolocations[l] = {"latitude": None, "longitude": None}
167 incomplete_geolocations += 1
169 if incomplete_geolocations > 0:
170 # Save the geolocations_file file with the placeholders for the unknown latitude and longitude
171 with open(geolocations_file, "w") as out_file:
172 json.dump(geolocations, out_file, sort_keys=True, indent=2)
173 logger.critical(
174 f"Please add the {incomplete_geolocations} new/missing unknown latitude and longitude to the file '{geolocations_file}'. Exiting..."
175 )
176 sys.exit(21)
178 logger.info(f"Found geolocation information for the {len(locations)} locations.")
179 return locations
182def add_videos_to_locations_array(uploaded_videos, locations):
183 logger.debug("Adding videos in each location array...")
184 for vid in uploaded_videos:
185 if not "videos" in locations[vid["location"]]:
186 locations[vid["location"]]["videos"] = []
187 locations[vid["location"]]["videos"].append(vid)
188 return locations
191def determine_locations_slug(locations):
192 logger.debug("Determining each location's slug...")
193 for loc in locations:
194 locations[loc]["slug"] = slugify(loc)
195 return locations
198def save_website_data(locations, website_data_file):
199 logger.debug("Save the updated dynamic data")
200 json_content = json.dumps(locations, sort_keys=True, indent=2)
201 # Make it JS (and not just JSON) for direct use in the HTML document
202 js_content = f"var locations = {json_content};"
203 with open(website_data_file, "w") as out_file:
204 out_file.write(js_content)
207def load_locations_from_website_data_file(website_data_file):
208 # Used by --no-fetch: rebuild the website from what's already committed
209 # to website/src/data.js instead of hitting the YouTube API, e.g. to
210 # build the production Docker image without live OAuth credentials.
211 logger.debug(f"Loading existing locations from {website_data_file} (--no-fetch)")
212 with open(website_data_file) as in_file:
213 # Remove the JS bits to keep only the JSON content
214 return json.loads(in_file.read()[16:-1])
217def flatten_videos_from_locations(locations):
218 uploaded_videos = []
219 for location in locations.values():
220 uploaded_videos.extend(location["videos"])
221 uploaded_videos.sort(key=lambda v: v["publishedAt"], reverse=True)
222 return uploaded_videos
225def ignored_files_in_prod(adir, filenames):
226 ignored_files = []
227 if "../website/src" == adir:
228 ignored_files = [
229 f"bootstrap-{BOOTSTRAP_VERSION}-dist",
230 f"bootstrap4-toggle-{BOOTSTRAP_TOGGLE_VERSION}",
231 f"jquery-{JQUERY_VERSION}.slim.min.js",
232 "leaflet",
233 ]
234 if "../website/src/aprender" == adir:
235 ignored_files = ["temp", "videos"]
236 return [filename for filename in filenames if filename in ignored_files]
239def get_stats(locations, uploaded_videos):
240 num_videos = len(uploaded_videos)
242 songs = []
243 skipped_titles = ["Vallenato at Epic", "La Guaneña navideña"]
244 for v in uploaded_videos:
245 song = v["title"].split(",")[0]
246 if song not in songs and song not in skipped_titles:
247 songs.append(song)
248 num_songs = len(songs)
250 num_places = len(locations)
252 countries = []
253 for l in locations:
254 country = l.split(",")[-1]
255 if country not in countries:
256 countries.append(country)
257 num_countries = len(countries)
259 navidad_2017 = datetime.date(2017, 12, 25)
260 today = datetime.date.today()
261 years = today.year - navidad_2017.year
262 if today.month == 12: # December
263 duration_since_navidad_2017 = f"{years} años"
264 elif today.month == 1: # January
265 duration_since_navidad_2017 = f"{years - 1} años"
266 else:
267 duration_since_navidad_2017 = f"{years - 1} años y {today.month} meses"
269 stats = f"El Vallenatero Francés les presenta {num_videos} videos de {num_songs} canciones tocadas en {num_places} lugares de {num_countries} paises. El empezo a aprender el Acordeón Vallenato en la Navidad 2017 (hace mas o menos {duration_since_navidad_2017})."
271 return stats
274def generate_website(locations, uploaded_videos):
275 logger.debug("Generate the production website files")
276 input_src_folder = "../website/src"
277 output_prod_folder = "../website/prod"
278 # The 2 index files in / and /aprender, 404
279 num_html_pages_created = 3
281 # Delete the previous production output folder (if existing)
282 if os.path.exists(output_prod_folder):
283 shutil.rmtree(output_prod_folder)
285 # Update statistics
286 stats = get_stats(locations, uploaded_videos)
287 index_src_file = f"{input_src_folder}/index.html"
288 with open(index_src_file, "r") as file:
289 index_data = file.read()
290 index_data = re.sub(
291 '<div id="stats">.*</div>', f'<div id="stats">{stats}</div>', index_data
292 )
293 with open(index_src_file, "w") as file:
294 file.write(index_data)
296 # Update the values accordingly for prod
297 # Main difference between development (src) and production websites:
298 # - src contains a full copy of the leaflet, Bootstrap and jQuery libraries
299 # - prod uses CDNs
301 # Copy src to prod folder, ignoring the files and folder replaced by CDNs in prod
302 # The videos are also not copied, as we're going to hard-link them
303 shutil.copytree(input_src_folder, output_prod_folder, ignore=ignored_files_in_prod)
305 # Create hard links for the videos in the prod folder
306 # (hard links can only be created for files, need to recreate the folder structure)
307 os.mkdir(f"{output_prod_folder}/aprender/videos")
308 # website/src/aprender/videos/ is gitignored (and .dockerignore'd) - it
309 # only exists locally once a tutorial's video files have been downloaded
310 # via --aprender. A fresh checkout (or a Docker build context, which
311 # never sends it in) has no such directory yet - nothing to hard-link.
312 videos_src_dir = f"{input_src_folder}/aprender/videos"
313 for d in os.listdir(videos_src_dir) if os.path.isdir(videos_src_dir) else []:
314 if d not in ["TODO", "blabla-bla"]:
315 # Create a folder for that tutorial's video files
316 # TODO: copy folder without content in order to keep the original folder's
317 # creation date, in order to not confuse the rsync upload process
318 os.mkdir(f"{output_prod_folder}/aprender/videos/{d}")
319 for f in os.listdir(f"{input_src_folder}/aprender/videos/{d}"):
320 # Create a hard link to the video file
321 os.link(
322 f"{input_src_folder}/aprender/videos/{d}/{f}",
323 f"{output_prod_folder}/aprender/videos/{d}/{f}",
324 )
326 # Update links to leaflet (CDN)
327 # Read the prod files
328 with open(f"{output_prod_folder}/index.html", "r") as file:
329 index_data = file.read()
330 with open(f"{output_prod_folder}/404.html", "r") as file:
331 page404_data = file.read()
332 with open(f"{output_prod_folder}/aprender/index.html", "r") as file:
333 index_aprender_data = file.read()
334 # Replace the target strings
335 # Leaflet
336 index_data = index_data.replace(
337 f'<link rel="stylesheet" href="leaflet/{LEAFLET_VERSION}/leaflet.css">',
338 f'<link rel="stylesheet" href="https://unpkg.com/leaflet@{LEAFLET_VERSION}/dist/leaflet.css"\n integrity="sha512-Zcn6bjR/8RZbLEpLIeOwNtzREBAJnUKESxces60Mpoj+2okopSAcSUIUOseddDm0cxnGQzxIR7vJgsLZbdLE3w=="\n crossorigin=""/>',
339 )
340 index_data = index_data.replace(
341 f'<script type = "text/javascript" src="leaflet/{LEAFLET_VERSION}/leaflet.js"></script>',
342 f'<script src="https://unpkg.com/leaflet@{LEAFLET_VERSION}/dist/leaflet.js"\n integrity="sha512-BwHfrr4c9kmRkLw6iXFdzcdWV/PGkVgiIyIWLLlTSXzWQzxuSg4DiQUCpauz/EWjgk5TYQqX/kvn9pG1NpYfqg=="\n crossorigin="">\n </script>',
343 )
344 # Bootstrap
345 index_data = index_data.replace(
346 f'<link rel="stylesheet" href="bootstrap-{BOOTSTRAP_VERSION}-dist/css/bootstrap.min.css">',
347 f'<link rel="stylesheet" href="https://cdn.jsdelivr.net/npm/bootstrap@{BOOTSTRAP_VERSION}/dist/css/bootstrap.min.css"\n integrity="sha384-xOolHFLEh07PJGoPkLv1IbcEPTNtaed2xpHsD9ESMhqIYd0nLMwNLD69Npy4HI+N"\n crossorigin="anonymous">',
348 )
349 index_data = index_data.replace(
350 f'<script src="bootstrap-{BOOTSTRAP_VERSION}-dist/js/bootstrap.min.js"></script>',
351 f'<script src="https://cdn.jsdelivr.net/npm/bootstrap@{BOOTSTRAP_VERSION}/dist/js/bootstrap.min.js"\n integrity="sha384-+sLIOodYLS7CIrQpBjl+C7nPvqq+FbNUBDunl/OZv93DB7Ln/533i8e/mZXLi/P+"\n crossorigin="anonymous"></script>',
352 )
353 page404_data = page404_data.replace(
354 f'<link rel="stylesheet" href="bootstrap-{BOOTSTRAP_VERSION}-dist/css/bootstrap.min.css">',
355 f'<link rel="stylesheet" href="https://cdn.jsdelivr.net/npm/bootstrap@{BOOTSTRAP_VERSION}/dist/css/bootstrap.min.css"\n integrity="sha384-xOolHFLEh07PJGoPkLv1IbcEPTNtaed2xpHsD9ESMhqIYd0nLMwNLD69Npy4HI+N"\n crossorigin="anonymous">',
356 )
357 page404_data = page404_data.replace(
358 f'<script src="bootstrap-{BOOTSTRAP_VERSION}-dist/js/bootstrap.min.js"></script>',
359 f'<script src="https://cdn.jsdelivr.net/npm/bootstrap@{BOOTSTRAP_VERSION}/dist/js/bootstrap.min.js"\n integrity="sha384-+sLIOodYLS7CIrQpBjl+C7nPvqq+FbNUBDunl/OZv93DB7Ln/533i8e/mZXLi/P+"\n crossorigin="anonymous"></script>',
360 )
361 index_aprender_data = index_aprender_data.replace(
362 f'<link rel="stylesheet" href="../bootstrap-{BOOTSTRAP_VERSION}-dist/css/bootstrap.min.css">',
363 f'<link rel="stylesheet" href="https://cdn.jsdelivr.net/npm/bootstrap@{BOOTSTRAP_VERSION}/dist/css/bootstrap.min.css"\n integrity="sha384-xOolHFLEh07PJGoPkLv1IbcEPTNtaed2xpHsD9ESMhqIYd0nLMwNLD69Npy4HI+N"\n crossorigin="anonymous">',
364 )
365 index_aprender_data = index_aprender_data.replace(
366 f'<script src="../bootstrap-{BOOTSTRAP_VERSION}-dist/js/bootstrap.min.js"></script>',
367 f'<script src="https://cdn.jsdelivr.net/npm/bootstrap@{BOOTSTRAP_VERSION}/dist/js/bootstrap.min.js"\n integrity="sha384-+sLIOodYLS7CIrQpBjl+C7nPvqq+FbNUBDunl/OZv93DB7Ln/533i8e/mZXLi/P+"\n crossorigin="anonymous"></script>',
368 )
369 # jQuery (for Bootstrap)
370 index_data = index_data.replace(
371 f'<script src="jquery-{JQUERY_VERSION}.slim.min.js"></script>',
372 f'<script src="https://code.jquery.com/jquery-{JQUERY_VERSION}.slim.min.js"\n integrity="sha384-5AkRS45j4ukf+JbWAfHL8P4onPA9p0KwwP7pUdjSQA3ss9edbJUJc/XcYAiheSSz"\n crossorigin="anonymous"></script>',
373 )
374 page404_data = page404_data.replace(
375 f'<script src="jquery-{JQUERY_VERSION}.slim.min.js"></script>',
376 f'<script src="https://code.jquery.com/jquery-{JQUERY_VERSION}.slim.min.js"\n integrity="sha384-5AkRS45j4ukf+JbWAfHL8P4onPA9p0KwwP7pUdjSQA3ss9edbJUJc/XcYAiheSSz"\n crossorigin="anonymous"></script>',
377 )
378 index_aprender_data = index_aprender_data.replace(
379 f'<script src="../jquery-{JQUERY_VERSION}.slim.min.js"></script>',
380 f'<script src="https://code.jquery.com/jquery-{JQUERY_VERSION}.slim.min.js"\n integrity="sha384-5AkRS45j4ukf+JbWAfHL8P4onPA9p0KwwP7pUdjSQA3ss9edbJUJc/XcYAiheSSz"\n crossorigin="anonymous"></script>',
381 )
382 # Bootstrap-toggle
383 index_aprender_data = index_aprender_data.replace(
384 f'<link rel="stylesheet" href="../bootstrap4-toggle-{BOOTSTRAP_TOGGLE_VERSION}/css/bootstrap4-toggle.min.css">',
385 f'<link rel="stylesheet" href="https://cdn.jsdelivr.net/gh/gitbrent/bootstrap4-toggle@{BOOTSTRAP_TOGGLE_VERSION}/css/bootstrap4-toggle.min.css"\n integrity="sha384-yakM86Cz9KJ6CeFVbopALOEQGGvyBFdmA4oHMiYuHcd9L59pLkCEFSlr6M9m434E"\n crossorigin="anonymous">',
386 )
387 index_aprender_data = index_aprender_data.replace(
388 f'<script src="../bootstrap4-toggle-{BOOTSTRAP_TOGGLE_VERSION}/js/bootstrap4-toggle.min.js"></script>',
389 f'<script src="https://cdn.jsdelivr.net/gh/gitbrent/bootstrap4-toggle@{BOOTSTRAP_TOGGLE_VERSION}/js/bootstrap4-toggle.min.js"\n integrity="sha384-Q9RsZ4GMzjlu4FFkJw4No9Hvvm958HqHmXI9nqo5Np2dA/uOVBvKVxAvlBQrDhk4"\n crossorigin="anonymous"></script>',
390 )
391 # Copyright year in the pages' footer
392 a = '<span class="text-muted">© YEAR El Vallenatero Francés</span>'
393 b = f'<span class="text-muted">© {datetime.date.today().year} El Vallenatero Francés</span>'
394 index_data = index_data.replace(a, b)
395 page404_data = page404_data.replace(a, b)
396 index_aprender_data = index_aprender_data.replace(a, b)
398 # Save edited prod files
399 with open(f"{output_prod_folder}/index.html", "w") as file:
400 file.write(index_data)
401 with open(f"{output_prod_folder}/404.html", "w") as file:
402 file.write(page404_data)
403 with open(f"{output_prod_folder}/aprender/index.html", "w") as file:
404 file.write(index_aprender_data)
406 # Create full HTML pages for Prod /aprender tutorials
407 with open("../website/src/aprender/tutoriales.js") as in_file:
408 # Remove the JS bits to keep only the JSON content
409 tutoriales_json_content = in_file.read()[17:-2]
410 tutoriales = json.loads(tutoriales_json_content)
411 num_html_pages_created += len(tutoriales)
412 for t in tutoriales:
413 output_prod_tutorial_file = "{}/aprender/{}.html".format(
414 output_prod_folder, t["slug"]
415 )
416 shutil.copy(
417 f"{output_prod_folder}/aprender/index.html", output_prod_tutorial_file
418 )
419 with open(output_prod_tutorial_file, "r") as file:
420 prod_tutorial_file_data = file.read()
421 if t["author"]:
422 tuto_title = "{} - {}".format(t["title"], t["author"])
423 else:
424 tuto_title = t["title"]
425 prod_tutorial_file_data = prod_tutorial_file_data.replace(
426 "<title>Aprender a tocar el Acordeón Vallenato - El Vallenatero Francés</title>",
427 f"<title>{tuto_title} - Aprender a tocar el Acordeón Vallenato</title>",
428 )
429 prod_tutorial_file_data = prod_tutorial_file_data.replace(
430 '<h1 id="tutorialFullTitle">TITLE</h1>',
431 f'<h1 id="tutorialFullTitle">{tuto_title}</h1>',
432 )
433 with open(output_prod_tutorial_file, "w") as file:
434 file.write(prod_tutorial_file_data)
436 # Create full HTML pages for Prod / videos
437 with open("../website/src/data.js") as in_file:
438 # Remove the JS bits to keep only the JSON content
439 videos_json_content = in_file.read()[16:-1]
440 locations = json.loads(videos_json_content)
441 num_html_pages_created += len(locations)
442 for l in locations:
443 # One page for each location
444 output_prod_video_file = "{}/{}.html".format(
445 output_prod_folder, locations[l]["slug"]
446 )
447 shutil.copy(f"{output_prod_folder}/index.html", output_prod_video_file)
448 with open(output_prod_video_file, "r") as file:
449 prod_video_file_data = file.read()
450 tuto_title = l
451 prod_video_file_data = prod_video_file_data.replace(
452 "<title>El Vallenatero Francés</title>",
453 f"<title>{tuto_title} - El Vallenatero Francés</title>",
454 )
455 prod_video_file_data = prod_video_file_data.replace(
456 '<h2 id="list_location"></h2>', f'<h2 id="list_location">{tuto_title}</h2>'
457 )
458 with open(output_prod_video_file, "w") as file:
459 file.write(prod_video_file_data)
461 num_html_pages_created += len(locations[l]["videos"])
462 for v in locations[l]["videos"]:
463 # One page for each video at that location
464 # Create folder
465 output_folder = "{}/{}".format(output_prod_folder, v["slug"])
466 if not os.path.isdir(output_folder):
467 os.mkdir(output_folder)
468 output_prod_video_file = "{}/{}.html".format(output_folder, v["id"])
469 shutil.copy(f"{output_prod_folder}/index.html", output_prod_video_file)
470 with open(output_prod_video_file, "r") as file:
471 prod_video_file_data = file.read()
472 tuto_title = v["title"]
473 prod_video_file_data = prod_video_file_data.replace(
474 "<title>El Vallenatero Francés</title>",
475 f"<title>{tuto_title} - El Vallenatero Francés</title>",
476 )
477 prod_video_file_data = prod_video_file_data.replace(
478 '<h2 id="list_location"></h2>',
479 f'<h2 id="list_location">{tuto_title}</h2>',
480 )
481 with open(output_prod_video_file, "w") as file:
482 file.write(prod_video_file_data)
484 logger.debug(f"Number of production HTML files created: {num_html_pages_created}")
487def generate_sitemap(sitemap_file, locations, uploaded_videos):
488 logger.debug("Generate the Sitemap")
489 base_url = "https://vallenato.fr"
490 sitemap = generator.Sitemap()
492 # vallenato.fr index
493 sitemap.add(
494 base_url,
495 # Timestamp of the most recently uploaded video
496 lastmod=uploaded_videos[0]["publishedAt"][:10],
497 changefreq="monthly",
498 priority="1.0",
499 )
501 # Locations and individual videos
502 sitemap.add(
503 f"{base_url}/mundo-entero",
504 # Timestamp of the most recently uploaded video
505 lastmod=uploaded_videos[0]["publishedAt"][:10],
506 changefreq="monthly",
507 priority="0.8",
508 )
509 for l in locations:
510 # Locations
511 sitemap.add(
512 "{}/{}".format(base_url, locations[l]["slug"]),
513 # Timestamp of the most recently uploaded video at that location
514 lastmod=locations[l]["videos"][0]["publishedAt"][:10],
515 changefreq="yearly",
516 priority="0.6",
517 )
518 for v in locations[l]["videos"]:
519 # Individual videos
520 sitemap.add(
521 "{}/{}/{}".format(base_url, v["slug"], v["id"]),
522 # Timestamp of that video
523 lastmod=v["publishedAt"][:10],
524 changefreq="yearly",
525 priority="0.5",
526 )
528 # Aprender index
529 sitemap.add(f"{base_url}/aprender/", changefreq="monthly", priority="0.9")
531 # Aprender: individual tutorials
532 with open("../website/src/aprender/tutoriales.js") as in_file:
533 # Remove the JS bits to keep only the JSON content
534 tutoriales_json_content = in_file.read()[17:-2]
535 tutoriales = json.loads(tutoriales_json_content)
536 for t in tutoriales:
537 tuto_url = "{}/aprender/{}".format(base_url, t["slug"])
538 sitemap.add(tuto_url, changefreq="yearly", priority="0.7")
540 sitemap_xml = sitemap.generate()
542 # Prettify the XML "by hand"
543 sitemap_xml = sitemap_xml.replace("<url>", " <url>")
544 sitemap_xml = sitemap_xml.replace("</url>", " </url>")
545 sitemap_xml = sitemap_xml.replace("<loc>", " <loc>")
546 sitemap_xml = sitemap_xml.replace("<lastmod>", " <lastmod>")
547 sitemap_xml = sitemap_xml.replace("<changefreq>", " <changefreq>")
548 sitemap_xml = sitemap_xml.replace("<priority>", " <priority>")
550 with open(sitemap_file, "w") as file:
551 file.write(sitemap_xml)
554def website(args):
555 if getattr(args, "no_fetch", False):
556 # Rebuild from the already-committed data instead of hitting the
557 # YouTube API - locations/videos already have their slugs set, so
558 # there's nothing left to do but regenerate the website files.
559 locations = load_locations_from_website_data_file(WEBSITE_DATA_FILE)
560 uploaded_videos = flatten_videos_from_locations(locations)
561 logger.info(
562 f"There are {len(uploaded_videos)} videos loaded from {WEBSITE_DATA_FILE} (--no-fetch)."
563 )
564 else:
565 # Retrieve the list of uploaded videos
566 uploaded_videos = get_uploaded_videos(args, UPLOADED_VIDEOS_DUMP_FILE)
567 logger.info(f"There are {len(uploaded_videos)} uploaded videos.")
569 # Identify each video's location
570 (uploaded_videos, locations) = identify_locations_names(
571 uploaded_videos, LOCATION_SPECIAL_CASES_FILE, UPLOADED_VIDEOS_DUMP_FILE
572 )
574 # Determine the geolocation of each location
575 locations = determine_geolocation(locations, GEOLOCATIONS_FILE)
577 # Create a slug for each location (to be used for the website URLs)
578 locations = determine_locations_slug(locations)
580 # Add the videos in each location array
581 locations = add_videos_to_locations_array(uploaded_videos, locations)
583 # Generate the JavaScript data file to be used by the website
584 save_website_data(locations, WEBSITE_DATA_FILE)
586 # Generate the development and production website files
587 generate_website(locations, uploaded_videos)
589 # Generate the Sitemap
590 generate_sitemap(SITEMAP_FILE, locations, uploaded_videos)