Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
74 changes: 74 additions & 0 deletions .gitignore
Original file line number Diff line number Diff line change
@@ -0,0 +1,74 @@

# Created by https://www.gitignore.io/api/windows,notepadpp,microsoftoffice,jupyternotebooks
# Edit at https://www.gitignore.io/?templates=windows,notepadpp,microsoftoffice,jupyternotebooks

### JupyterNotebooks ###
# gitignore template for Jupyter Notebooks
# website: http://jupyter.org/

.ipynb_checkpoints
*/.ipynb_checkpoints/*

# IPython
profile_default/
ipython_config.py

# Remove previous ipynb_checkpoints
# git rm -r .ipynb_checkpoints/

### MicrosoftOffice ###
*.tmp

# Word temporary
~$*.doc*

# Word Auto Backup File
Backup of *.doc*

# Excel temporary
~$*.xls*

# Excel Backup File
*.xlk

# PowerPoint temporary
~$*.ppt*

# Visio autosave temporary files
*.~vsd*

### NotepadPP ###
# Notepad++ backups #
*.bak

### Windows ###
# Windows thumbnail cache files
Thumbs.db
Thumbs.db:encryptable
ehthumbs.db
ehthumbs_vista.db

# Dump file
*.stackdump

# Folder config file
[Dd]esktop.ini

# Recycle Bin used on file shares
$RECYCLE.BIN/

# Windows Installer files
*.cab
*.msi
*.msix
*.msm
*.msp
brainstorm project 2
*.csv
*.ipynb


# Windows shortcuts
*.lnk

# End of https://www.gitignore.io/api/windows,notepadpp,microsoftoffice,jupyternotebooks
155 changes: 155 additions & 0 deletions Top_50_Barcelona_names/Barcelona_Top_50_names.sql
Original file line number Diff line number Diff line change
@@ -0,0 +1,155 @@
CREATE DATABASE Population;

#DESCRIPTIVE QUESTIONS!
#question longest(F/M)
select * from most_frequent_names
where Gender = 'Female'
group by name
order by max(length(`name`)) desc
limit 10;
#Maria Teresa, Maria Carmen, Encarnación. All have 11 letters

#Question longest (M)
select * from most_frequent_names
where Gender = 'Male'
group by name
order by max(length(`name`)) desc
limit 10;
#Francisco Javier is the clear winner in the male seccion

#question shortest(F)
select * from most_frequent_names
where Gender = 'Female'
group by name
order by min(length(`name`)) asc
limit 10;
#here we have a another technical draw between Noa, Ona, Lia, Mia, Mar, Eva and Ana all with 3 lettersw
#question shortest (m)
select * from most_frequent_names
where Gender = 'Male'
group by name
order by min(length(`name`)) asc
limit 10;
#here we have a technical draw once more between Max, Nil, Teo, Roc, Pau, Jan, Pol, Ian and Leo

#Average length of the names group by genre
SELECT gender, ROUND(AVG(length(name)),2) as Average_length
FROM most_frequent_names
GROUP BY 1;
#We can see Female names are slightly longer than males names.

#Are our names in the table? In which decade do they appear first?
SELECT Name, Decade as First_appearance_decade
FROM most_frequent_names
WHERE name = 'Francisco Javier' or 'Andreu'
GROUP BY 1
ORDER BY 2
LIMIT 1;
#Francisco Javier is on the table and appeared in the top 50 for the first time in 1950.
#Andreu has never been on the top 50 male names

#questiononetimeglow(F)
select name, decade, sum(frequency)
from most_frequent_names
where gender = 'Female'
group by name having count(decade) = 1
order by sum(frequency) desc
limit 10;
#here we have many names that have only been on the to 50 for only one decade, but we choses the 10 with more frequency which are:
#Jessica(1980), Noemi(1970), Eva Maria (1970), Valentina (2010), Lourdes(1960), Olivia(2010), Lidia(1970), Ruth(1970), MªJosefa (1940) and Nora(2010)

#questiononetimeglow(M)
select name, decade, sum(frequency)
from most_frequent_names
where gender = 'Male'
group by name having count(decade) = 1
order by sum(frequency) desc
Limit 10;
#here we have many names that have only been on the to 50 for only one decade, but we choses the 10 with more frequency which are:
#Leo(2010), Roberto(1970), Juan Jose(1960), Mateu (2010), Geal(2010), Teo(2010), Roc(2010), Adam(2010), Luca(2010),Ian(2010)

#Does appear any name in the top 50 in both genders? Which one?
SELECT name
FROM most_frequent_names
WHERE gender = 'Male' AND name in (SELECT name FROM most_frequent_names where gender = 'Female');
#No name has ever been on the top 50 names on both genders (no matter in which decade it appeared)

#Are there any names in the first decade and in the last? which ones?
SELECT name, gender
from most_frequent_names
where name in (SELECT name FROM most_frequent_names where decade = 'Before 1930') and decade = 2010
GROUP BY NAME;
#There are exactly 5 names for each gender that appear in the top 50 before 1930 and in 2010
#The female names are Maria, Nuria, Ana, Julia, Elena
#The males names are Joan, Jordi, Tomas, Daniel, Alejandro



#ANALYTICAL QUESTIONS!

#How does the lenght of the names change through the decades? (F)
SELECT gender, decade, round(AVG(length(name)),2) as Length
FROM most_frequent_names
where gender = 'Female' and decade not like 'Total'
GROUP BY 1, 2;
#It can be appreciated that during the first half of the century the tendecy of name's length was to increase, while after 1950 each decade it decreases

#How does the lenght of the names change through the decades? (M)
SELECT gender, decade, round(AVG(length(name)),2) as Length
FROM most_frequent_names
where gender = 'Male' and decade not like 'Total'
GROUP BY 1, 2;
#Here we can see the same pattern as female's names does, but it peaks 2 decades later.

#Question All time Favourites (the premises are that if any name has been on the top 50 all decades is the winner, but if there are more,
#we decide the winner by frequency (M)
select name, gender, sum(frequency) as Frequency
from most_frequent_names
where decade not like 'Total' and gender = 'Male'
group by name having count(decade) = 10
order by gender, sum(frequency) desc;
#Only two male names have been on the top all century! the winner is 'Jordi', followed by the runner-up 'Joan'
#as we also want to know the all time favourites for female names, now we are checking if there are any that have been up there al least 9 decades of the century

#Question All time Favourites (the premises are that if any name has been on the top 50 all decades is the winner, but if there are more,
#we decide the winner by frequency (F)
select name, sum(frequency) as Frequency
from most_frequent_names
where decade not like 'Total' and gender = 'Female'
group by name having count(decade) = 10
order by sum(frequency) desc;
#here we do find two female names! We have 'Maria' as a winner, followed by the runners-up 'Nuria', 'Ana' and 'Elena'

#Question modern names (only appear on the top the last two decades)
select name, gender, sum(frequency)
from most_frequent_names
where decade >= 2000 and name not in (select name
from most_frequent_names
where decade < 2000)
group by name having count(decade) = 2
order by sum(frequency) desc , gender asc
limit 10;
#Top 3 modern names for females are Martina, Emma and Noa
#top 3 modern names for Males are Jan, Biel and Hugo
#we can also see that out of the 10 most frequent modern names 7 out of 10 are male names
#so we could say that male names have modernized more than female names the past two decades

#Top names of the century (We decided to assing 50 points for the top name of each decade and 1 to the last name, descending accordingly.
#Afterewards, we multipled the result by the frequency of each name. In order to make the number more readable, we divided the result by 1M)
select name, round((sum(51 - mfn.order) * sum(Frequency)) / 1000000,2) as Score
from most_frequent_names mfn
where decade not like 'Total'
GROUP BY Name
Order BY Score desc
LIMIT 3;
#The clear winner in Barcelona is maria, with an score of 8.65, followed by Antonio (5.60) and the bronze goes to Jordi (5.21)

#In which gender names tends society to innovate more? (We sum the frequency of the names by gender and then divide it by the count the unique names of each gender, the value which is nearest to 1 is the most innovative)
Select gender, count(DISTINCT(name)) as Total_Unique_Names, sum(Frequency) as Frequency_of_the_names, sum(Frequency) / count(DISTINCT(name)) as Score
From most_frequent_names
where decade not like 'Total'
Group by 1;
#Female names have are more innovative than male names, eventhough their frequency ir higher the number of unique names is also higher, making it closer to 1 than the males names.



69 changes: 69 additions & 0 deletions Top_50_Barcelona_names/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,69 @@
<img src="https://bit.ly/2VnXWr2" alt="Ironhack Logo" width="100"/>

# Top 50 names in Barcelona
*Andreu Carreño and Javier Hita*

*[Data Analytics, Barcelona, March 2020]*

## Content
- [Project Description](#project-description)
- [Questions & Hypotheses](#questions-hypotheses)
- [Dataset](#dataset)
- [Workflow](#workflow)
- [Organization](#organization)
- [Links](#links)


## Project Description
This project is an analisis of a dataset about the most frequent names in Barcelona throughout the last century.
The analisis has allowed us to answer some questions that we wouldn't have been able to answer a week ago by using what we have learned in this week's lessons.
In this project we learned how to work as a team and to ask the most insightful questions about a dataset.
We also learnd how to use MySQL to extract answers from data and be able to get interesting conclusions.
we also improved our skills on how to present a project in team.

## Questions & Hypotheses

###Questions
What are the questions you would like to answer with your analysis? What did you feel were the answers to those questions before answering them with data?
Which are the longest names in the top 50 in Barcelona?
Which are the shortest ones?
Which names are longer, male or female?
Are Andreu and Francisco Javier (our names) in the Top 50? If so, in which decade did they appear?
Which names have only appeared in just one decade? Why?
Does any name appear in the Top 50 for both genders? Which one?
Are there any names in the first decade and in the last?

###Hypotheses
How does the length of the names change through the decades? (Is it true that names have shroten throughout the decades?)
Which are the all time favourite's names? (Is it true that some names have been present in all decades?)
Which are the modern names? (Is it true that there are some names that only appear on the last 2 decades?)
Which are the top names of the century? (Is there a way to know which have been the best names of the century?
In which gender society tends to innovate more? (Is it true that society innovates more with male names?

## Dataset
We used the dataset about the most frequent names in Barcelona. This data set is from Barcelona's council house website.
This data set contain information regarding the names during each decade and it's frecuency sorted by gender.
[Open data Bcn](https://opendata-ajuntament.barcelona.cat/en/)

## Workflow
First we decided which dataset were we going to analyse. Once the choice had been made the first task we did was to brainstorm about which questions would be interesting to answer.
We divided the questions into 2 types. Descriptive and analytical, we defined under which premises we were going to approach the analytical questions.
We ran all the necessary queries to answer the questions, and we proceeded to analyse the results.

## Organization
Once all the tasks were defined, we created a Trello board to decided who would do each task and how.
First we defined the question together, then we assigned each one of us half of them and we started coding.
Once we had the results, we analysed them together.

Our repository is compossed by 4 main files
- README.md
- Dataset
- Top_50_Barcelona_names
- .gitignore

## Links
Include links to your repository, slides and kanban board. Feel free to include any other links associated with your project.

[Repository](https://github.com/Javierhvb/Project-50_Top_names)
[Slides](https://docs.google.com/presentation/d/15uxZwiwrCGTDJR8E2kbv0jEvrwJpJdZNTIdukp7ehgE/edit#slide=id.p)
[Trello](https://trello.com/b/18xAHAGL/project-week-2)
43 changes: 0 additions & 43 deletions your-project/README.md

This file was deleted.