1) 파일 확장자
2) 확장자에 따른 파일 불러오는 함수
pandas 라이브러리의 read_확장자() 함수를 사용import pandas as pd
df = pd.read_csv('file.csv')
import pandas as pd
df = pd.read_json('file.json')
import pandas as pd
df = pd.read_csv('file.txt', delimiter='\t') # 만약 탭으로 구분되어 있다면 delimiter='\t'를 사용합니다.
colab 환경에서는 내 컴퓨터에서 사용하는 것이 아니기 때문에 구글 드라이브에 파일을 업로드 후, 파일을 마운트하고 불러와야 함from google.colab import drive
drive.mount('/content/drive')
root = "/content/drive/MyDrive/파일이나 폴더"
import pandas as pd
df = pd.read_excel(root) # root나 기타 다른 변수로 파일 경로를 만들어서 넣어주면 됨
3) 파일 저장하기
import pandas as pd
data = {
'Name': ['John', 'Emily', 'Michael'],
'Age': [30, 25, 35],
'City': ['New York', 'Los Angeles', 'Chicago']
}
df = pd.DataFrame(data)
excel_file_path = '/content/sample_data/data.csv'
df.to_csv(excel_file_path, index = False)
# index란 맨 왼쪽에 있는 행 번호들인데 보통은 넣는 경우가 거의 없어서 False
print("csv 파일이 생성되었습니다.")
import json
data = {
'Name': ['John', 'Emily', 'Michael'],
'Age': [30, 25, 35],
'City': ['New York', 'Los Angeles', 'Chicago']
}
json_file_path = '/content/sample_data/data.json'
# json 파일을 쓰기모드('w')로 열어서 data를 거기에 덮어씌우게 됩니다.
# with를 사용하면 파일을 열고 자동으로 닫아줌
with open(json_file_path, 'w') as jsonfile:
json.dump(data, jsonfile, indent=4)
# .dump는 덮어씌우는 메소드
print("JSON 파일이 생성되었습니다.")
1) 패키지란?
numpy와 matplotlib# 아래와 같이 보통은 필요한 패키지를 한번에 다 불러온 다음 코딩을 진행합니다
import pandas as pd
import numpy as np
import tensorflow as tf
import matplotlib.pyplot as plt
import seaborn
2) 다양한 종류의 패키지
pandas
데이터 조작과 분석을 위한 라이브러리로, 데이터를 효과적으로 조작하고 분석할 때
numpy
과학적 계산을 위한 핵심 라이브러리로, 다차원 배열과 행렬 연산을 지원
matplotlib
데이터 시각화를 위한 라이브러리로, 다양한 그래프와 플롯을 생성
import matplotlib.pyplot as plt
plt.plot([1, 2, 3, 4], [1, 4, 9, 16])
plt.xlabel('X-axis')
plt.ylabel('Y-axis')
plt.show()
seabornimport seaborn as sns
import pandas as pd
data_sample = pd.DataFrame({'x':[1, 2, 3, 4], 'y':[1, 4, 9, 16]})
sns.barplot(data=data_sample, x='x', y='y')
scikit-learn
from sklearn.datasets import load_iris # 내가 가져올 함수를 작성
from sklearn.linear_model import LinearRegression
# Iris 데이터셋 불러오기
iris = load_iris()
# Iris 데이터셋에서 특정 범위의 데이터 슬라이싱하기
X_train = iris.data[:,:-1] # 데이터 값들 추출
print("학습 데이터:", X_train)
y_train = iris.data[:,-1:] # 정답값 추출
print("학습 데이터:", y_train)
model = LinearRegression()
model.fit(X_train, y_train)
statsmodelsimport statsmodels.api as sm
model = sm.OLS(y_train, X_train)
result = model.fit()
print(result.summary())
scipyimport numpy as np
from scipy.integrate import quad
# 적분할 함수 정의
def integrand(x):
return np.exp(-x ** 2)
# 정적분 구간
a = 0
b = np.inf
# 적분 계산
result, error = quad(integrand, a, b)
print("결과:", result)
print("오차:", error)
tensorflow ⭐import tensorflow as tf
input_size = 3
model = tf.keras.Sequential([
tf.keras.layers.Dense(10, activation='relu', input_shape=(input_size,)),
tf.keras.layers.Dense(1)
])
model.compile(optimizer='adam', loss='mse')
pytorch1) 포맷팅이란?
f-string (가장 직관적이고 편함, python 3.6 이상)x = 10
print(f"변수 x의 값은 {x}입니다.")
.format 메소드 활용 (예전 방식)x = 10
print("변수 x의 값은 {}입니다.".format(x))
x = 10
print("변수 x의 값은 %d입니다." % (x))
1) 리스트 컴프리헨션이란?
True인 경우에만 리스트에 추가# 기본적인 구조
[표현식 for 항목 in iterable if 조건문]
2) 리스트 컴프리헨션 예시
# 예시: 1부터 10까지의 숫자를 제곱한 리스트 생성
squares = [x**2 for x in range(1, 11)]
print(squares) # 출력: [1, 4, 9, 16, 25, 36, 49, 64, 81, 100]
# 예시: 리스트에서 짝수만 선택하여 제곱한 리스트 생성
even_squares = [x**2 for x in range(1, 11) if x % 2 == 0]
print(even_squares) # 출력: [4, 16, 36, 64, 100]
# 예시: 문자열 리스트에서 각 문자열의 길이를 저장한 리스트 생성
words = ["apple", "banana", "grape", "orange"]
word_lengths = [len(word) for word in words]
print(word_lengths) # 출력: [5, 6, 5, 6]
# 예시: 리스트 컴프리헨션을 중첩하여 2차원 리스트 생성
matrix = [[i for i in range(1, 4)] for j in range(3)]
print(matrix) # 출력: [[1, 2, 3], [1, 2, 3], [1, 2, 3]]
1) lambda(람다)란?
2) lambda와 함수의 차이점
def, 람다 함수는 lambda로 정의3) lambda를 쓰는 이유
4) lambda 함수의 예시
add = lambda x, y: x + y
print(add(3, 5)) # 출력: 8
square = lambda x: x ** 2
print(square(4)) # 출력: 16
filter 함수 사용)filter(조건 함수, 반복 가능한 데이터)
numbers = [1, 2, 3, 4, 5, 6, 7, 8, 9, 10]
even_numbers = list(filter(lambda x: x % 2 == 0, numbers))
print(even_numbers) # 출력: [2, 4, 6, 8, 10]
map 함수 사용)map(함수, 반복 가능한 데이터)
numbers = [1, 2, 3, 4, 5]
squared_numbers = list(map(lambda x: x ** 2, numbers))
print(squared_numbers) # 출력: [1, 4, 9, 16, 25]
1) glob란?
import glob
# 현재 경로의 모든 파일을 찾기
file_list1 = glob.glob('*')
# 단일 파일 패턴으로 파일을 찾기
file_list2 = glob.glob('drive')
# 디렉토리 안의 모든 파일 찾기
file_list3 = glob.glob('sample_data/*')
# 특정 확장자를 가진 파일만 찾기
file_list4 = glob.glob('sample_data/*.csv')
1) os란?
os 모듈은 파일 시스템을 관리하고, 디렉토리를 탐색하고, 파일을 조작하는 운영 체제와 상호 작용하기 위한 함수를 제공
주요 기능
2) os 사용 예시
# 현재 작업 디렉토리 가져오기
import os
cwd = os.getcwd()
print(cwd)
# 디렉토리 생성
import os
os.mkdir('sample_data/new_directory')
# 파일 이름 변경
import os
os.rename('sample_data/new_directory', 'sample_data/new_directory2')
# 파일 삭제
import os
os.remove(file_adress)
import os
os.remove('sample_data/data.csv')
# 파일 목록(경로) 가져오기
import os
files = os.listdir('/content')
print(files)
# 경로 조작
import os
path = os.path.join('/content', 'sample_data', 'mnist_test.csv')
print(path)
1) split이란?
문자열이나 경로를 쪼개기 위해 사용하며, 파일 경로로부터 파일 제목을 추출하는 등의 상황에서 사용
문자열을 공백 기준으로 분할하여 리스트로 변환하기
sentence = "Hello, how are you doing today?"
words = sentence.split()
print(words) # 출력: ['Hello,', 'how', 'are', 'you', 'doing', 'today?']
data = "apple,banana,grape,orange"
fruits = data.split(',')
print(fruits) # 출력: ['apple', 'banana', 'grape', 'orange']
words = ['Hello,', 'how', 'are', 'you', 'doing', 'today?']
sentence = ' '.join(words)
print(sentence) # 출력: Hello, how are you doing today?
fruits = ['apple', 'banana', 'grape', 'orange']
data = ','.join(fruits)
print(data) # 출력: apple,banana,grape,orange
text = """First line
Second line
Third line"""
lines = text.split('\n')
print(lines) # 출력: ['First line', 'Second line', 'Third line']
sentence = "Hello, how are you doing today?"
words = sentence.split()
first_three_words = words[:3]
print(first_three_words) # 출력: ['Hello,', 'how', 'are']
text = " Hello how are you "
cleaned_text = text.strip()
words = cleaned_text.split()
print(words) # 출력: ['Hello', 'how', 'are', 'you']
file_path라는 문자열 변수에 데이터의 경로를 저장하고, split() 함수를 사용하여 문자열을 / 기준으로 분할rsplit() 함수를 사용하여 오른쪽에서부터 최대 1회만 분할하도록 설정하여 파일명과 디렉토리로 나눈다.directory와 filename 변수에 할당하여 출력1) 클래스란?
# 아래의 초기값은 반드시 맨 처음에 작성되어야 한다
class ClassName:
def __init__(self, parameter1, parameter2):
self.attribute1 = parameter1
self.attribute2 = parameter2
# self라는 클래스 자기 자신을 나타내는 매개변수는 반드시 작성
def method1(self, parameter1, parameter2):
# 메서드 내용 작성
pass
__init__ : 클래스 생성자 메소드, 객체가 생성될 때 호출하여 초기화 작업을 수행self는 무조건 반드시 사용해야한다. 이것은 해당 메소드가 속한 객체를 가리킨다.class Person:
def __init__(self, name, age):
self.name = name
self.age = age
# 객체 생성
person1 = Person("Alice", 30)
person2 = Person("Bob", 25)
class Animal:
def sound(self):
print("Some generic sound")
class Dog(Animal):
def sound(self):
print("Woof")
class Cat(Animal):
def sound(self):
print("Meow")
# 다형성 활용
animals = [Dog(), Cat()]
for animal in animals:
animal.sound()
2) 클래스와 함수의 차이점
공통점 : 클래스와 함수는 모두 파이썬에서 코드를 조직화하고 재사용성을 높이는 데 사용
함수(Function)
3) 클래스의 속성과 메서드
클래스(Class)는 객체(Object)를 생성하기 위한 템플릿이며, 메서드(Method)와 속성(Attribute)을 가진다.
메서드와 속성은 클래스의 행동과 상태를 정의하는 데 사용
메서드(Method)
self 매개변수를 첫 번째 매개변수로 사용하여 메서드가 속한 인스턴스를 참조class Car:
def __init__(self, brand):
self.brand = brand
def start_engine(self):
print(f"{self.brand}의 엔진을 가동합니다.")
# Car 클래스의 인스턴스 생성
my_car = Car("Toyota")
# start_engine() 메서드 호출
my_car.start_engine() # 출력: Toyota의 엔진을 가동합니다.
class Dog:
def __init__(self, name):
self.name = name # 인스턴스 속성
# Dog 클래스의 인스턴스 생성
my_dog = Dog("Buddy")
print(my_dog.name) # 출력: Buddy
4) 클래스가 데이터 분석에서 사용되는 예시
class Stock:
def __init__(self, symbol, price, volume):
self.symbol = symbol
self.price = price
self.volume = volume
# 객체 생성
stock1 = Stock("AAPL", 150.25, 100000)
stock2 = Stock("GOOG", 2800.75, 50000)
raw_data = [1,2,4,5,6,87,2,253654]
class DataPreprocessor:
def __init__(self, data):
self.data = data
def normalize_data(self):
# 데이터 정규화 메소드 작업 수행
pass
def handle_missing_values(self):
# 결측치 처리 메소드 작업 수행
pass
def remove_outliers(self):
# 이상치 제거 메소드 작업 수행
pass
# 데이터 전처리 객체 생성
preprocessor = DataPreprocessor(raw_data)
preprocessor.normalize_data()
preprocessor.handle_missing_values()
preprocessor.remove_outliers()
class LinearRegressionModel:
def __init__(self, data):
self.data = data
def train(self):
# 선형 회귀 모델 학습
pass
def predict(self, new_data):
# 새로운 데이터에 대한 예측 수행
pass
def evaluate(self):
# 모델 평가 수행
pass
# 선형 회귀 모델 객체 생성
lr_model = LinearRegressionModel(training_data)
lr_model.train()
predictions = lr_model.predict(new_data)
evaluation_result = lr_model.evaluate()
1) 불리언 인덱싱이란?
import numpy as np
# 배열 생성
arr = np.array([1, 2, 3, 4, 5])
# 불리언 배열 생성 (조건에 따라 True 또는 False 값을 갖는 배열)
condition = np.array([True, False, True, False, True])
# 불리언 인덱싱을 사용하여 조건에 맞는 요소 선택
result = arr[condition]
# 결과 출력
print("Result using boolean indexing:", result) # 출력: [1 3 5]
# 불리언 인덱싱을 사용하여 배열에서 짝수인 요소만 선택
evens = arr[arr % 2 == 0]
# 결과 출력
print("Even numbers using boolean indexing:", evens) # 출력: [2 4]
arr과 조건을 담은 불리언 배열 condition을 생성해준다.1) 데코레이션이란?
데코레이터(Decorator)는 파이썬에서 함수나 메서드의 기능을 확장하거나 수정하는 강력한 도구로 함수나 메서드를 인자로 받아 해당 함수나 메서드를 변경하거나 래핑하는 함수
즉, 기존의 함수를 따로 수정하지 않고도 추가 기능을 넣고 싶을 때 사용
데코레이션은 따로 함수 내부의 구조를 바꾸지 않고 함수 외부에 간단한 명령어를 작성하여 작동이 된다.
사용 예시 1
def my_decorator(func):
def wrapper():
print("Something is happening before the function is called.")
func()
print("Something is happening after the function is called.")
return wrapper
@my_decorator
def say_hello():
print("Hello!")
say_hello()
# 출력 :
#'Something is happening before the function is called.'
#'Hello!'
#'Something is happening after the function is called.'
import tensorflow as tf
# 계산량이 적은 경우 - 더 빠름
@tf.function # The decorator converts `add` into a `Function`.
def add(a, b):
return a + b
add(tf.ones([2, 2]), tf.ones([2, 2])) # [[2., 2.], [2., 2.]]
# 계산량이 큰 경우 - 큰 차이 없음
import timeit
conv_layer = tf.keras.layers.Conv2D(100, 3)
@tf.function
def conv_fn(image):
return conv_layer(image)
image = tf.zeros([1, 200, 200, 100])
print("Eager conv:", timeit.timeit(lambda: conv_layer(image), number=10))
print("Function conv:", timeit.timeit(lambda: conv_fn(image), number=10))
print("Note how there's not much difference in performance for convolutions")
참고
- 즉시 실행 모드 (Eager Execution)
- 즉시 실행 모드는 파이썬 코드를 순차적으로 실행하면서 연산을 즉시 평가하는 방식입니다.
- 각각의 연산은 실행될 때마다 결과가 즉시 반환되어 사용자가 바로 확인할 수 있습니다.
- 파이썬의 일반적인 제어 흐름과 함께 사용되며, 디버깅 및 코드 작성이 용이합니다.
- TensorFlow 2.0부터는 즉시 실행 모드가 기본적으로 활성화되어 있습니다.
- 그래프 모드 (Graph Mode)
- 그래프 모드는 파이썬 코드를 그래프로 변환하고, 이를 최적화한 후에 실행하는 방식입니다.
- 먼저 그래프를 정의하고, 그래프를 실행하기 위해 세션을 통해 입력을 제공해야 합니다.
- 그래프는 연산의 순서와 의존성을 명확하게 표현하므로, 병렬 실행과 하드웨어 가속화에 최적화되어 있습니다.
- TensorFlow 1.x에서는 주로 그래프 모드를 사용했으나, TensorFlow 2.0부터는 즉시 실행 모드가 기본으로 제공되어 그래프 모드를 명시적으로 사용하지 않아도 됩니다.
- 차이점
- 즉시 실행 모드는 파이썬 코드를 사용하여 연산을 즉시 평가하고 결과를 반환합니다. 반면에, 그래프 모드는 그래프를 먼저 정의하고 세션을 통해 실행해야 합니다.
- 즉시 실행 모드는 디버깅과 코드 작성이 용이하지만, 그래프 모드는 병렬 실행과 하드웨어 가속화에 최적화되어 있습니다.
- 즉시 실행 모드는 각각의 연산을 바로 평가하기 때문에 코드를 작성하고 실행하는 데에 편리합니다. 반면에, 그래프 모드는 전체 그래프를 먼저 정의하고 실행해야 하므로 초기 설정이 더 복잡할 수 있습니다.
- 그래프 모드는 동일한 그래프를 여러번 실행할 때 재사용할 수 있어서 효율적인 반복 작업에 유용합니다.
1) 파이썬 에러 확인하는 법
Traceback (most recent call last):
File "codeit.py", line 12, in <module> # 어떤 파일에서 몇번째 줄에 에러가 떴는지를 알려줌
numbers[right] = temp[left] # 어떤 코드가 잘못 되었는지를 알려줌
TypeError: 'int' object is not subscriptable # 에러의 종류 : 에러에 대한 세부 정보
2) 대표적인 에러들
1. SyntaxError (구문 오류)
2. IndentationError (들여쓰기 오류)
3. NameError (이름 오류)
4. TypeError (타입 오류)
5. IndexError (인덱스 오류)
6. KeyError (키 오류)
7. FileNotFoundError (파일을 찾을 수 없음 오류)
3) 위의 예시에 없는 에러의 경우